From 1a255ed9d51b4519ebf999aefa5b170a3fb17659 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Thu, 10 Sep 2026 21:44:57 +0800 Subject: [PATCH 01/17] fix(sql): bind minted identity by kind instead of always the decimal surrogate A keyless schemaless row and a timeseries row were self-bound under the decimal surrogate string, the same convention auto-_rowid rows use. That decimal string is not the document storage key those rows are actually stored under, so RETURNING id, SELECT id, and later lookups by that id disagreed with each other. FreshSurrogateKind now names the binding convention a fresh row takes (AutoRowId vs DocumentStorageKey), and assign_fresh returns the bound identity string alongside the surrogate so every caller uses the same value the allocator minted instead of re-deriving it. fresh_identity_string is the single formatter both the allocator and the SQL-plan converter's placeholder paths call. The response envelope also forwards its id onto a scan-wrapper body that has none, so a keyless row's identity survives projection. --- nodedb-physical/src/lib.rs | 2 +- nodedb-physical/src/surrogate.rs | 32 +++- .../planner/sql_plan_convert/convert.rs | 41 ++++- .../planner/sql_plan_convert/dml/insert.rs | 34 +++- .../sql_plan_convert/scan/timeseries.rs | 15 +- .../control/server/response_shape/project.rs | 8 +- .../surrogate/assign/core/assign_ops.rs | 49 ++++-- .../src/control/surrogate/assign/core/mod.rs | 1 + nodedb/src/control/surrogate/assign/mod.rs | 1 + nodedb/src/control/surrogate/mod.rs | 1 + nodedb/src/control/surrogate/physical_impl.rs | 9 +- .../control/target_identity/document_id.rs | 18 +- .../src/control/target_identity/surrogate.rs | 25 ++- .../sql_transactions_pk_scan_consistency.rs | 163 ++++++++++++++++++ 14 files changed, 337 insertions(+), 62 deletions(-) diff --git a/nodedb-physical/src/lib.rs b/nodedb-physical/src/lib.rs index 9b7489305..d80c6046a 100644 --- a/nodedb-physical/src/lib.rs +++ b/nodedb-physical/src/lib.rs @@ -18,5 +18,5 @@ pub mod visitor; pub use convert_context::SharedConvertContext; pub use error::ConvertError; -pub use surrogate::{SurrogateAssignError, SurrogateAssigner}; +pub use surrogate::{FreshSurrogateKind, SurrogateAssignError, SurrogateAssigner}; pub use visitor::{PhysicalTaskVisitor, dispatch}; diff --git a/nodedb-physical/src/surrogate.rs b/nodedb-physical/src/surrogate.rs index 6dab200b9..d0b6a7624 100644 --- a/nodedb-physical/src/surrogate.rs +++ b/nodedb-physical/src/surrogate.rs @@ -24,6 +24,20 @@ pub enum SurrogateAssignError { Backend(String), } +/// The identity convention a freshly minted row's surrogate binds under. +/// +/// This enum names the convention. It never formats one. +/// `nodedb`'s allocator turns the variant into the identity string. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum FreshSurrogateKind { + /// The row's identity is its document storage key. Covers a schemaless + /// row with no declared `PRIMARY KEY`, and a timeseries row. + DocumentStorageKey, + /// The row's identity is the strict-schema auto `_rowid` column value. + /// The Data Plane writes that column as an `Int64`. + AutoRowId, +} + /// Allocate stable, cross-engine surrogates for `(collection, pk_bytes)`. /// /// Implementations must be: @@ -51,18 +65,20 @@ pub trait SurrogateAssigner: Send + Sync { /// Allocate a FRESH, never-before-issued surrogate for a row that has no /// content primary key — i.e. a collection whose primary key is the - /// auto-generated `_rowid` (no `PRIMARY KEY` was declared at CREATE). Each - /// call returns a new value; there is no `pk_bytes` to content-address on, - /// so repeated calls do NOT collapse to the same surrogate (which is - /// exactly the bug that content-addressing an empty key would cause). + /// auto-generated `_rowid` (no `PRIMARY KEY` was declared at CREATE), or + /// a timeseries row. /// - /// The Data Plane sets the row's `_rowid` equal to this surrogate, so - /// implementations should bind the surrogate to its own value for reverse - /// `_rowid = N` point lookups. + /// Every call allocates a new value. There is no `pk_bytes` to + /// content-address on, so repeated calls never collapse onto one + /// surrogate. + /// + /// `kind` picks the convention the binding uses. The returned `String` is + /// the identity. Callers use it verbatim and never re-derive it. fn assign_fresh( &self, database_id: DatabaseId, tenant_id: TenantId, collection: &str, - ) -> Result; + kind: FreshSurrogateKind, + ) -> Result<(Surrogate, String), SurrogateAssignError>; } diff --git a/nodedb/src/control/planner/sql_plan_convert/convert.rs b/nodedb/src/control/planner/sql_plan_convert/convert.rs index 7887e2056..087f06bf1 100644 --- a/nodedb/src/control/planner/sql_plan_convert/convert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/convert.rs @@ -193,15 +193,35 @@ impl ConvertContext { .unwrap_or(nodedb_types::Surrogate::ZERO)) } - /// Allocate a new surrogate only while producing executable work. - /// Metadata plans use a zero placeholder because no fresh identity exists. - pub fn fresh_surrogate(&self, collection: &str) -> crate::Result { + /// Allocate a new surrogate and its bound identity string, only while + /// producing executable work. + /// + /// A metadata plan, or a plan with no wired assigner, returns a + /// `Surrogate::ZERO` placeholder. Its identity string comes from + /// [`fresh_identity_string`](crate::control::surrogate::fresh_identity_string), + /// the same function the allocator calls. + pub fn fresh_surrogate( + &self, + collection: &str, + kind: nodedb_physical::FreshSurrogateKind, + ) -> crate::Result<(nodedb_types::Surrogate, String)> { + let placeholder = || { + Ok(( + nodedb_types::Surrogate::ZERO, + crate::control::surrogate::fresh_identity_string( + kind, + nodedb_types::Surrogate::ZERO, + ), + )) + }; if self.is_metadata() { - return Ok(nodedb_types::Surrogate::ZERO); + return placeholder(); } match self.surrogate_assigner.as_ref() { - Some(assigner) => assigner.assign_fresh(self.database_id, self.tenant_id, collection), - None => Ok(nodedb_types::Surrogate::ZERO), + Some(assigner) => { + assigner.assign_fresh(self.database_id, self.tenant_id, collection, kind) + } + None => placeholder(), } } @@ -340,7 +360,14 @@ mod tests { .as_u32(), 0 ); - assert_eq!(metadata.fresh_surrogate("users").unwrap().as_u32(), 0); + assert_eq!( + metadata + .fresh_surrogate("users", nodedb_physical::FreshSurrogateKind::AutoRowId) + .unwrap() + .0 + .as_u32(), + 0 + ); assert_eq!( assigner .lookup(DatabaseId::DEFAULT, TenantId::new(1), "users", b"new-user") diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert.rs index 8bd48d4fa..f720e6485 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/insert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert.rs @@ -157,8 +157,12 @@ pub(super) fn resolve_doc_identity( } if is_auto_rowid_pk(primary_key) { - let s = assign_fresh(ctx, collection)?; - return Ok((s.as_u32().to_string(), s)); + let (s, pk) = assign_fresh( + ctx, + collection, + nodedb_physical::FreshSurrogateKind::AutoRowId, + )?; + return Ok((pk, s)); } match extract_doc_id(row, primary_key) { DocId::Present(id) => { @@ -166,8 +170,12 @@ pub(super) fn resolve_doc_identity( Ok((id, s)) } DocId::ExplicitNull | DocId::Absent => { - let s = assign_fresh(ctx, collection)?; - Ok((s.as_u32().to_string(), s)) + let (s, pk) = assign_fresh( + ctx, + collection, + nodedb_physical::FreshSurrogateKind::DocumentStorageKey, + )?; + Ok((pk, s)) } } } @@ -181,11 +189,19 @@ pub(super) fn assign_for_pk( } /// Allocate a fresh, unique surrogate for a row whose primary key is the -/// auto-generated `_rowid` (no `PRIMARY KEY` declared). Content-addressing an -/// empty pk here would collapse every such row onto one surrogate — a -/// duplicate-key violation on the second insert. -pub(super) fn assign_fresh(ctx: &ConvertContext, collection: &str) -> crate::Result { - ctx.fresh_surrogate(collection) +/// auto-generated `_rowid` (no `PRIMARY KEY` declared), or that carries no +/// content primary key at all. +/// +/// Content-addressing an empty pk collapses every such row onto one +/// surrogate, a duplicate-key violation on the second insert. +/// +/// Returns the identity string `kind` binds. The caller uses it verbatim. +pub(super) fn assign_fresh( + ctx: &ConvertContext, + collection: &str, + kind: nodedb_physical::FreshSurrogateKind, +) -> crate::Result<(Surrogate, String)> { + ctx.fresh_surrogate(collection, kind) } /// Whether a collection's declared primary key is the auto-generated `_rowid` diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs b/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs index 43e05ec5a..7c4616c99 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs @@ -110,11 +110,16 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_timeseries_ingest( // A timeseries row's natural identity is its (timestamp, tag-set) // tuple, which is not a cross-engine surrogate and carries no PK // column. Mint a FRESH unique surrogate per row (mirroring the - // columnar auto-`_rowid` path in `dml/insert.rs`) so every row - // occupies its own transaction-overlay slot for statement-time - // read-your-own-writes staging. Content-addressing an empty PK would - // collapse every row onto `Surrogate::ZERO` and merge distinct rows. - let s = ctx.fresh_surrogate(collection)?; + // document-storage-key path a keyless schemaless row takes in + // `dml/insert.rs`), so every row holds its own transaction-overlay + // slot for read-your-own-writes staging. Content-addressing an empty + // PK collapses every row onto `Surrogate::ZERO` and merges distinct + // rows. Nothing looks a timeseries row up by this binding, so the + // identity string is discarded. + let (s, _) = ctx.fresh_surrogate( + collection, + nodedb_physical::FreshSurrogateKind::DocumentStorageKey, + )?; surrogates.push(s); } Ok(vec![PhysicalTask { diff --git a/nodedb/src/control/server/response_shape/project.rs b/nodedb/src/control/server/response_shape/project.rs index 53e4848da..a4483634e 100644 --- a/nodedb/src/control/server/response_shape/project.rs +++ b/nodedb/src/control/server/response_shape/project.rs @@ -36,8 +36,14 @@ pub fn push_flat_rows( } serde_json::Value::Object(mut map) => { if is_scan_wrapper(&map) - && let Some(serde_json::Value::Object(inner)) = map.remove("data") + && let Some(serde_json::Value::Object(mut inner)) = map.remove("data") { + // The envelope carries the row's storage-key identity. A body + // with no `id` field carries it nowhere else. `or_insert` + // leaves a declared primary key as the authority. + if let Some(id) = map.remove("id") { + inner.entry("id").or_insert(id); + } out.push(inner); return; } diff --git a/nodedb/src/control/surrogate/assign/core/assign_ops.rs b/nodedb/src/control/surrogate/assign/core/assign_ops.rs index 41c3e9435..7eaa70b22 100644 --- a/nodedb/src/control/surrogate/assign/core/assign_ops.rs +++ b/nodedb/src/control/surrogate/assign/core/assign_ops.rs @@ -9,6 +9,24 @@ use nodedb_types::Surrogate; use super::types::SurrogateAssigner; +/// Format the identity string a freshly minted surrogate binds under. +/// +/// This is the one site in `nodedb` that formats a minted surrogate. +/// [`assign_fresh`](SurrogateAssigner::assign_fresh) calls it on the +/// allocation path. The SQL-plan converter's placeholder paths call it for +/// `Surrogate::ZERO`, so both stay on one rule. +pub(crate) fn fresh_identity_string( + kind: nodedb_physical::FreshSurrogateKind, + surrogate: Surrogate, +) -> String { + match kind { + nodedb_physical::FreshSurrogateKind::AutoRowId => surrogate.as_u32().to_string(), + nodedb_physical::FreshSurrogateKind::DocumentStorageKey => { + crate::engine::document::store::surrogate_to_doc_id(surrogate) + } + } +} + impl SurrogateAssigner { /// Resolve `(collection, pk_bytes)` to a stable surrogate. /// @@ -106,22 +124,26 @@ impl SurrogateAssigner { /// Allocate a FRESH surrogate for a row with no content primary key — a /// collection whose primary key is the auto-generated `_rowid` (no - /// `PRIMARY KEY` was declared). Unlike [`assign`](Self::assign), there is - /// no fast-path lookup: every call allocates a new value, so N rows get N - /// distinct surrogates instead of collapsing onto the binding for an empty - /// key. + /// `PRIMARY KEY` was declared), or a timeseries row. /// - /// The surrogate is self-bound (pk = its own decimal string). The Data - /// Plane sets the row's `_rowid` field equal to this surrogate, so the - /// self-binding makes a later `WHERE _rowid = N` point lookup resolve back - /// to it, and reuses the same durable bind/flush machinery as `assign` so - /// the hwm advance is persisted and Raft-proposed identically. + /// [`assign`](Self::assign) has a fast-path lookup. This does not. Every + /// call allocates a new value, so N rows get N distinct surrogates. + /// + /// The surrogate self-binds to the identity string `kind` picks. + /// `AutoRowId` binds the decimal string, matching the `_rowid` value the + /// Data Plane writes, so `WHERE _rowid = N` resolves back to it. + /// `DocumentStorageKey` binds the 8-hex key the row is stored under. + /// + /// Both reuse `assign`'s bind/flush machinery, so the hwm advance + /// persists and Raft-proposes identically. Returns the bound identity + /// string, which the caller uses verbatim. pub fn assign_fresh( &self, database_id: DatabaseId, tenant_id: TenantId, collection: &str, - ) -> crate::Result { + kind: nodedb_physical::FreshSurrogateKind, + ) -> crate::Result<(Surrogate, String)> { let catalog = self.credential_store.catalog(); // Allocate + self-bind + maybe-flush under the registry write lock, @@ -140,10 +162,7 @@ impl SurrogateAssigner { continue; } }; - // Self-bind: pk is the surrogate's own decimal string, matching the - // `_rowid` value the Data Plane writes (surrogate as i64) once run - // through `sql_value_to_string`, so `WHERE _rowid = N` resolves. - let pk = surrogate.as_u32().to_string(); + let pk = fresh_identity_string(kind, surrogate); let pk_bytes = pk.as_bytes(); catalog.put_surrogate(database_id, tenant_id, collection, pk_bytes, surrogate)?; self.wal_appender.record_bind_to_wal( @@ -154,7 +173,7 @@ impl SurrogateAssigner { pk_bytes, )?; self.maybe_flush(®istry, catalog)?; - return Ok(surrogate); + return Ok((surrogate, pk)); } } diff --git a/nodedb/src/control/surrogate/assign/core/mod.rs b/nodedb/src/control/surrogate/assign/core/mod.rs index 3b12349fe..30539977c 100644 --- a/nodedb/src/control/surrogate/assign/core/mod.rs +++ b/nodedb/src/control/surrogate/assign/core/mod.rs @@ -22,4 +22,5 @@ mod assign_ops; mod flush; mod types; +pub(crate) use assign_ops::fresh_identity_string; pub use types::{SurrogateAssigner, SurrogateRegistryHandle}; diff --git a/nodedb/src/control/surrogate/assign/mod.rs b/nodedb/src/control/surrogate/assign/mod.rs index 42aed71ce..226382bce 100644 --- a/nodedb/src/control/surrogate/assign/mod.rs +++ b/nodedb/src/control/surrogate/assign/mod.rs @@ -6,4 +6,5 @@ pub(super) mod cluster_reserve; pub mod core; +pub(crate) use core::fresh_identity_string; pub use core::{SurrogateAssigner, SurrogateRegistryHandle}; diff --git a/nodedb/src/control/surrogate/mod.rs b/nodedb/src/control/surrogate/mod.rs index 755ed73c8..2a2bba6de 100644 --- a/nodedb/src/control/surrogate/mod.rs +++ b/nodedb/src/control/surrogate/mod.rs @@ -14,6 +14,7 @@ pub mod physical_impl; pub mod registry; pub mod wal_appender; +pub(crate) use assign::fresh_identity_string; pub use assign::{SurrogateAssigner, SurrogateRegistryHandle}; pub use bootstrap::bootstrap_registry; pub use persist::{SURROGATE_HWM, SurrogateHwmPersist, SystemCatalogHwm}; diff --git a/nodedb/src/control/surrogate/physical_impl.rs b/nodedb/src/control/surrogate/physical_impl.rs index 50acd0d50..54958d3b8 100644 --- a/nodedb/src/control/surrogate/physical_impl.rs +++ b/nodedb/src/control/surrogate/physical_impl.rs @@ -4,7 +4,9 @@ //! `nodedb_physical::SurrogateAssigner` trait so the shared converter //! can allocate surrogates without depending on Origin internals. -use nodedb_physical::{SurrogateAssignError, SurrogateAssigner as PhysicalSurrogateAssigner}; +use nodedb_physical::{ + FreshSurrogateKind, SurrogateAssignError, SurrogateAssigner as PhysicalSurrogateAssigner, +}; use super::assign::SurrogateAssigner; @@ -29,8 +31,9 @@ impl PhysicalSurrogateAssigner for SurrogateAssigner { database_id: nodedb_types::DatabaseId, tenant_id: nodedb_types::TenantId, collection: &str, - ) -> Result { - Self::assign_fresh(self, database_id, tenant_id, collection) + kind: FreshSurrogateKind, + ) -> Result<(nodedb_types::Surrogate, String), SurrogateAssignError> { + Self::assign_fresh(self, database_id, tenant_id, collection, kind) .map_err(|e| SurrogateAssignError::Backend(e.to_string())) } } diff --git a/nodedb/src/control/target_identity/document_id.rs b/nodedb/src/control/target_identity/document_id.rs index 974ec8322..bba51d62f 100644 --- a/nodedb/src/control/target_identity/document_id.rs +++ b/nodedb/src/control/target_identity/document_id.rs @@ -7,22 +7,28 @@ use nodedb_types::Surrogate; use super::pk::{TargetPk, extract_pk_value}; +use crate::control::surrogate::fresh_identity_string; /// The user-visible primary key (`document_id`) for a row written on this /// target, mirroring the plain-`INSERT` identity path (`insert.rs`): an /// auto-`_rowid` row's PK is the decimal surrogate the Data Plane also writes -/// into `_rowid`; a declared-PK row's PK is the field value extracted from -/// the body. +/// into `_rowid`. A declared-PK row's PK is the field value extracted from +/// the body. A row with no content key is stored under its document storage +/// key. +/// +/// Both forms come from `fresh_identity_string`, the same formatter +/// `assign_target_surrogate` binds a freshly minted row under. pub(crate) fn derive_document_id( target_pk: &TargetPk, body: &[u8], surrogate: Surrogate, ) -> String { + use nodedb_physical::FreshSurrogateKind; match target_pk { - TargetPk::AutoRowId => surrogate.as_u32().to_string(), - TargetPk::Field { name, .. } => { - extract_pk_value(body, name).unwrap_or_else(|| surrogate.as_u32().to_string()) - } + TargetPk::AutoRowId => fresh_identity_string(FreshSurrogateKind::AutoRowId, surrogate), + TargetPk::Field { name, .. } => extract_pk_value(body, name).unwrap_or_else(|| { + fresh_identity_string(FreshSurrogateKind::DocumentStorageKey, surrogate) + }), } } diff --git a/nodedb/src/control/target_identity/surrogate.rs b/nodedb/src/control/target_identity/surrogate.rs index 59af4b8f8..5d52e7c91 100644 --- a/nodedb/src/control/target_identity/surrogate.rs +++ b/nodedb/src/control/target_identity/surrogate.rs @@ -20,9 +20,13 @@ pub(crate) fn assign_target_surrogate( ) -> crate::Result { match target_pk { TargetPk::AutoRowId => { - state - .surrogate_assigner - .assign_fresh(database_id, tenant_id, target_collection) + let (surrogate, _) = state.surrogate_assigner.assign_fresh( + database_id, + tenant_id, + target_collection, + nodedb_physical::FreshSurrogateKind::AutoRowId, + )?; + Ok(surrogate) } TargetPk::Field { name, declared } => match extract_pk_value(body, name) { // The empty string is a key like any other. Minting a fresh @@ -43,10 +47,17 @@ pub(crate) fn assign_target_surrogate( }), // Undeclared `id`-by-convention field: mint a fresh unique // surrogate rather than collapsing every keyless row onto one - // binding. - _ => state - .surrogate_assigner - .assign_fresh(database_id, tenant_id, target_collection), + // binding. The row's identity is its document storage key, so + // the allocator binds the hex form. + _ => { + let (surrogate, _) = state.surrogate_assigner.assign_fresh( + database_id, + tenant_id, + target_collection, + nodedb_physical::FreshSurrogateKind::DocumentStorageKey, + )?; + Ok(surrogate) + } }, } } diff --git a/nodedb/tests/wire/cases/sql_transactions_pk_scan_consistency.rs b/nodedb/tests/wire/cases/sql_transactions_pk_scan_consistency.rs index e7df7d5ae..96381c0c4 100644 --- a/nodedb/tests/wire/cases/sql_transactions_pk_scan_consistency.rs +++ b/nodedb/tests/wire/cases/sql_transactions_pk_scan_consistency.rs @@ -231,3 +231,166 @@ async fn point_lookup_scan_and_count_agree_after_restart() { "COUNT(*) must equal the scanned row count after restart, got {count:?}" ); } + +// ── Minted identity (no declared PRIMARY KEY) ─────────────────────────────── +// +// A schemaless collection with no declared `PRIMARY KEY` mints its row +// identity from the surrogate counter. That identity must be one value, not +// one per read path: `RETURNING id`, `SELECT id`, `SELECT *`, the scan +// predicate, the aggregate predicate, and the UPDATE key must all name it the +// same way. The tests above cover the declared-key case only, where the body +// carries `id` and every path reads it from there. + +/// Insert one row into a collection with no declared `PRIMARY KEY` and return +/// the id `RETURNING` reports. +async fn insert_minted_row(server: &TestServer, table: &str) -> String { + let returned = server + .query_text(&format!( + "INSERT INTO {table} (v) VALUES ('x') RETURNING id" + )) + .await + .unwrap(); + assert_eq!( + returned.len(), + 1, + "RETURNING id must report exactly one id, got {returned:?}" + ); + returned[0].clone() +} + +/// `RETURNING id` and `SELECT id` name the same row, so they must answer the +/// same string. Two encodings of one surrogate is one identity too many. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn minted_identity_agrees_between_returning_and_projection() { + let server = TestServer::start().await; + server + .exec("CREATE COLLECTION mint_ret (v TEXT)") + .await + .unwrap(); + + let returned = insert_minted_row(&server, "mint_ret").await; + let projected = server.query_text("SELECT id FROM mint_ret").await.unwrap(); + + assert_eq!( + projected, + vec![returned.clone()], + "SELECT id must answer the same identity RETURNING id reported ('{returned}')" + ); +} + +/// The identity a read returns must address the row it came from. A client +/// that reads an id and cannot fetch that row again holds a value that names +/// nothing. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn minted_identity_addresses_its_own_row() { + let server = TestServer::start().await; + server + .exec("CREATE COLLECTION mint_addr (v TEXT)") + .await + .unwrap(); + + let returned = insert_minted_row(&server, "mint_addr").await; + let projected = server.query_text("SELECT id FROM mint_addr").await.unwrap(); + + for id in [returned.as_str(), projected[0].as_str()] { + let rows = server + .query_text(&format!("SELECT v FROM mint_addr WHERE id = '{id}'")) + .await + .unwrap(); + assert_eq!( + rows, + vec!["x".to_string()], + "id '{id}' was returned by a read of this row and must address it" + ); + } +} + +/// The scan path and the aggregate path answer the same predicate over `id` +/// identically. A predicate that finds the row on one path and misses it on +/// the other is a silently wrong count. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn minted_identity_scan_and_aggregate_agree_on_equality() { + let server = TestServer::start().await; + server + .exec("CREATE COLLECTION mint_agg (v TEXT)") + .await + .unwrap(); + + let returned = insert_minted_row(&server, "mint_agg").await; + let projected = server.query_text("SELECT id FROM mint_agg").await.unwrap(); + + // Both encodings a read has ever produced for this row are probed. Any + // value that addresses the row on one path must address it on both. + for id in [returned.as_str(), projected[0].as_str()] { + let scan = server + .query_text(&format!("SELECT v FROM mint_agg WHERE id = '{id}'")) + .await + .unwrap(); + let count = server + .query_text(&format!("SELECT count(*) FROM mint_agg WHERE id = '{id}'")) + .await + .unwrap(); + assert_eq!( + count, + vec![scan.len().to_string()], + "scan and count(*) must agree on `id = '{id}'`; scan got {scan:?}, count got {count:?}" + ); + } +} + +/// `SELECT *` exposes `id` whenever `SELECT id` resolves it. A projection that +/// drops the row's only identity leaves the client no way to address it. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn minted_identity_star_projection_includes_id() { + let server = TestServer::start().await; + server + .exec("CREATE COLLECTION mint_star (v TEXT)") + .await + .unwrap(); + + let returned = insert_minted_row(&server, "mint_star").await; + let star = server.query_rows("SELECT * FROM mint_star").await.unwrap(); + + assert_eq!(star.len(), 1, "one row inserted, got {star:?}"); + assert!( + star[0].iter().any(|cell| cell == &returned), + "SELECT * must carry the row's identity '{returned}', got {star:?}" + ); +} + +/// A read-then-write round trip addresses the row by the identity the read +/// reported. `UPDATE` keyed on that value touches that row and no other. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn minted_identity_addresses_its_own_row_for_update() { + let server = TestServer::start().await; + server + .exec("CREATE COLLECTION mint_upd (v TEXT)") + .await + .unwrap(); + + let returned = insert_minted_row(&server, "mint_upd").await; + server + .exec(&format!( + "UPDATE mint_upd SET v = 'y' WHERE id = '{returned}'" + )) + .await + .unwrap(); + + let rows = server.query_text("SELECT v FROM mint_upd").await.unwrap(); + assert_eq!( + rows, + vec!["y".to_string()], + "UPDATE keyed on the id RETURNING reported ('{returned}') must reach the row, got {rows:?}" + ); + + // The update addressed an existing row, so it minted no second one. + let count = server + .query_text("SELECT count(*) FROM mint_upd") + .await + .unwrap(); + assert_eq!( + count, + vec!["1".to_string()], + "the collection must still hold exactly one row, got {count:?}" + ); +} From c99e07f2f1eb32cdd6a429e8b6b86060014cf55c Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 11 Sep 2026 00:26:48 +0800 Subject: [PATCH 02/17] feat(sql): reject duplicate natural-key PRIMARY KEY on columnar INSERT A PRIMARY KEY declared on a columnar collection's natural key column (not id / document_id) tombstoned and re-appended the row on a duplicate insert, matching the id-based convention but not ANSI SQL uniqueness semantics. ColumnarInsertIntent::InsertUnique now checks the PK index first and raises RejectedConstraint (23505) instead, while ON CONFLICT DO NOTHING keeps its existing InsertIfAbsent path regardless of the declared key. build_columnar_schema and build_schema_bytes take the declared primary-key column name so the schema seeded at bootstrap and the one built per-statement agree on which column is the key. columnar_row_surrogates and resolve_doc_identity mint per-row surrogates off the same declared column rather than guessing from the id/document_id/key convention, so distinct natural keys never collapse onto one surrogate. The insert conversion module splits into insert/{convert,identity, schema}.rs along those three concerns, and its wire test grows a declared_key.rs and spatial.rs case file alongside the existing columnar.rs cases. --- nodedb-physical/src/physical_plan/columnar.rs | 5 + nodedb/src/bootstrap/data_plane.rs | 1 + .../dml/{insert.rs => insert/convert.rs} | 279 +++--------------- .../sql_plan_convert/dml/insert/identity.rs | 193 ++++++++++++ .../sql_plan_convert/dml/insert/mod.rs | 13 + .../sql_plan_convert/dml/insert/schema.rs | 81 +++++ .../planner/sql_plan_convert/dml/upsert.rs | 15 +- .../handlers/columnar_write/row_ingest.rs | 47 ++- .../handlers/transaction/sub_plan_kv.rs | 6 +- .../columnar.rs} | 152 ---------- .../declared_key.rs | 241 +++++++++++++++ .../cases/sql_insert_conflict_columnar/mod.rs | 20 ++ .../sql_insert_conflict_columnar/spatial.rs | 53 ++++ 13 files changed, 715 insertions(+), 391 deletions(-) rename nodedb/src/control/planner/sql_plan_convert/dml/{insert.rs => insert/convert.rs} (55%) create mode 100644 nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs create mode 100644 nodedb/src/control/planner/sql_plan_convert/dml/insert/mod.rs create mode 100644 nodedb/src/control/planner/sql_plan_convert/dml/insert/schema.rs rename nodedb/tests/wire/cases/{sql_insert_conflict_columnar.rs => sql_insert_conflict_columnar/columnar.rs} (58%) create mode 100644 nodedb/tests/wire/cases/sql_insert_conflict_columnar/declared_key.rs create mode 100644 nodedb/tests/wire/cases/sql_insert_conflict_columnar/mod.rs create mode 100644 nodedb/tests/wire/cases/sql_insert_conflict_columnar/spatial.rs diff --git a/nodedb-physical/src/physical_plan/columnar.rs b/nodedb-physical/src/physical_plan/columnar.rs index fc0131130..4ac6ce611 100644 --- a/nodedb-physical/src/physical_plan/columnar.rs +++ b/nodedb-physical/src/physical_plan/columnar.rs @@ -42,6 +42,11 @@ pub enum ColumnarInsertIntent { /// assignments (with `EXCLUDED.col` bound to the incoming row), and /// writes the merged result. Put, + /// Plain `INSERT` on a collection whose `PRIMARY KEY` is declared on a + /// natural key column (not `id` / `document_id`). Duplicate PK refuses + /// the row with `RejectedConstraint` (SQLSTATE 23505) instead of the + /// `Insert` tombstone-and-append behavior above. + InsertUnique, } /// Base columnar physical operations. diff --git a/nodedb/src/bootstrap/data_plane.rs b/nodedb/src/bootstrap/data_plane.rs index 16427cffa..9c93f5962 100644 --- a/nodedb/src/bootstrap/data_plane.rs +++ b/nodedb/src/bootstrap/data_plane.rs @@ -257,6 +257,7 @@ pub fn load_columnar_schema_seed( .filter_map(|(database_id, coll)| { let schema = crate::control::planner::sql_plan_convert::dml::build_columnar_schema( &coll.fields, + coll.declared_primary_key.as_deref(), )?; Some(( database_id, diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs similarity index 55% rename from nodedb/src/control/planner/sql_plan_convert/dml/insert.rs rename to nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs index f720e6485..8f771994b 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/insert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs @@ -2,237 +2,23 @@ use nodedb_sql::types::{SqlValue, WriteRoute}; use nodedb_types::Surrogate; -use nodedb_types::columnar::{ColumnDef, ColumnType, ColumnarSchema}; use crate::bridge::envelope::PhysicalPlan; use crate::types::{TenantId, VShardId}; -use nodedb_physical::physical_plan::ColumnarInsertIntent; use nodedb_physical::physical_plan::*; -use super::super::convert::ConvertContext; -use super::super::value::{ - expand_row_defaults, row_to_msgpack, rows_to_msgpack_array, sql_value_to_string, -}; +use super::super::super::convert::ConvertContext; +use super::super::super::value::{expand_row_defaults, row_to_msgpack, rows_to_msgpack_array}; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; -/// Build a `ColumnarSchema` from raw catalog column-type strings. -/// -/// `column_schema` is the list of `(column_name, type_str)` pairs from the -/// DDL catalog (`stored.fields`). Unknown type strings are treated as -/// `ColumnType::String` (matching the memtable's existing fallback). -/// -/// The `id` column is treated as the primary key when present; all other -/// columns are treated as nullable. -/// -/// Returns `None` when `column_schema` is empty (no catalog schema available -/// — test fixtures and legacy paths) or the resulting schema fails -/// validation. -/// -/// This is the single source of truth for turning a catalog's raw -/// `(name, type_str)` field list into a typed `ColumnarSchema` — shared by -/// the live SQL insert path (via [`build_schema_bytes`]) and -/// `bootstrap::data_plane::load_columnar_schema_seed`, which pre-registers -/// each columnar-family collection's real schema before WAL replay so a -/// fresh `MutationEngine` never falls back to type-lossy inference. -pub(crate) fn build_columnar_schema(column_schema: &[(String, String)]) -> Option { - if column_schema.is_empty() { - return None; - } - let mut cols = Vec::with_capacity(column_schema.len()); - let mut has_id = false; - for (name, type_str) in column_schema { - // `type_str` may contain SQL modifiers such as `NOT NULL` or `PRIMARY KEY` - // (e.g. "BIGINT NOT NULL"). Strip everything after the first token so that - // `ColumnType::from_str` receives the bare type name (e.g. "BIGINT"). - let bare_type = type_str - .split_whitespace() - .next() - .unwrap_or(type_str.as_str()); - let col_type = bare_type - .parse::() - .unwrap_or(ColumnType::String); - let is_id = name == "id" || name == "document_id"; - if is_id { - has_id = true; - cols.push(ColumnDef::required(name.clone(), col_type).with_primary_key()); - } else { - cols.push(ColumnDef::nullable(name.clone(), col_type)); - } - } - // If no PK column found in stored.fields, inject a synthetic one. - if !has_id { - cols.insert( - 0, - ColumnDef::required("id", ColumnType::String).with_primary_key(), - ); - } - ColumnarSchema::new(cols).ok() -} - -/// Build a `ColumnarSchema` from raw catalog column-type strings, then -/// serialize it as MessagePack for the `ColumnarOp::Insert::schema_bytes` field. -/// -/// Returns an empty `Vec` when `column_schema` is empty or fails validation -/// — see [`build_columnar_schema`] for the typed builder this wraps. -pub(super) fn build_schema_bytes(column_schema: &[(String, String)]) -> Vec { - build_columnar_schema(column_schema) - .map(|schema| zerompk::to_msgpack_vec(&schema).unwrap_or_default()) - .unwrap_or_default() -} - -/// A row's primary-key column as found during identity derivation. -/// -/// `Present("")` is the empty string — a real key, not an absence. -pub(super) enum DocId { - Present(String), - ExplicitNull, - Absent, -} - -/// Extract the document-id value from a row, keyed off the declared -/// `primary_key` column when present, falling back to the legacy -/// `id`/`document_id`/`key` convention otherwise. -pub(super) fn extract_doc_id(row: &[(String, SqlValue)], primary_key: Option<&str>) -> DocId { - match row.iter().find(|(k, _)| match primary_key { - Some(pk) => k == pk, - None => k == "id" || k == "document_id" || k == "key", - }) { - Some((_, SqlValue::Null)) => DocId::ExplicitNull, - Some((_, v)) => DocId::Present(sql_value_to_string(v)), - None => DocId::Absent, - } -} - -/// `collection`'s DDL-declared `PRIMARY KEY` column name, if any. -/// -/// `primary_key` cannot answer this: schemaless, columnar, and spatial -/// collections resolve it to `id` by convention with nothing declared. The -/// catalog's `declared_primary_key` is set only by the keyword itself, and -/// names the column the keyword applied `NOT NULL` to. A catalog miss reads -/// as not declared — nothing to enforce. -pub(in super::super) fn declared_primary_key_name( - ctx: &ConvertContext, - collection: &str, -) -> crate::Result> { - let Some(credentials) = ctx.credentials.as_ref() else { - return Ok(None); - }; - credentials - .catalog() - .declared_primary_key(ctx.database_id, ctx.tenant_id.as_u64(), collection) -} - -/// Resolve a row's document id + surrogate, refusing a `NULL`/omitted -/// declared primary key first: a declared `PRIMARY KEY` implies `NOT NULL`. -/// -/// Enforcement keys on the DDL-declared column, not the resolved -/// `primary_key`: those diverge whenever a natural key sits on a column -/// other than `id` (e.g. `metrics (sku TEXT PRIMARY KEY)` resolves -/// `primary_key` to `id` but declares `sku`). `_rowid` carries no -/// declaration, so it skips the check and mints a surrogate. -/// -/// Identity minting then runs on the resolved `primary_key`: an auto-`_rowid` -/// pk or a missing/null key mints a fresh surrogate; a present key -/// content-addresses one via [`assign_for_pk`]. The two steps are one call so -/// no caller can mint an identity without the NOT NULL check running first. -pub(super) fn resolve_doc_identity( - ctx: &ConvertContext, - collection: &str, - primary_key: Option<&str>, - row: &[(String, SqlValue)], -) -> crate::Result<(String, Surrogate)> { - if !is_auto_rowid_pk(primary_key) - && let Some(declared) = declared_primary_key_name(ctx, collection)? - { - match extract_doc_id(row, Some(&declared)) { - DocId::Present(_) => {} - DocId::ExplicitNull | DocId::Absent => { - return Err(crate::Error::RejectedConstraint { - collection: collection.to_string(), - constraint: "not_null".to_string(), - detail: format!("primary key '{declared}' cannot be NULL or omitted"), - }); - } - } - } - - if is_auto_rowid_pk(primary_key) { - let (s, pk) = assign_fresh( - ctx, - collection, - nodedb_physical::FreshSurrogateKind::AutoRowId, - )?; - return Ok((pk, s)); - } - match extract_doc_id(row, primary_key) { - DocId::Present(id) => { - let s = assign_for_pk(ctx, collection, id.as_bytes())?; - Ok((id, s)) - } - DocId::ExplicitNull | DocId::Absent => { - let (s, pk) = assign_fresh( - ctx, - collection, - nodedb_physical::FreshSurrogateKind::DocumentStorageKey, - )?; - Ok((pk, s)) - } - } -} - -pub(super) fn assign_for_pk( - ctx: &ConvertContext, - collection: &str, - pk_bytes: &[u8], -) -> crate::Result { - ctx.surrogate_for_pk(collection, pk_bytes) -} - -/// Allocate a fresh, unique surrogate for a row whose primary key is the -/// auto-generated `_rowid` (no `PRIMARY KEY` declared), or that carries no -/// content primary key at all. -/// -/// Content-addressing an empty pk collapses every such row onto one -/// surrogate, a duplicate-key violation on the second insert. -/// -/// Returns the identity string `kind` binds. The caller uses it verbatim. -pub(super) fn assign_fresh( - ctx: &ConvertContext, - collection: &str, - kind: nodedb_physical::FreshSurrogateKind, -) -> crate::Result<(Surrogate, String)> { - ctx.fresh_surrogate(collection, kind) -} - -/// Whether a collection's declared primary key is the auto-generated `_rowid` -/// sentinel — injected by strict-schema construction when no `PRIMARY KEY` was -/// declared. Such rows carry no user identity: each needs a fresh surrogate. -pub(super) fn is_auto_rowid_pk(primary_key: Option<&str>) -> bool { - primary_key == Some("_rowid") -} - -/// Mirrors the document-engine identity path (`resolve_doc_identity`) for -/// columnar/spatial rows. The declared `primary_key` — not the legacy -/// `id`/`document_id`/`key` name guess — determines each row's identity, so a -/// natural key on any column (e.g. `sku`) gets its own surrogate. A -/// missing/empty key mints a fresh unique surrogate rather than collapsing -/// onto `Surrogate::ZERO`, which would silently merge distinct rows. -pub(super) fn columnar_row_surrogates( - ctx: &ConvertContext, - collection: &str, - columnar_rows: &[&Vec<(String, SqlValue)>], - primary_key: Option<&str>, -) -> crate::Result> { - let mut out = Vec::with_capacity(columnar_rows.len()); - for row in columnar_rows { - let (_, surrogate) = resolve_doc_identity(ctx, collection, primary_key, row)?; - out.push(surrogate); - } - Ok(out) -} +use super::identity::{ + columnar_row_surrogates, declared_primary_key_name, is_auto_rowid_pk, + resolve_doc_identity_with_declared, +}; +use super::schema::build_schema_bytes; /// Bundled arguments for [`convert_insert`]. -pub(in super::super) struct ConvertInsertArgs<'a> { +pub(crate) struct ConvertInsertArgs<'a> { pub collection: &'a str, /// The lowering these rows take, decided by `nodedb-sql`. pub route: WriteRoute, @@ -245,9 +31,7 @@ pub(in super::super) struct ConvertInsertArgs<'a> { pub ctx: &'a ConvertContext, } -pub(in super::super) fn convert_insert( - args: ConvertInsertArgs<'_>, -) -> crate::Result> { +pub(crate) fn convert_insert(args: ConvertInsertArgs<'_>) -> crate::Result> { let ConvertInsertArgs { collection, route, @@ -259,7 +43,7 @@ pub(in super::super) fn convert_insert( tenant_id, ctx, } = args; - let coll_qualified = super::super::convert::db_qualified(ctx.database_id, collection); + let coll_qualified = super::super::super::convert::db_qualified(ctx.database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); let collection = coll_qualified.as_str(); let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); @@ -271,7 +55,7 @@ pub(in super::super) fn convert_insert( // // `IF NOT EXISTS` (ON CONFLICT DO NOTHING → `if_absent`) cannot be honored by // `CrdtOp::DocUpsert`, which is an unconditional LWW full-replace: reject. - let gates = super::balanced_gate::document_collection_write_gates(ctx, collection)?; + let gates = super::super::balanced_gate::document_collection_write_gates(ctx, collection)?; let is_crdt = gates.crdt; if is_crdt && if_absent { return Err(crate::Error::BadRequest { @@ -308,6 +92,14 @@ pub(in super::super) fn convert_insert( // refuses a key the declaration supplies. let expanded_rows = expand_row_defaults(rows, column_defaults, tenant_id, ctx)?; + // One catalog read for the whole statement. `_rowid` carries no + // declaration, so it skips the read. + let declared_pk = if is_auto_rowid_pk(primary_key) { + None + } else { + declared_primary_key_name(ctx, collection)? + }; + for row in &expanded_rows { match route { WriteRoute::ColumnarFamily => { @@ -315,7 +107,13 @@ pub(in super::super) fn convert_insert( } WriteRoute::Document => { let value_bytes = row_to_msgpack(row)?; - let (doc_id, surrogate) = resolve_doc_identity(ctx, collection, primary_key, row)?; + let (doc_id, surrogate) = resolve_doc_identity_with_declared( + ctx, + collection, + primary_key, + declared_pk.as_deref(), + row, + )?; // One page for the whole statement: the rows of a balanced // INSERT are judged together, so they may not be split across // one task — one boundary — per row. @@ -328,7 +126,7 @@ pub(in super::super) fn convert_insert( PhysicalPlan::Crdt(CrdtOp::DocUpsert { collection: qualified_collection.clone(), document_id: doc_id, - fields_json: super::crdt_gate::row_to_fields_json(row)?, + fields_json: super::super::crdt_gate::row_to_fields_json(row)?, surrogate, partial: false, returning: None, @@ -366,8 +164,8 @@ pub(in super::super) fn convert_insert( } if !balanced_documents.is_empty() { - tasks.push(super::balanced_gate::balanced_batch_task( - super::balanced_gate::BalancedBatch { + tasks.push(super::super::balanced_gate::balanced_batch_task( + super::super::balanced_gate::BalancedBatch { collection, tenant_id, vshard, @@ -380,13 +178,28 @@ pub(in super::super) fn convert_insert( if !columnar_rows.is_empty() { let payload = rows_to_msgpack_array(&columnar_rows)?; + // `ON CONFLICT DO NOTHING` means skip, not refuse: `if_absent` keeps + // `InsertIfAbsent` regardless of the declared key. Otherwise, a + // `PRIMARY KEY` declared on a natural key column (not `id` / + // `document_id`) refuses a duplicate rather than tombstoning it. let intent = if if_absent { ColumnarInsertIntent::InsertIfAbsent + } else if declared_pk + .as_deref() + .is_some_and(|pk| pk != "id" && pk != "document_id") + { + ColumnarInsertIntent::InsertUnique } else { ColumnarInsertIntent::Insert }; - let surrogates = columnar_row_surrogates(ctx, collection, &columnar_rows, primary_key)?; - let schema_bytes = build_schema_bytes(column_schema); + let surrogates = columnar_row_surrogates( + ctx, + collection, + &columnar_rows, + primary_key, + declared_pk.as_deref(), + )?; + let schema_bytes = build_schema_bytes(column_schema, declared_pk.as_deref()); tasks.push(PhysicalTask { tenant_id, vshard_id: vshard, diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs new file mode 100644 index 000000000..19e406944 --- /dev/null +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs @@ -0,0 +1,193 @@ +// SPDX-License-Identifier: BUSL-1.1 + +use nodedb_sql::types::SqlValue; +use nodedb_types::Surrogate; + +use super::super::super::convert::ConvertContext; +use super::super::super::value::sql_value_to_string; + +/// A row's primary-key column as found during identity derivation. +/// +/// `Present("")` is the empty string — a real key, not an absence. +pub(super) enum DocId { + Present(String), + ExplicitNull, + Absent, +} + +/// Extract the document-id value from a row, keyed off the declared +/// `primary_key` column when present, falling back to the legacy +/// `id`/`document_id`/`key` convention otherwise. +pub(super) fn extract_doc_id(row: &[(String, SqlValue)], primary_key: Option<&str>) -> DocId { + match row.iter().find(|(k, _)| match primary_key { + Some(pk) => k == pk, + None => k == "id" || k == "document_id" || k == "key", + }) { + Some((_, SqlValue::Null)) => DocId::ExplicitNull, + Some((_, v)) => DocId::Present(sql_value_to_string(v)), + None => DocId::Absent, + } +} + +/// `collection`'s DDL-declared `PRIMARY KEY` column name, if any. +/// +/// `primary_key` cannot answer this: schemaless, columnar, and spatial +/// collections resolve it to `id` by convention with nothing declared. The +/// catalog's `declared_primary_key` is set only by the keyword itself, and +/// names the column the keyword applied `NOT NULL` to. A catalog miss reads +/// as not declared — nothing to enforce. +pub(crate) fn declared_primary_key_name( + ctx: &ConvertContext, + collection: &str, +) -> crate::Result> { + let Some(credentials) = ctx.credentials.as_ref() else { + return Ok(None); + }; + credentials + .catalog() + .declared_primary_key(ctx.database_id, ctx.tenant_id.as_u64(), collection) +} + +/// Resolve a row's document id + surrogate, refusing a `NULL`/omitted +/// declared primary key first: a declared `PRIMARY KEY` implies `NOT NULL`. +/// +/// Enforcement keys on the DDL-declared column, not the resolved +/// `primary_key`: those diverge whenever a natural key sits on a column +/// other than `id` (e.g. `metrics (sku TEXT PRIMARY KEY)` resolves +/// `primary_key` to `id` but declares `sku`). `_rowid` carries no +/// declaration, so it skips the check and mints a surrogate. +/// +/// Identity minting then runs on the same declared name. A present declared +/// key content-addresses a surrogate via [`assign_for_pk`]. With no declared +/// key, minting falls back to the resolved `primary_key` argument. An +/// auto-`_rowid` pk or a missing/null key mints a fresh surrogate instead. +/// The two steps are one call so no caller can mint an identity without the +/// NOT NULL check running first. +pub(crate) fn resolve_doc_identity( + ctx: &ConvertContext, + collection: &str, + primary_key: Option<&str>, + row: &[(String, SqlValue)], +) -> crate::Result<(String, Surrogate)> { + // `_rowid` carries no declaration: skip the catalog read entirely and + // mint a surrogate below. + let declared = if is_auto_rowid_pk(primary_key) { + None + } else { + declared_primary_key_name(ctx, collection)? + }; + resolve_doc_identity_with_declared(ctx, collection, primary_key, declared.as_deref(), row) +} + +/// Core of [`resolve_doc_identity`], given the declared PK name already +/// resolved by the caller instead of reading it from the catalog here. +/// +/// [`columnar_row_surrogates`] resolves `declared` once for the whole +/// statement and calls this directly for every row, so a multi-row INSERT +/// hits the catalog once rather than once per row. +pub(super) fn resolve_doc_identity_with_declared( + ctx: &ConvertContext, + collection: &str, + primary_key: Option<&str>, + declared: Option<&str>, + row: &[(String, SqlValue)], +) -> crate::Result<(String, Surrogate)> { + if let Some(declared) = declared { + match extract_doc_id(row, Some(declared)) { + DocId::Present(_) => {} + DocId::ExplicitNull | DocId::Absent => { + return Err(crate::Error::RejectedConstraint { + collection: collection.to_string(), + constraint: "not_null".to_string(), + detail: format!("primary key '{declared}' cannot be NULL or omitted"), + }); + } + } + } + + if is_auto_rowid_pk(primary_key) { + let (s, pk) = assign_fresh( + ctx, + collection, + nodedb_physical::FreshSurrogateKind::AutoRowId, + )?; + return Ok((pk, s)); + } + let mint_key = declared.or(primary_key); + match extract_doc_id(row, mint_key) { + DocId::Present(id) => { + let s = assign_for_pk(ctx, collection, id.as_bytes())?; + Ok((id, s)) + } + DocId::ExplicitNull | DocId::Absent => { + let (s, pk) = assign_fresh( + ctx, + collection, + nodedb_physical::FreshSurrogateKind::DocumentStorageKey, + )?; + Ok((pk, s)) + } + } +} + +pub(crate) fn assign_for_pk( + ctx: &ConvertContext, + collection: &str, + pk_bytes: &[u8], +) -> crate::Result { + ctx.surrogate_for_pk(collection, pk_bytes) +} + +/// Allocate a fresh, unique surrogate for a row whose primary key is the +/// auto-generated `_rowid` (no `PRIMARY KEY` declared), or that carries no +/// content primary key at all. +/// +/// Content-addressing an empty pk collapses every such row onto one +/// surrogate, a duplicate-key violation on the second insert. +/// +/// Returns the identity string `kind` binds. The caller uses it verbatim. +pub(super) fn assign_fresh( + ctx: &ConvertContext, + collection: &str, + kind: nodedb_physical::FreshSurrogateKind, +) -> crate::Result<(Surrogate, String)> { + ctx.fresh_surrogate(collection, kind) +} + +/// Whether a collection's declared primary key is the auto-generated `_rowid` +/// sentinel — injected by strict-schema construction when no `PRIMARY KEY` was +/// declared. Such rows carry no user identity: each needs a fresh surrogate. +pub(super) fn is_auto_rowid_pk(primary_key: Option<&str>) -> bool { + primary_key == Some("_rowid") +} + +/// Mirrors the document-engine identity path (`resolve_doc_identity`) for +/// columnar/spatial rows. The declared `primary_key` — not the legacy +/// `id`/`document_id`/`key` name guess — determines each row's identity, so a +/// natural key on any column (e.g. `sku`) gets its own surrogate. A +/// missing/empty key mints a fresh unique surrogate rather than collapsing +/// onto `Surrogate::ZERO`, which would silently merge distinct rows. +/// +/// `declared_pk` is the catalog's declared PK name, resolved once by the +/// caller for the whole statement — not re-read here per row. +pub(crate) fn columnar_row_surrogates( + ctx: &ConvertContext, + collection: &str, + columnar_rows: &[&Vec<(String, SqlValue)>], + primary_key: Option<&str>, + declared_pk: Option<&str>, +) -> crate::Result> { + // `_rowid` carries no declaration, matching `resolve_doc_identity`. + let declared = if is_auto_rowid_pk(primary_key) { + None + } else { + declared_pk + }; + let mut out = Vec::with_capacity(columnar_rows.len()); + for row in columnar_rows { + let (_, surrogate) = + resolve_doc_identity_with_declared(ctx, collection, primary_key, declared, row)?; + out.push(surrogate); + } + Ok(out) +} diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert/mod.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert/mod.rs new file mode 100644 index 000000000..c7ac18de0 --- /dev/null +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert/mod.rs @@ -0,0 +1,13 @@ +// SPDX-License-Identifier: BUSL-1.1 + +mod convert; +mod identity; +mod schema; + +pub(crate) use schema::build_columnar_schema; +pub(super) use schema::build_schema_bytes; + +pub(crate) use identity::declared_primary_key_name; +pub(super) use identity::{assign_for_pk, columnar_row_surrogates, resolve_doc_identity}; + +pub(crate) use convert::{ConvertInsertArgs, convert_insert}; diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert/schema.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert/schema.rs new file mode 100644 index 000000000..4540d68de --- /dev/null +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert/schema.rs @@ -0,0 +1,81 @@ +// SPDX-License-Identifier: BUSL-1.1 + +use nodedb_types::columnar::{ColumnDef, ColumnType, ColumnarSchema}; + +/// Build a `ColumnarSchema` from raw catalog column-type strings. +/// +/// `column_schema` is the list of `(column_name, type_str)` pairs from the +/// DDL catalog (`stored.fields`). Unknown type strings are treated as +/// `ColumnType::String` (matching the memtable's existing fallback). +/// +/// `declared_pk` names the collection's DDL-declared `PRIMARY KEY` column, +/// when one was declared (see [`declared_primary_key_name`]). A column +/// matching that name is the schema primary key. When `declared_pk` is +/// `None`, the legacy `id` / `document_id` name convention applies instead. +/// +/// Returns `None` when `column_schema` is empty (no catalog schema available +/// — test fixtures and legacy paths) or the resulting schema fails +/// validation. +/// +/// This is the single source of truth for turning a catalog's raw +/// `(name, type_str)` field list into a typed `ColumnarSchema` — shared by +/// the live SQL insert path (via [`build_schema_bytes`]) and +/// `bootstrap::data_plane::load_columnar_schema_seed`, which pre-registers +/// each columnar-family collection's real schema before WAL replay so a +/// fresh `MutationEngine` never falls back to type-lossy inference. +pub(crate) fn build_columnar_schema( + column_schema: &[(String, String)], + declared_pk: Option<&str>, +) -> Option { + if column_schema.is_empty() { + return None; + } + let mut cols = Vec::with_capacity(column_schema.len()); + let mut has_id = false; + for (name, type_str) in column_schema { + // `type_str` may contain SQL modifiers such as `NOT NULL` or `PRIMARY KEY` + // (e.g. "BIGINT NOT NULL"). Strip everything after the first token so that + // `ColumnType::from_str` receives the bare type name (e.g. "BIGINT"). + let bare_type = type_str + .split_whitespace() + .next() + .unwrap_or(type_str.as_str()); + let col_type = bare_type + .parse::() + .unwrap_or(ColumnType::String); + let is_id = match declared_pk { + Some(pk) => name == pk, + None => name == "id" || name == "document_id", + }; + if is_id { + has_id = true; + cols.push(ColumnDef::required(name.clone(), col_type).with_primary_key()); + } else { + cols.push(ColumnDef::nullable(name.clone(), col_type)); + } + } + // If no PK column found in stored.fields, inject a synthetic one. + if !has_id { + cols.insert( + 0, + ColumnDef::required("id", ColumnType::String).with_primary_key(), + ); + } + ColumnarSchema::new(cols).ok() +} + +/// Build a `ColumnarSchema` from raw catalog column-type strings, then +/// serialize it as MessagePack for the `ColumnarOp::Insert::schema_bytes` field. +/// +/// `declared_pk` is forwarded to [`build_columnar_schema`] unchanged. +/// +/// Returns an empty `Vec` when `column_schema` is empty or fails validation +/// — see [`build_columnar_schema`] for the typed builder this wraps. +pub(crate) fn build_schema_bytes( + column_schema: &[(String, String)], + declared_pk: Option<&str>, +) -> Vec { + build_columnar_schema(column_schema, declared_pk) + .map(|schema| zerompk::to_msgpack_vec(&schema).unwrap_or_default()) + .unwrap_or_default() +} diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs b/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs index 9813c9305..05f980231 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs @@ -17,7 +17,9 @@ use super::super::convert::ConvertContext; use super::super::value::{ assignments_to_update_values, expand_row_defaults, row_to_msgpack, rows_to_msgpack_array, }; -use super::insert::{build_schema_bytes, columnar_row_surrogates, resolve_doc_identity}; +use super::insert::{ + build_schema_bytes, columnar_row_surrogates, declared_primary_key_name, resolve_doc_identity, +}; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; /// Bundled arguments for [`convert_upsert`]. @@ -130,8 +132,15 @@ pub(in super::super) fn convert_upsert( if !columnar_rows.is_empty() { let payload = rows_to_msgpack_array(&columnar_rows)?; - let surrogates = columnar_row_surrogates(ctx, collection, &columnar_rows, primary_key)?; - let schema_bytes = build_schema_bytes(column_schema); + let declared_pk = declared_primary_key_name(ctx, collection)?; + let surrogates = columnar_row_surrogates( + ctx, + collection, + &columnar_rows, + primary_key, + declared_pk.as_deref(), + )?; + let schema_bytes = build_schema_bytes(column_schema, declared_pk.as_deref()); tasks.push(PhysicalTask { tenant_id, vshard_id: vshard, diff --git a/nodedb/src/data/executor/handlers/columnar_write/row_ingest.rs b/nodedb/src/data/executor/handlers/columnar_write/row_ingest.rs index 042e961e2..d3cbab52c 100644 --- a/nodedb/src/data/executor/handlers/columnar_write/row_ingest.rs +++ b/nodedb/src/data/executor/handlers/columnar_write/row_ingest.rs @@ -55,7 +55,8 @@ impl CoreLoop { /// `params.engine_key`, applying intent-specific ON CONFLICT semantics /// (upsert-overwrite for `Insert` and `Put`, silent skip for /// `InsertIfAbsent`, merge-via-`apply_on_conflict_updates` for `Put` - /// with non-empty `on_conflict_updates`). + /// with non-empty `on_conflict_updates`, `RejectedConstraint` error for + /// `InsertUnique` on a PK the index already carries). /// /// Returns the accepted row count (and, on request, the stored post-images), /// or `Err(Response)` on the first unrecoverable error (short-circuits the @@ -209,7 +210,10 @@ impl CoreLoop { } } } - _ => values, + ColumnarInsertIntent::Put + | ColumnarInsertIntent::Insert + | ColumnarInsertIntent::InsertIfAbsent + | ColumnarInsertIntent::InsertUnique => values, }; // The row that will actually exist afterwards is decided here, not @@ -241,6 +245,45 @@ impl CoreLoop { let row_surrogate = surrogates.get(row_idx).copied(); let result = match intent { ColumnarInsertIntent::InsertIfAbsent => engine.insert_if_absent(&final_values), + ColumnarInsertIntent::InsertUnique => { + let pk_bytes = match engine.encode_pk_from_row(&final_values) { + Ok(b) => b, + Err(e) => { + return Err(self.response_error( + task, + ErrorCode::Internal { + detail: format!("columnar insert: pk encode failed: {e}"), + }, + )); + } + }; + if engine.pk_index().contains(&pk_bytes) { + let key_desc = schema + .columns + .iter() + .zip(final_values.iter()) + .filter(|(col, _)| col.primary_key) + .map(|(col, v)| format!("{}={v:?}", col.name)) + .collect::>() + .join(", "); + return Err(self.response_error( + task, + crate::Error::RejectedConstraint { + collection: engine_key.2.clone(), + constraint: "unique".to_string(), + detail: format!( + "duplicate key value '{key_desc}' violates primary-key \ + uniqueness on '{}'", + engine_key.2 + ), + }, + )); + } + match row_surrogate { + Some(s) => engine.insert_with_surrogate(&final_values, s), + None => engine.insert(&final_values), + } + } ColumnarInsertIntent::Insert | ColumnarInsertIntent::Put => match row_surrogate { Some(s) => engine.insert_with_surrogate(&final_values, s), None => engine.insert(&final_values), diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv.rs index 342e870b8..3b87d26b4 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv.rs @@ -167,7 +167,11 @@ impl CoreLoop { }; match intent { - ColumnarInsertIntent::InsertIfAbsent => { + // `InsertUnique` never displaces a prior row. A real + // conflict fails the statement before this undo entry is + // used. On the success path it matches `InsertIfAbsent`: + // record the new PK only when nothing occupies it. + ColumnarInsertIntent::InsertIfAbsent | ColumnarInsertIntent::InsertUnique => { if !engine.pk_index().contains(&pk_bytes) { inserted_pks.push(pk_bytes); } diff --git a/nodedb/tests/wire/cases/sql_insert_conflict_columnar.rs b/nodedb/tests/wire/cases/sql_insert_conflict_columnar/columnar.rs similarity index 58% rename from nodedb/tests/wire/cases/sql_insert_conflict_columnar.rs rename to nodedb/tests/wire/cases/sql_insert_conflict_columnar/columnar.rs index 9a192f085..15b20b03d 100644 --- a/nodedb/tests/wire/cases/sql_insert_conflict_columnar.rs +++ b/nodedb/tests/wire/cases/sql_insert_conflict_columnar/columnar.rs @@ -1,18 +1,5 @@ // SPDX-License-Identifier: BUSL-1.1 -//! INSERT conflict semantics for columnar-family engines. -//! -//! Columnar storage is OLAP-shaped (append-only segments, zonemap pruning), -//! but `PRIMARY KEY` appears in the ANSI SQL surface and must mean the same -//! thing across every engine NodeDB ships. The resolution is to treat PK on -//! a columnar collection as both a sort key (enforced at segment flush) and -//! a logical uniqueness constraint enforced via a sparse PK index plus -//! positional deletes: duplicate INSERTs tombstone the prior row rather -//! than raising 23505, and readers skip tombstoned row-ids. -//! -//! Spatial extends columnar and inherits the same semantics. Timeseries is -//! a different profile (append-only, time-keyed) and is not covered here. - use crate::harness::TestServer; // ── Plain columnar ────────────────────────────────────────────────────────── @@ -264,142 +251,3 @@ async fn columnar_order_by_sort_key_accepted() { "ORDER BY without PK must not dedup; got: {rows:?}" ); } - -// ── Natural-key PRIMARY KEY on a non-`id` column ──────────────────────────── - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn columnar_natural_key_pk_on_non_id_column_keeps_distinct_rows() { - // A PRIMARY KEY declared on a non-`id` column must drive row identity on - // columnar too. Two rows with DISTINCT natural keys must both stay visible - // — the built-in synthetic `id` (NULL for every row) must NOT make them - // collide into a single tombstoned row (silent data loss). - let server = TestServer::start().await; - - server - .exec( - "CREATE COLLECTION metrics (\ - sku TEXT PRIMARY KEY, region TEXT, value FLOAT\ - ) WITH (engine='columnar')", - ) - .await - .unwrap(); - server - .exec("CREATE UNIQUE INDEX metrics_pk ON metrics (sku)") - .await - .unwrap(); - - server - .exec("INSERT INTO metrics (sku, region, value) VALUES ('a', 'us-east', 1.0)") - .await - .unwrap(); - // Distinct natural key: must NOT collide on an empty synthetic `id`. - server - .exec("INSERT INTO metrics (sku, region, value) VALUES ('b', 'us-west', 2.0)") - .await - .unwrap(); - - let rows = server - .query_rows("SELECT sku, region FROM metrics ORDER BY sku") - .await - .unwrap(); - assert_eq!( - rows.len(), - 2, - "both distinct natural-key rows must stay visible, got: {rows:?}" - ); - assert_eq!(rows[0][0], "a", "got: {rows:?}"); - assert_eq!(rows[1][0], "b", "got: {rows:?}"); -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn spatial_natural_key_pk_on_non_id_column_keeps_distinct_rows() { - // Spatial inherits columnar identity: a non-`id` PRIMARY KEY must keep - // distinct natural-key rows distinct rather than colliding on synthetic id. - let server = TestServer::start().await; - - server - .exec( - "CREATE COLLECTION places (\ - code TEXT PRIMARY KEY, geom GEOMETRY SPATIAL_INDEX, label TEXT\ - ) WITH (engine='spatial')", - ) - .await - .unwrap(); - server - .exec("CREATE UNIQUE INDEX places_pk ON places (code)") - .await - .unwrap(); - - server - .exec("INSERT INTO places (code, geom, label) VALUES ('p1', ST_Point(0.0, 0.0), 'origin')") - .await - .unwrap(); - server - .exec("INSERT INTO places (code, geom, label) VALUES ('p2', ST_Point(1.0, 1.0), 'other')") - .await - .unwrap(); - - let rows = server - .query_rows("SELECT code, label FROM places ORDER BY code") - .await - .unwrap(); - assert_eq!( - rows.len(), - 2, - "both distinct natural-key rows must stay visible, got: {rows:?}" - ); - assert_eq!(rows[0][0], "p1", "got: {rows:?}"); - assert_eq!(rows[1][0], "p2", "got: {rows:?}"); -} - -// ── Spatial inherits columnar PK semantics ────────────────────────────────── - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn spatial_insert_duplicate_pk_keeps_latest() { - let server = TestServer::start().await; - - server - .exec( - "CREATE COLLECTION places (\ - id TEXT PRIMARY KEY, geom GEOMETRY SPATIAL_INDEX, label TEXT\ - ) WITH (engine='spatial')", - ) - .await - .unwrap(); - server - .exec("CREATE UNIQUE INDEX places_pk ON places (id)") - .await - .unwrap(); - - server - .exec("INSERT INTO places (id, geom, label) VALUES ('p1', ST_Point(0.0, 0.0), 'origin')") - .await - .unwrap(); - - server - .exec("INSERT INTO places (id, geom, label) VALUES ('p1', ST_Point(1.0, 1.0), 'moved')") - .await - .unwrap(); - - let rows = server - .query_rows("SELECT id, label FROM places WHERE id = 'p1'") - .await - .unwrap(); - - assert_eq!( - rows.len(), - 1, - "spatial duplicate PK must not produce two rows, got: {rows:?}" - ); - // row[0]=id, row[1]=label - assert_eq!( - rows[0][1], "moved", - "expected latest (moved), got: {:?}", - rows[0] - ); - assert_ne!( - rows[0][1], "origin", - "prior row must be tombstoned, got: {:?}", - rows[0] - ); -} diff --git a/nodedb/tests/wire/cases/sql_insert_conflict_columnar/declared_key.rs b/nodedb/tests/wire/cases/sql_insert_conflict_columnar/declared_key.rs new file mode 100644 index 000000000..296efa2d1 --- /dev/null +++ b/nodedb/tests/wire/cases/sql_insert_conflict_columnar/declared_key.rs @@ -0,0 +1,241 @@ +// SPDX-License-Identifier: BUSL-1.1 + +use crate::harness::TestServer; + +// ── Natural-key PRIMARY KEY on a non-`id` column ──────────────────────────── + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn columnar_natural_key_pk_on_non_id_column_keeps_distinct_rows() { + // A PRIMARY KEY declared on a non-`id` column must drive row identity on + // columnar too. Two rows with DISTINCT natural keys must both stay visible + // — the built-in synthetic `id` (NULL for every row) must NOT make them + // collide into a single tombstoned row (silent data loss). + let server = TestServer::start().await; + + server + .exec( + "CREATE COLLECTION metrics (\ + sku TEXT PRIMARY KEY, region TEXT, value FLOAT\ + ) WITH (engine='columnar')", + ) + .await + .unwrap(); + server + .exec("CREATE UNIQUE INDEX metrics_pk ON metrics (sku)") + .await + .unwrap(); + + server + .exec("INSERT INTO metrics (sku, region, value) VALUES ('a', 'us-east', 1.0)") + .await + .unwrap(); + // Distinct natural key: must NOT collide on an empty synthetic `id`. + server + .exec("INSERT INTO metrics (sku, region, value) VALUES ('b', 'us-west', 2.0)") + .await + .unwrap(); + + let rows = server + .query_rows("SELECT sku, region FROM metrics ORDER BY sku") + .await + .unwrap(); + assert_eq!( + rows.len(), + 2, + "both distinct natural-key rows must stay visible, got: {rows:?}" + ); + assert_eq!(rows[0][0], "a", "got: {rows:?}"); + assert_eq!(rows[1][0], "b", "got: {rows:?}"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn spatial_natural_key_pk_on_non_id_column_keeps_distinct_rows() { + // Spatial inherits columnar identity: a non-`id` PRIMARY KEY must keep + // distinct natural-key rows distinct rather than colliding on synthetic id. + let server = TestServer::start().await; + + server + .exec( + "CREATE COLLECTION places (\ + code TEXT PRIMARY KEY, geom GEOMETRY SPATIAL_INDEX, label TEXT\ + ) WITH (engine='spatial')", + ) + .await + .unwrap(); + server + .exec("CREATE UNIQUE INDEX places_pk ON places (code)") + .await + .unwrap(); + + server + .exec("INSERT INTO places (code, geom, label) VALUES ('p1', ST_Point(0.0, 0.0), 'origin')") + .await + .unwrap(); + server + .exec("INSERT INTO places (code, geom, label) VALUES ('p2', ST_Point(1.0, 1.0), 'other')") + .await + .unwrap(); + + let rows = server + .query_rows("SELECT code, label FROM places ORDER BY code") + .await + .unwrap(); + assert_eq!( + rows.len(), + 2, + "both distinct natural-key rows must stay visible, got: {rows:?}" + ); + assert_eq!(rows[0][0], "p1", "got: {rows:?}"); + assert_eq!(rows[1][0], "p2", "got: {rows:?}"); +} + +// ── Declared non-`id` PRIMARY KEY is a uniqueness constraint ──────────────── +// +// The natural-key tests above create a `UNIQUE INDEX` alongside the +// declaration, so they exercise the index rather than the declaration. +// `PRIMARY KEY` implies uniqueness on every engine, and declaring it is +// enough — these tests carry no index. + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn columnar_declared_non_id_primary_key_refuses_duplicate() { + let server = TestServer::start().await; + + server + .exec( + "CREATE COLLECTION metrics_pk (\ + sku TEXT PRIMARY KEY, value FLOAT\ + ) WITH (engine='columnar')", + ) + .await + .unwrap(); + + server + .exec("INSERT INTO metrics_pk (sku, value) VALUES ('a', 1.0)") + .await + .unwrap(); + + match server + .client + .simple_query("INSERT INTO metrics_pk (sku, value) VALUES ('a', 2.0)") + .await + { + Ok(_) => panic!("expected unique_violation on the declared primary key, got success"), + Err(e) => { + let db_err = e.as_db_error().expect("expected DbError"); + assert_eq!( + db_err.code().code(), + "23505", + "expected SQLSTATE 23505, got {}: {}", + db_err.code().code(), + db_err.message() + ); + } + } + + // The refused insert left one row under the key. A duplicate that commits + // makes every later point read and keyed DML touch an unbounded row count. + let rows = server + .query_rows("SELECT value FROM metrics_pk WHERE sku = 'a'") + .await + .unwrap(); + assert_eq!( + rows.len(), + 1, + "exactly one row may exist per declared primary key, got: {rows:?}" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn spatial_declared_non_id_primary_key_refuses_duplicate() { + let server = TestServer::start().await; + + server + .exec( + "CREATE COLLECTION places_pk (\ + code TEXT PRIMARY KEY, geom GEOMETRY SPATIAL_INDEX, label TEXT\ + ) WITH (engine='spatial')", + ) + .await + .unwrap(); + + server + .exec( + "INSERT INTO places_pk (code, geom, label) VALUES ('p1', ST_Point(0.0, 0.0), 'origin')", + ) + .await + .unwrap(); + + match server + .client + .simple_query( + "INSERT INTO places_pk (code, geom, label) VALUES ('p1', ST_Point(1.0, 1.0), 'moved')", + ) + .await + { + Ok(_) => panic!("expected unique_violation on the declared primary key, got success"), + Err(e) => { + let db_err = e.as_db_error().expect("expected DbError"); + assert_eq!( + db_err.code().code(), + "23505", + "expected SQLSTATE 23505, got {}: {}", + db_err.code().code(), + db_err.message() + ); + } + } + + let rows = server + .query_rows("SELECT label FROM places_pk WHERE code = 'p1'") + .await + .unwrap(); + assert_eq!( + rows.len(), + 1, + "exactly one row may exist per declared primary key, got: {rows:?}" + ); +} + +/// `ON CONFLICT` resolves against the declared key. The upsert path derives +/// identity through the same helper as the insert path, so it must recognise +/// the same conflict rather than appending a second row. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn columnar_declared_non_id_primary_key_upsert_updates_in_place() { + let server = TestServer::start().await; + + server + .exec( + "CREATE COLLECTION metrics_up (\ + sku TEXT PRIMARY KEY, value FLOAT\ + ) WITH (engine='columnar')", + ) + .await + .unwrap(); + + server + .exec("INSERT INTO metrics_up (sku, value) VALUES ('a', 1.0)") + .await + .unwrap(); + server + .exec( + "INSERT INTO metrics_up (sku, value) VALUES ('a', 7.0) \ + ON CONFLICT (sku) DO UPDATE SET value = EXCLUDED.value", + ) + .await + .unwrap(); + + let rows = server + .query_rows("SELECT value FROM metrics_up WHERE sku = 'a'") + .await + .unwrap(); + assert_eq!( + rows.len(), + 1, + "the upsert must resolve against the declared key, got: {rows:?}" + ); + assert_eq!( + rows[0][0], "7.0", + "expected the EXCLUDED value, got: {:?}", + rows[0] + ); +} diff --git a/nodedb/tests/wire/cases/sql_insert_conflict_columnar/mod.rs b/nodedb/tests/wire/cases/sql_insert_conflict_columnar/mod.rs new file mode 100644 index 000000000..40da6a588 --- /dev/null +++ b/nodedb/tests/wire/cases/sql_insert_conflict_columnar/mod.rs @@ -0,0 +1,20 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! INSERT conflict semantics for columnar-family engines. +//! +//! Columnar storage is OLAP-shaped (append-only segments, zonemap pruning), +//! but `PRIMARY KEY` appears in the ANSI SQL surface and must mean the same +//! thing across every engine NodeDB ships. The resolution is to treat PK on +//! a columnar collection as both a sort key (enforced at segment flush) and +//! a logical uniqueness constraint enforced via a sparse PK index plus +//! positional deletes: duplicate INSERTs tombstone the prior row rather +//! than raising 23505, and readers skip tombstoned row-ids. A `PRIMARY KEY` +//! declared on any column other than `id` or `document_id` refuses a +//! duplicate with 23505 instead. +//! +//! Spatial extends columnar and inherits the same semantics. Timeseries is +//! a different profile (append-only, time-keyed) and is not covered here. + +mod columnar; +mod declared_key; +mod spatial; diff --git a/nodedb/tests/wire/cases/sql_insert_conflict_columnar/spatial.rs b/nodedb/tests/wire/cases/sql_insert_conflict_columnar/spatial.rs new file mode 100644 index 000000000..978b9b47f --- /dev/null +++ b/nodedb/tests/wire/cases/sql_insert_conflict_columnar/spatial.rs @@ -0,0 +1,53 @@ +// SPDX-License-Identifier: BUSL-1.1 + +use crate::harness::TestServer; + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn spatial_insert_duplicate_pk_keeps_latest() { + let server = TestServer::start().await; + + server + .exec( + "CREATE COLLECTION places (\ + id TEXT PRIMARY KEY, geom GEOMETRY SPATIAL_INDEX, label TEXT\ + ) WITH (engine='spatial')", + ) + .await + .unwrap(); + server + .exec("CREATE UNIQUE INDEX places_pk ON places (id)") + .await + .unwrap(); + + server + .exec("INSERT INTO places (id, geom, label) VALUES ('p1', ST_Point(0.0, 0.0), 'origin')") + .await + .unwrap(); + + server + .exec("INSERT INTO places (id, geom, label) VALUES ('p1', ST_Point(1.0, 1.0), 'moved')") + .await + .unwrap(); + + let rows = server + .query_rows("SELECT id, label FROM places WHERE id = 'p1'") + .await + .unwrap(); + + assert_eq!( + rows.len(), + 1, + "spatial duplicate PK must not produce two rows, got: {rows:?}" + ); + // row[0]=id, row[1]=label + assert_eq!( + rows[0][1], "moved", + "expected latest (moved), got: {:?}", + rows[0] + ); + assert_ne!( + rows[0][1], "origin", + "prior row must be tombstoned, got: {:?}", + rows[0] + ); +} From 56871b8c85078c6857fbb0ad56dd15a3b3f1ecfd Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 11 Sep 2026 06:12:05 +0800 Subject: [PATCH 03/17] fix(sql): enforce declared PRIMARY KEY uniqueness on every column MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A duplicate INSERT only raised 23505 when the declared primary key sat on a column other than id/document_id; those two names silently fell back to tombstone-and-reinsert (UPSERT semantics) even without an explicit UPSERT. A declared PRIMARY KEY now means uniqueness on every column it names, id included — convert_insert checks declared_pk.is_some() rather than excluding the id/document_id convention. primary_key threads through ConvertInsertArgs, ConvertUpsertArgs, resolve_doc_identity_with_declared, columnar_row_surrogates and build_schema_bytes as a plain &str instead of Option<&str>: the planner now always resolves one before reaching these converters, and the DML visitor arms reject a missing resolution as a PlanError instead of letting callers guess a fallback. DEFAULT_IDENTITY_COLUMN replaces the "id" literal scattered across schema-building and catalog conversion so the synthetic primary-key name has one definition. resolve_doc_identity and its is_auto_rowid_pk/ declared_primary_key_name helpers move to narrower pub(in ...) visibility now that upsert resolves the declared key once per statement instead of once per row, mirroring convert_insert. Row ids surfaced from change events, array surrogate scans, and vector search/upsert/write now go through the same surrogate_to_doc_id encoding the document store uses for its storage key, instead of ad hoc decimal or "{:08x}" formatting that could disagree with the key a row is actually stored under. Wire tests for columnar and spatial INSERT-conflict now expect 23505 on a duplicate declared-PK insert and confirm the original row is untouched; the prior tombstone/UPSERT-shaped coverage moves to an explicit UPSERT statement, which still merges as before. --- nodedb/src/bootstrap/data_plane.rs | 4 +- .../planner/catalog_adapter/type_convert.rs | 2 +- .../sql_plan_convert/dml/insert/convert.rs | 27 +++--- .../sql_plan_convert/dml/insert/identity.rs | 90 +++++++------------ .../sql_plan_convert/dml/insert/mod.rs | 10 ++- .../sql_plan_convert/dml/insert/schema.rs | 38 ++++---- .../planner/sql_plan_convert/dml/mod.rs | 2 +- .../planner/sql_plan_convert/dml/upsert.rs | 29 ++++-- .../sql_plan_convert/visitor/arms_dml.rs | 12 +++ .../dispatch_utils/change_events/extract.rs | 7 +- .../executor/dispatch/array/surrogate_scan.rs | 3 +- .../handlers/columnar_write/row_ingest.rs | 2 +- .../executor/handlers/vector_search_exec.rs | 4 +- .../data/executor/handlers/vector_upsert.rs | 3 +- .../data/executor/handlers/vector_write.rs | 3 +- .../sql_insert_conflict_columnar/columnar.rs | 51 +++++------ .../cases/sql_insert_conflict_columnar/mod.rs | 14 ++- .../sql_insert_conflict_columnar/spatial.rs | 35 +++++--- 18 files changed, 178 insertions(+), 158 deletions(-) diff --git a/nodedb/src/bootstrap/data_plane.rs b/nodedb/src/bootstrap/data_plane.rs index 9c93f5962..c6c5803ff 100644 --- a/nodedb/src/bootstrap/data_plane.rs +++ b/nodedb/src/bootstrap/data_plane.rs @@ -257,7 +257,9 @@ pub fn load_columnar_schema_seed( .filter_map(|(database_id, coll)| { let schema = crate::control::planner::sql_plan_convert::dml::build_columnar_schema( &coll.fields, - coll.declared_primary_key.as_deref(), + coll.declared_primary_key.as_deref().unwrap_or( + crate::control::planner::sql_plan_convert::dml::DEFAULT_IDENTITY_COLUMN, + ), )?; Some(( database_id, diff --git a/nodedb/src/control/planner/catalog_adapter/type_convert.rs b/nodedb/src/control/planner/catalog_adapter/type_convert.rs index 8aa2a1dc8..7438b2666 100644 --- a/nodedb/src/control/planner/catalog_adapter/type_convert.rs +++ b/nodedb/src/control/planner/catalog_adapter/type_convert.rs @@ -132,7 +132,7 @@ pub(super) fn convert_collection_type( } else { EngineType::Columnar }; - let pk_name = "id"; + let pk_name = crate::control::planner::sql_plan_convert::dml::DEFAULT_IDENTITY_COLUMN; // If the DDL declared its own `id` field, the synthetic primary key // adopts that declared type and is client-supplied — an explicit // `id INT PRIMARY KEY` must stay INT rather than being dropped in diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs index 8f771994b..dc4a3b110 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs @@ -18,7 +18,7 @@ use super::identity::{ use super::schema::build_schema_bytes; /// Bundled arguments for [`convert_insert`]. -pub(crate) struct ConvertInsertArgs<'a> { +pub(in super::super::super) struct ConvertInsertArgs<'a> { pub collection: &'a str, /// The lowering these rows take, decided by `nodedb-sql`. pub route: WriteRoute, @@ -26,12 +26,14 @@ pub(crate) struct ConvertInsertArgs<'a> { pub column_defaults: &'a [(String, String)], pub column_schema: &'a [(String, String)], pub if_absent: bool, - pub primary_key: Option<&'a str>, + pub primary_key: &'a str, pub tenant_id: TenantId, pub ctx: &'a ConvertContext, } -pub(crate) fn convert_insert(args: ConvertInsertArgs<'_>) -> crate::Result> { +pub(in super::super::super) fn convert_insert( + args: ConvertInsertArgs<'_>, +) -> crate::Result> { let ConvertInsertArgs { collection, route, @@ -179,15 +181,13 @@ pub(crate) fn convert_insert(args: ConvertInsertArgs<'_>) -> crate::Result) -> crate::Result) -> DocId { - match row.iter().find(|(k, _)| match primary_key { - Some(pk) => k == pk, - None => k == "id" || k == "document_id" || k == "key", - }) { +/// Extract the document-id value from a row, keyed off `primary_key`. +pub(super) fn extract_doc_id(row: &[(String, SqlValue)], primary_key: &str) -> DocId { + match row.iter().find(|(k, _)| k == primary_key) { Some((_, SqlValue::Null)) => DocId::ExplicitNull, Some((_, v)) => DocId::Present(sql_value_to_string(v)), None => DocId::Absent, @@ -36,7 +31,7 @@ pub(super) fn extract_doc_id(row: &[(String, SqlValue)], primary_key: Option<&st /// catalog's `declared_primary_key` is set only by the keyword itself, and /// names the column the keyword applied `NOT NULL` to. A catalog miss reads /// as not declared — nothing to enforce. -pub(crate) fn declared_primary_key_name( +pub(in super::super::super) fn declared_primary_key_name( ctx: &ConvertContext, collection: &str, ) -> crate::Result> { @@ -48,52 +43,31 @@ pub(crate) fn declared_primary_key_name( .declared_primary_key(ctx.database_id, ctx.tenant_id.as_u64(), collection) } -/// Resolve a row's document id + surrogate, refusing a `NULL`/omitted -/// declared primary key first: a declared `PRIMARY KEY` implies `NOT NULL`. +/// Resolve a row's document id and surrogate, refusing a NULL or omitted +/// declared primary key first. A declared `PRIMARY KEY` implies `NOT NULL`. /// /// Enforcement keys on the DDL-declared column, not the resolved -/// `primary_key`: those diverge whenever a natural key sits on a column -/// other than `id` (e.g. `metrics (sku TEXT PRIMARY KEY)` resolves -/// `primary_key` to `id` but declares `sku`). `_rowid` carries no -/// declaration, so it skips the check and mints a surrogate. +/// `primary_key`. The two diverge whenever a natural key sits on a column +/// other than `id`. `metrics (sku TEXT PRIMARY KEY)` resolves `primary_key` +/// to `id` and declares `sku`. /// -/// Identity minting then runs on the same declared name. A present declared -/// key content-addresses a surrogate via [`assign_for_pk`]. With no declared -/// key, minting falls back to the resolved `primary_key` argument. An -/// auto-`_rowid` pk or a missing/null key mints a fresh surrogate instead. -/// The two steps are one call so no caller can mint an identity without the -/// NOT NULL check running first. -pub(crate) fn resolve_doc_identity( - ctx: &ConvertContext, - collection: &str, - primary_key: Option<&str>, - row: &[(String, SqlValue)], -) -> crate::Result<(String, Surrogate)> { - // `_rowid` carries no declaration: skip the catalog read entirely and - // mint a surrogate below. - let declared = if is_auto_rowid_pk(primary_key) { - None - } else { - declared_primary_key_name(ctx, collection)? - }; - resolve_doc_identity_with_declared(ctx, collection, primary_key, declared.as_deref(), row) -} - -/// Core of [`resolve_doc_identity`], given the declared PK name already -/// resolved by the caller instead of reading it from the catalog here. +/// Identity minting runs on the same declared name. A present declared key +/// content-addresses a surrogate via [`assign_for_pk`]. With no declared key, +/// minting uses the resolved `primary_key`. An auto-`_rowid` pk or a missing +/// key mints a fresh surrogate. /// -/// [`columnar_row_surrogates`] resolves `declared` once for the whole -/// statement and calls this directly for every row, so a multi-row INSERT -/// hits the catalog once rather than once per row. -pub(super) fn resolve_doc_identity_with_declared( +/// The caller resolves `declared` once per statement, so a multi-row INSERT +/// reads the catalog once rather than once per row. Both steps are one call, +/// so no caller mints an identity without the NOT NULL check running first. +pub(in super::super) fn resolve_doc_identity_with_declared( ctx: &ConvertContext, collection: &str, - primary_key: Option<&str>, + primary_key: &str, declared: Option<&str>, row: &[(String, SqlValue)], ) -> crate::Result<(String, Surrogate)> { if let Some(declared) = declared { - match extract_doc_id(row, Some(declared)) { + match extract_doc_id(row, declared) { DocId::Present(_) => {} DocId::ExplicitNull | DocId::Absent => { return Err(crate::Error::RejectedConstraint { @@ -113,7 +87,7 @@ pub(super) fn resolve_doc_identity_with_declared( )?; return Ok((pk, s)); } - let mint_key = declared.or(primary_key); + let mint_key: &str = declared.unwrap_or(primary_key); match extract_doc_id(row, mint_key) { DocId::Present(id) => { let s = assign_for_pk(ctx, collection, id.as_bytes())?; @@ -130,7 +104,7 @@ pub(super) fn resolve_doc_identity_with_declared( } } -pub(crate) fn assign_for_pk( +pub(in super::super) fn assign_for_pk( ctx: &ConvertContext, collection: &str, pk_bytes: &[u8], @@ -157,27 +131,27 @@ pub(super) fn assign_fresh( /// Whether a collection's declared primary key is the auto-generated `_rowid` /// sentinel — injected by strict-schema construction when no `PRIMARY KEY` was /// declared. Such rows carry no user identity: each needs a fresh surrogate. -pub(super) fn is_auto_rowid_pk(primary_key: Option<&str>) -> bool { - primary_key == Some("_rowid") +pub(in super::super) fn is_auto_rowid_pk(primary_key: &str) -> bool { + primary_key == "_rowid" } -/// Mirrors the document-engine identity path (`resolve_doc_identity`) for -/// columnar/spatial rows. The declared `primary_key` — not the legacy -/// `id`/`document_id`/`key` name guess — determines each row's identity, so a -/// natural key on any column (e.g. `sku`) gets its own surrogate. A -/// missing/empty key mints a fresh unique surrogate rather than collapsing -/// onto `Surrogate::ZERO`, which would silently merge distinct rows. +/// Mirrors the document-engine identity path for columnar and spatial rows. +/// +/// The declared primary key determines each row's identity, so a natural key +/// on any column gets its own surrogate. A missing or empty key mints a fresh +/// surrogate rather than collapsing onto `Surrogate::ZERO`, which merges +/// distinct rows. /// /// `declared_pk` is the catalog's declared PK name, resolved once by the /// caller for the whole statement — not re-read here per row. -pub(crate) fn columnar_row_surrogates( +pub(in super::super) fn columnar_row_surrogates( ctx: &ConvertContext, collection: &str, columnar_rows: &[&Vec<(String, SqlValue)>], - primary_key: Option<&str>, + primary_key: &str, declared_pk: Option<&str>, ) -> crate::Result> { - // `_rowid` carries no declaration, matching `resolve_doc_identity`. + // `_rowid` carries no declaration. let declared = if is_auto_rowid_pk(primary_key) { None } else { diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert/mod.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert/mod.rs index c7ac18de0..acbc2dc53 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/insert/mod.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert/mod.rs @@ -4,10 +4,12 @@ mod convert; mod identity; mod schema; -pub(crate) use schema::build_columnar_schema; pub(super) use schema::build_schema_bytes; +pub(crate) use schema::{DEFAULT_IDENTITY_COLUMN, build_columnar_schema}; -pub(crate) use identity::declared_primary_key_name; -pub(super) use identity::{assign_for_pk, columnar_row_surrogates, resolve_doc_identity}; +pub(in super::super) use identity::declared_primary_key_name; +pub(super) use identity::{ + assign_for_pk, columnar_row_surrogates, is_auto_rowid_pk, resolve_doc_identity_with_declared, +}; -pub(crate) use convert::{ConvertInsertArgs, convert_insert}; +pub(in super::super) use convert::{ConvertInsertArgs, convert_insert}; diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert/schema.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert/schema.rs index 4540d68de..a132779a8 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/insert/schema.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert/schema.rs @@ -8,14 +8,14 @@ use nodedb_types::columnar::{ColumnDef, ColumnType, ColumnarSchema}; /// DDL catalog (`stored.fields`). Unknown type strings are treated as /// `ColumnType::String` (matching the memtable's existing fallback). /// -/// `declared_pk` names the collection's DDL-declared `PRIMARY KEY` column, -/// when one was declared (see [`declared_primary_key_name`]). A column -/// matching that name is the schema primary key. When `declared_pk` is -/// `None`, the legacy `id` / `document_id` name convention applies instead. +/// `identity_column` names the column that carries the row's identity: the +/// DDL-declared `PRIMARY KEY` when one exists, else the engine's resolved +/// primary key. The column of that name is the schema primary key. When no +/// column carries that name, a required `String` column is synthesized under +/// it. /// -/// Returns `None` when `column_schema` is empty (no catalog schema available -/// — test fixtures and legacy paths) or the resulting schema fails -/// validation. +/// Returns `None` when `column_schema` is empty, meaning no catalog schema is +/// available, or when the resulting schema fails validation. /// /// This is the single source of truth for turning a catalog's raw /// `(name, type_str)` field list into a typed `ColumnarSchema` — shared by @@ -23,9 +23,13 @@ use nodedb_types::columnar::{ColumnDef, ColumnType, ColumnarSchema}; /// `bootstrap::data_plane::load_columnar_schema_seed`, which pre-registers /// each columnar-family collection's real schema before WAL replay so a /// fresh `MutationEngine` never falls back to type-lossy inference. +/// The identity column a columnar-family collection carries when its DDL +/// declares no `PRIMARY KEY`. The planner resolves the same name. +pub(crate) const DEFAULT_IDENTITY_COLUMN: &str = "id"; + pub(crate) fn build_columnar_schema( column_schema: &[(String, String)], - declared_pk: Option<&str>, + identity_column: &str, ) -> Option { if column_schema.is_empty() { return None; @@ -43,22 +47,18 @@ pub(crate) fn build_columnar_schema( let col_type = bare_type .parse::() .unwrap_or(ColumnType::String); - let is_id = match declared_pk { - Some(pk) => name == pk, - None => name == "id" || name == "document_id", - }; - if is_id { + if name == identity_column { has_id = true; cols.push(ColumnDef::required(name.clone(), col_type).with_primary_key()); } else { cols.push(ColumnDef::nullable(name.clone(), col_type)); } } - // If no PK column found in stored.fields, inject a synthetic one. + // No column carries the identity: synthesize it. if !has_id { cols.insert( 0, - ColumnDef::required("id", ColumnType::String).with_primary_key(), + ColumnDef::required(identity_column, ColumnType::String).with_primary_key(), ); } ColumnarSchema::new(cols).ok() @@ -67,15 +67,15 @@ pub(crate) fn build_columnar_schema( /// Build a `ColumnarSchema` from raw catalog column-type strings, then /// serialize it as MessagePack for the `ColumnarOp::Insert::schema_bytes` field. /// -/// `declared_pk` is forwarded to [`build_columnar_schema`] unchanged. +/// `identity_column` is forwarded to [`build_columnar_schema`] unchanged. /// /// Returns an empty `Vec` when `column_schema` is empty or fails validation /// — see [`build_columnar_schema`] for the typed builder this wraps. -pub(crate) fn build_schema_bytes( +pub(in super::super) fn build_schema_bytes( column_schema: &[(String, String)], - declared_pk: Option<&str>, + identity_column: &str, ) -> Vec { - build_columnar_schema(column_schema, declared_pk) + build_columnar_schema(column_schema, identity_column) .map(|schema| zerompk::to_msgpack_vec(&schema).unwrap_or_default()) .unwrap_or_default() } diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/mod.rs b/nodedb/src/control/planner/sql_plan_convert/dml/mod.rs index d87cf86b8..42d087059 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/mod.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/mod.rs @@ -8,8 +8,8 @@ mod merge; mod update_delete; mod upsert; -pub(crate) use insert::build_columnar_schema; pub(super) use insert::{ConvertInsertArgs, convert_insert, declared_primary_key_name}; +pub(crate) use insert::{DEFAULT_IDENTITY_COLUMN, build_columnar_schema}; pub(super) use kv_and_vector::{ VectorPrimaryInsertCfg, convert_kv_insert, convert_vector_primary_insert, }; diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs b/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs index 05f980231..70a5d679e 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs @@ -3,8 +3,8 @@ //! `UPSERT` / `INSERT ... ON CONFLICT DO UPDATE` lowering. //! //! Split from `insert.rs`, which lowers plain `INSERT`. The two share the row -//! identity helper there (`resolve_doc_identity`) so a row's surrogate is -//! derived identically whichever statement wrote it. +//! identity helper there (`resolve_doc_identity_with_declared`) so a row's +//! surrogate is derived identically whichever statement wrote it. use nodedb_sql::types::{SqlExpr, SqlValue, WriteRoute}; @@ -18,7 +18,8 @@ use super::super::value::{ assignments_to_update_values, expand_row_defaults, row_to_msgpack, rows_to_msgpack_array, }; use super::insert::{ - build_schema_bytes, columnar_row_surrogates, declared_primary_key_name, resolve_doc_identity, + build_schema_bytes, columnar_row_surrogates, declared_primary_key_name, is_auto_rowid_pk, + resolve_doc_identity_with_declared, }; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; @@ -31,7 +32,7 @@ pub(in super::super) struct ConvertUpsertArgs<'a> { pub column_defaults: &'a [(String, String)], pub column_schema: &'a [(String, String)], pub on_conflict_updates: &'a [(String, SqlExpr)], - pub primary_key: Option<&'a str>, + pub primary_key: &'a str, pub tenant_id: TenantId, pub ctx: &'a ConvertContext, } @@ -83,11 +84,25 @@ pub(in super::super) fn convert_upsert( // declaration promises — see `expand_row_defaults`. let expanded_rows = expand_row_defaults(rows, column_defaults, tenant_id, ctx)?; + // One catalog read for the whole statement, mirroring `convert_insert`. + // `_rowid` carries no declaration, so it skips the read. + let declared_pk = if is_auto_rowid_pk(primary_key) { + None + } else { + declared_primary_key_name(ctx, collection)? + }; + for row in &expanded_rows { match route { WriteRoute::Document => { let value_bytes = row_to_msgpack(row)?; - let (doc_id, surrogate) = resolve_doc_identity(ctx, collection, primary_key, row)?; + let (doc_id, surrogate) = resolve_doc_identity_with_declared( + ctx, + collection, + primary_key, + declared_pk.as_deref(), + row, + )?; let plan = if is_crdt { PhysicalPlan::Crdt(CrdtOp::DocUpsert { collection: qualified_collection.clone(), @@ -132,7 +147,6 @@ pub(in super::super) fn convert_upsert( if !columnar_rows.is_empty() { let payload = rows_to_msgpack_array(&columnar_rows)?; - let declared_pk = declared_primary_key_name(ctx, collection)?; let surrogates = columnar_row_surrogates( ctx, collection, @@ -140,7 +154,8 @@ pub(in super::super) fn convert_upsert( primary_key, declared_pk.as_deref(), )?; - let schema_bytes = build_schema_bytes(column_schema, declared_pk.as_deref()); + let schema_bytes = + build_schema_bytes(column_schema, declared_pk.as_deref().unwrap_or(primary_key)); tasks.push(PhysicalTask { tenant_id, vshard_id: vshard, diff --git a/nodedb/src/control/planner/sql_plan_convert/visitor/arms_dml.rs b/nodedb/src/control/planner/sql_plan_convert/visitor/arms_dml.rs index 90dffec3d..5b16ac971 100644 --- a/nodedb/src/control/planner/sql_plan_convert/visitor/arms_dml.rs +++ b/nodedb/src/control/planner/sql_plan_convert/visitor/arms_dml.rs @@ -18,6 +18,12 @@ macro_rules! impl_dml_arms_for_convert_visitor { column_schema, primary_key, } = args; + let primary_key = primary_key.ok_or_else(|| crate::Error::PlanError { + detail: format!( + "insert converter reached collection '{collection}' with no resolved \ + primary key" + ), + })?; super::super::dml::convert_insert(super::super::dml::ConvertInsertArgs { collection, route, @@ -45,6 +51,12 @@ macro_rules! impl_dml_arms_for_convert_visitor { column_schema, primary_key, } = args; + let primary_key = primary_key.ok_or_else(|| crate::Error::PlanError { + detail: format!( + "upsert converter reached collection '{collection}' with no resolved \ + primary key" + ), + })?; super::super::dml::convert_upsert(super::super::dml::ConvertUpsertArgs { collection, route, diff --git a/nodedb/src/control/server/dispatch_utils/change_events/extract.rs b/nodedb/src/control/server/dispatch_utils/change_events/extract.rs index 3a3b05e41..81272e7d7 100644 --- a/nodedb/src/control/server/dispatch_utils/change_events/extract.rs +++ b/nodedb/src/control/server/dispatch_utils/change_events/extract.rs @@ -5,6 +5,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::change_stream::ChangeOperation; +use crate::engine::document::store::surrogate_to_doc_id; use crate::types::TenantId; use nodedb_physical::physical_plan::{ ArrayOp, ClusterArrayOp, ColumnarOp, CrdtOp, DocumentOp, DocumentResolvedMutation, KvOp, @@ -283,7 +284,7 @@ pub(super) fn extract_write_metadata( .. }) => vec![( collection.to_string(), - surrogate.as_u32().to_string(), + surrogate_to_doc_id(*surrogate), ChangeOperation::Insert, )], PhysicalPlan::Vector(_) => Vec::new(), @@ -633,11 +634,13 @@ mod tests { rls_filters: Vec::new(), }); let meta = extract_write_metadata(&plan, TenantId::new(1)); + // The row id is the document storage key, so a consumer can address + // the row the event describes. assert_eq!( meta, vec![( "embeddings".to_string(), - "42".to_string(), + "0000002a".to_string(), ChangeOperation::Insert )] ); diff --git a/nodedb/src/data/executor/dispatch/array/surrogate_scan.rs b/nodedb/src/data/executor/dispatch/array/surrogate_scan.rs index e760ebf5f..5e2601f98 100644 --- a/nodedb/src/data/executor/dispatch/array/surrogate_scan.rs +++ b/nodedb/src/data/executor/dispatch/array/surrogate_scan.rs @@ -18,6 +18,7 @@ use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::response_codec::encode_raw_document_rows; use crate::data::executor::task::ExecutionTask; +use crate::engine::document::store::surrogate_to_doc_id; impl CoreLoop { pub(in crate::data::executor) fn dispatch_array_surrogate_bitmap_scan( @@ -98,7 +99,7 @@ impl CoreLoop { if sur.as_u32() == 0 { continue; } - let hex = format!("{:08x}", sur.as_u32()); + let hex = surrogate_to_doc_id(*sur); // Empty msgpack map as the row body — the consumer // (`collect_surrogates`) only reads `id`. rows.push((hex, vec![0x80])); diff --git a/nodedb/src/data/executor/handlers/columnar_write/row_ingest.rs b/nodedb/src/data/executor/handlers/columnar_write/row_ingest.rs index d3cbab52c..3df97e8ac 100644 --- a/nodedb/src/data/executor/handlers/columnar_write/row_ingest.rs +++ b/nodedb/src/data/executor/handlers/columnar_write/row_ingest.rs @@ -263,7 +263,7 @@ impl CoreLoop { .iter() .zip(final_values.iter()) .filter(|(col, _)| col.primary_key) - .map(|(col, v)| format!("{}={v:?}", col.name)) + .map(|(col, v)| format!("{}={v}", col.name)) .collect::>() .join(", "); return Err(self.response_error( diff --git a/nodedb/src/data/executor/handlers/vector_search_exec.rs b/nodedb/src/data/executor/handlers/vector_search_exec.rs index 0cd26d262..d0f0aa513 100644 --- a/nodedb/src/data/executor/handlers/vector_search_exec.rs +++ b/nodedb/src/data/executor/handlers/vector_search_exec.rs @@ -13,6 +13,8 @@ use super::vector_search_ann::{ResolvedAnnOptions, apply_ann_options, quantizati use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; +use crate::engine::document::store::surrogate_to_doc_id; +use nodedb_types::Surrogate; /// Parameters for [`CoreLoop::search_ivf`]. struct SearchIvfParams<'a> { @@ -53,7 +55,7 @@ impl CoreLoop { if !attach { return hit; } - let hex = format!("{:08x}", hit.id); + let hex = surrogate_to_doc_id(Surrogate::new(hit.id)); if let Ok(Some(bytes)) = self.sparse.get(database_id, tid, collection, &hex) { let format = self.sparse_body_format( crate::types::DatabaseId::new(database_id), diff --git a/nodedb/src/data/executor/handlers/vector_upsert.rs b/nodedb/src/data/executor/handlers/vector_upsert.rs index 943cfd0ce..2ff41c826 100644 --- a/nodedb/src/data/executor/handlers/vector_upsert.rs +++ b/nodedb/src/data/executor/handlers/vector_upsert.rs @@ -24,6 +24,7 @@ use tracing::debug; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; +use crate::engine::document::store::surrogate_to_doc_id; /// Decode MessagePack payload bytes into `HashMap` and /// lower-case all field names so bitmap inserts agree with SELECT @@ -214,7 +215,7 @@ impl CoreLoop { // row scannable at all, so skipping it made such a row invisible to // `SELECT *` while every other path still counted it as stored. An // empty tagged map is the honest sidecar for "no non-vector columns". - let row_key = format!("{:08x}", surrogate.as_u32()); + let row_key = surrogate_to_doc_id(surrogate); let sidecar: std::borrow::Cow<'_, [u8]> = if payload.is_empty() { match zerompk::to_msgpack_vec(&HashMap::::new()) { Ok(bytes) => std::borrow::Cow::Owned(bytes), diff --git a/nodedb/src/data/executor/handlers/vector_write.rs b/nodedb/src/data/executor/handlers/vector_write.rs index 577875b28..bf1f0b767 100644 --- a/nodedb/src/data/executor/handlers/vector_write.rs +++ b/nodedb/src/data/executor/handlers/vector_write.rs @@ -11,6 +11,7 @@ use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::vector_upsert::decode_payload_lowercased; use crate::data::executor::task::ExecutionTask; +use crate::engine::document::store::surrogate_to_doc_id; use crate::types::TenantId; use nodedb_types::DatabaseId; @@ -139,7 +140,7 @@ impl CoreLoop { .and_then(|c| c.get_surrogate(vector_id)); if let Some(surrogate) = surrogate_opt { - let row_key = format!("{:08x}", surrogate.as_u32()); + let row_key = surrogate_to_doc_id(surrogate); let fields = match self .sparse diff --git a/nodedb/tests/wire/cases/sql_insert_conflict_columnar/columnar.rs b/nodedb/tests/wire/cases/sql_insert_conflict_columnar/columnar.rs index 15b20b03d..1efd86dec 100644 --- a/nodedb/tests/wire/cases/sql_insert_conflict_columnar/columnar.rs +++ b/nodedb/tests/wire/cases/sql_insert_conflict_columnar/columnar.rs @@ -5,7 +5,7 @@ use crate::harness::TestServer; // ── Plain columnar ────────────────────────────────────────────────────────── #[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn columnar_insert_duplicate_pk_keeps_latest() { +async fn columnar_insert_duplicate_pk_refuses_with_23505() { let server = TestServer::start().await; server @@ -26,45 +26,43 @@ async fn columnar_insert_duplicate_pk_keeps_latest() { .await .unwrap(); - // Duplicate PK — must NOT raise 23505 on columnar (OLAP-shaped), but - // must also NOT produce two visible rows (the silent-duplicate bug). - server - .exec("INSERT INTO metrics (id, region, value) VALUES ('m1', 'us-west', 2.0)") + // A declared PRIMARY KEY means uniqueness on every column, `id` included. + match server + .client + .simple_query("INSERT INTO metrics (id, region, value) VALUES ('m1', 'us-west', 2.0)") .await - .unwrap(); + { + Ok(_) => panic!("expected unique_violation on the declared primary key, got success"), + Err(e) => { + let db_err = e.as_db_error().expect("expected DbError"); + assert_eq!( + db_err.code().code(), + "23505", + "expected SQLSTATE 23505, got {}: {}", + db_err.code().code(), + db_err.message() + ); + } + } let rows = server .query_rows("SELECT id, region, value FROM metrics WHERE id = 'm1'") .await .unwrap(); - - // Regression guard: the original bug was two rows visible for one PK. assert_eq!( rows.len(), 1, - "duplicate PK must not produce two visible rows, got: {rows:?}" + "the refused insert must leave exactly one row, got: {rows:?}" ); - - // Latest-write-wins on the tombstoned prior row. row[0]=id, row[1]=region, row[2]=value. assert_eq!( - rows[0][1], "us-west", - "expected latest write (us-west), got: {:?}", - rows[0] - ); - assert!( - rows[0][2].contains('2'), - "expected value 2.0, got: {:?}", - rows[0] - ); - assert_ne!( rows[0][1], "us-east", - "prior row must be tombstoned, got: {:?}", + "the original row must be unchanged, got: {:?}", rows[0] ); } #[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn columnar_full_scan_hides_tombstoned_duplicate() { +async fn columnar_full_scan_hides_tombstoned_upsert() { let server = TestServer::start().await; server @@ -80,8 +78,11 @@ async fn columnar_full_scan_hides_tombstoned_duplicate() { .exec("INSERT INTO m (id, v) VALUES ('a', 1), ('b', 2), ('c', 3)") .await .unwrap(); + // UPSERT merges into the existing row rather than refusing it, and does + // so by tombstoning the prior row and writing a new one. A full scan + // must hide the tombstoned row and expose only the latest one. server - .exec("INSERT INTO m (id, v) VALUES ('b', 20)") + .exec("UPSERT INTO m (id, v) VALUES ('b', 20)") .await .unwrap(); @@ -93,7 +94,7 @@ async fn columnar_full_scan_hides_tombstoned_duplicate() { assert_eq!( rows.len(), 3, - "full scan must return 3 rows after dup, got: {rows:?}" + "full scan must return 3 rows after upsert, got: {rows:?}" ); // ORDER BY id → a, b, c with latest-wins on b. row[0]=id, row[1]=v. assert_eq!( diff --git a/nodedb/tests/wire/cases/sql_insert_conflict_columnar/mod.rs b/nodedb/tests/wire/cases/sql_insert_conflict_columnar/mod.rs index 40da6a588..d5be33b61 100644 --- a/nodedb/tests/wire/cases/sql_insert_conflict_columnar/mod.rs +++ b/nodedb/tests/wire/cases/sql_insert_conflict_columnar/mod.rs @@ -2,15 +2,11 @@ //! INSERT conflict semantics for columnar-family engines. //! -//! Columnar storage is OLAP-shaped (append-only segments, zonemap pruning), -//! but `PRIMARY KEY` appears in the ANSI SQL surface and must mean the same -//! thing across every engine NodeDB ships. The resolution is to treat PK on -//! a columnar collection as both a sort key (enforced at segment flush) and -//! a logical uniqueness constraint enforced via a sparse PK index plus -//! positional deletes: duplicate INSERTs tombstone the prior row rather -//! than raising 23505, and readers skip tombstoned row-ids. A `PRIMARY KEY` -//! declared on any column other than `id` or `document_id` refuses a -//! duplicate with 23505 instead. +//! A declared `PRIMARY KEY` means uniqueness, on every engine and every +//! column. A duplicate raises 23505 whatever the column is named. An +//! explicit `UPSERT` or `ON CONFLICT DO UPDATE` still merges into the +//! existing row instead of refusing it. A collection with no declared +//! primary key mints a fresh identity per row and never dedups. //! //! Spatial extends columnar and inherits the same semantics. Timeseries is //! a different profile (append-only, time-keyed) and is not covered here. diff --git a/nodedb/tests/wire/cases/sql_insert_conflict_columnar/spatial.rs b/nodedb/tests/wire/cases/sql_insert_conflict_columnar/spatial.rs index 978b9b47f..709552dc4 100644 --- a/nodedb/tests/wire/cases/sql_insert_conflict_columnar/spatial.rs +++ b/nodedb/tests/wire/cases/sql_insert_conflict_columnar/spatial.rs @@ -3,7 +3,7 @@ use crate::harness::TestServer; #[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn spatial_insert_duplicate_pk_keeps_latest() { +async fn spatial_insert_duplicate_pk_refuses_with_23505() { let server = TestServer::start().await; server @@ -24,30 +24,39 @@ async fn spatial_insert_duplicate_pk_keeps_latest() { .await .unwrap(); - server - .exec("INSERT INTO places (id, geom, label) VALUES ('p1', ST_Point(1.0, 1.0), 'moved')") + // A declared PRIMARY KEY means uniqueness on every column, `id` included. + match server + .client + .simple_query( + "INSERT INTO places (id, geom, label) VALUES ('p1', ST_Point(1.0, 1.0), 'moved')", + ) .await - .unwrap(); + { + Ok(_) => panic!("expected unique_violation on the declared primary key, got success"), + Err(e) => { + let db_err = e.as_db_error().expect("expected DbError"); + assert_eq!( + db_err.code().code(), + "23505", + "expected SQLSTATE 23505, got {}: {}", + db_err.code().code(), + db_err.message() + ); + } + } let rows = server .query_rows("SELECT id, label FROM places WHERE id = 'p1'") .await .unwrap(); - assert_eq!( rows.len(), 1, - "spatial duplicate PK must not produce two rows, got: {rows:?}" + "the refused insert must leave exactly one row, got: {rows:?}" ); - // row[0]=id, row[1]=label assert_eq!( - rows[0][1], "moved", - "expected latest (moved), got: {:?}", - rows[0] - ); - assert_ne!( rows[0][1], "origin", - "prior row must be tombstoned, got: {:?}", + "the original row must be unchanged, got: {:?}", rows[0] ); } From 3efbf033b82af02f5df4637711ccd64d3be52d7d Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 11 Sep 2026 13:54:08 +0800 Subject: [PATCH 04/17] fix(document): derive a row's client-visible identity from its storage key A document's storage key (fixed-width hex) and its client-visible identity (decimal surrogate, or a user/DDL-declared primary key) were conflated across the codebase: change events, array surrogate scans, and vector search/upsert/write formatted or compared identities with ad hoc decimal or "{:08x}" logic that could disagree with the key a row is actually stored under. StorageKey and RowIdentity in engine::document::store::key are now the only two representations: StorageKey wraps a surrogate as the redb key it is stored under, RowIdentity is what a client sees. identity_of(doc_id) is the single place that reinterprets a storage key's hex shape into a RowIdentity, falling back to the raw string for a non-minted (user-supplied or declared) key. surrogate_to_doc_id and doc_id_to_surrogate become thin wrappers over the new types so the 200+ existing call sites that hold a plain String keep working. assign_fresh drops the FreshSurrogateKind parameter now that every fresh row binds through the same RowIdentity convention, removing the enum and its call sites in nodedb-physical and the surrogate assigner. ConvertCollection's MetaOp gains a source_storage_mode field, resolved from the catalog before the DDL mutates collection_type, so the Data Plane decodes rows being converted the same way the collection's own register path would; convert_collection also re-registers the Data Plane's doc_configs entry after a successful conversion so later reads see the new storage mode without a restart. merge_orchestrated's apply module splits into apply/{mod,orchestrate, insert_rows,update_rows}.rs, and the sql_typeguard_defaults wire test splits into sql_typeguard_defaults/{mod,convert,defaults,validate}.rs, which gains coverage confirming a strict-mode conversion preserves a minted row's identity. --- nodedb-physical/src/lib.rs | 2 +- nodedb-physical/src/physical_plan/meta.rs | 7 + nodedb-physical/src/surrogate.rs | 19 +- .../control/planner/rls_injection/context.rs | 17 +- .../control/planner/rls_injection/document.rs | 15 +- .../planner/sql_plan_convert/convert.rs | 28 +- .../sql_plan_convert/dml/insert/identity.rs | 17 +- .../sql_plan_convert/scan/timeseries.rs | 5 +- .../control/server/response_shape/project.rs | 15 +- .../server/response_translate/text_hybrid.rs | 18 +- .../shared/ddl/neutral/convert/driver.rs | 31 ++ .../predicate/txn_buffering/classify.rs | 1 + .../surrogate/assign/core/assign_ops.rs | 36 +- .../src/control/surrogate/assign/core/mod.rs | 1 - nodedb/src/control/surrogate/assign/mod.rs | 1 - nodedb/src/control/surrogate/mod.rs | 1 - nodedb/src/control/surrogate/physical_impl.rs | 7 +- .../control/target_identity/document_id.rs | 15 +- .../src/control/target_identity/surrogate.rs | 11 +- .../src/data/executor/core_loop/deferred.rs | 5 +- .../src/data/executor/core_loop/event_emit.rs | 52 +- .../data/executor/core_loop/filter_match.rs | 12 +- nodedb/src/data/executor/dispatch/meta.rs | 2 + .../data/executor/handlers/bulk_dml/delete.rs | 18 +- .../data/executor/handlers/bulk_dml/scan.rs | 6 +- .../data/executor/handlers/bulk_dml/update.rs | 17 +- .../handlers/bulk_dml/update_project.rs | 24 +- .../control/calvin_overlay_stage_bulk.rs | 6 +- .../executor/handlers/control/crdt_doc.rs | 25 +- .../handlers/control/crdt_materialize.rs | 9 +- .../handlers/control/range_scan_versioned.rs | 13 +- nodedb/src/data/executor/handlers/convert.rs | 229 ++++++-- .../executor/handlers/document/read/scan.rs | 7 +- .../handlers/document/resolve/apply.rs | 4 +- .../handlers/document/resolve/apply_row.rs | 20 +- .../handlers/document/resolve/bulk.rs | 17 +- .../handlers/document/resolve/context.rs | 4 +- .../handlers/document/resolve/point.rs | 13 +- .../handlers/document/resolve/upsert.rs | 17 +- .../handlers/document/write/batch_insert.rs | 13 +- .../src/data/executor/handlers/graph_rag.rs | 17 +- .../executor/handlers/graph_rag_triple.rs | 4 +- .../src/data/executor/handlers/kv/atomic.rs | 8 +- .../data/executor/handlers/kv/crud/delete.rs | 2 +- .../executor/handlers/kv/crud/write_basic.rs | 13 +- .../executor/handlers/kv/crud/write_upsert.rs | 2 +- .../executor/handlers/kv/predicate/apply.rs | 2 +- .../executor/handlers/kv/resolve/apply.rs | 4 +- nodedb/src/data/executor/handlers/kv/rls.rs | 7 +- .../src/data/executor/handlers/kv/transfer.rs | 8 +- .../handlers/merge_orchestrated/apply.rs | 513 ------------------ .../merge_orchestrated/apply/insert_rows.rs | 198 +++++++ .../handlers/merge_orchestrated/apply/mod.rs | 17 + .../merge_orchestrated/apply/orchestrate.rs | 278 ++++++++++ .../merge_orchestrated/apply/update_rows.rs | 246 +++++++++ .../merge_orchestrated/apply_support.rs | 34 +- .../merge_orchestrated/delete_arms.rs | 9 +- .../handlers/merge_orchestrated/plan.rs | 23 +- .../data/executor/handlers/point/delete.rs | 18 +- .../src/data/executor/handlers/point/get.rs | 2 +- .../data/executor/handlers/point/insert.rs | 19 +- .../src/data/executor/handlers/point/put.rs | 12 +- .../executor/handlers/point/update/exec.rs | 12 +- .../data/executor/handlers/returning_doc.rs | 20 +- .../data/executor/handlers/returning_rows.rs | 7 +- .../data/executor/handlers/rls_write_gate.rs | 11 +- .../executor/handlers/transaction/batch.rs | 10 +- .../handlers/transaction/overlay/merge.rs | 19 +- .../transaction/stage_write/dispatch.rs | 4 +- .../stage_write/stage_bulk_delete.rs | 3 +- .../stage_write/stage_bulk_update.rs | 3 +- .../stage_write/stage_kv_atomic.rs | 2 +- .../stage_write/stage_kv_conflict.rs | 2 +- .../stage_write/stage_kv_delete.rs | 4 +- .../stage_write/stage_point_document.rs | 14 +- .../transaction/stage_write/stage_upsert.rs | 2 +- nodedb/src/data/executor/handlers/truncate.rs | 7 +- .../handlers/update_from_join_collect.rs | 3 +- .../handlers/update_from_join_write.rs | 15 +- .../executor/handlers/upsert/exec/insert.rs | 19 +- .../handlers/upsert/exec/overwrite.rs | 10 +- .../src/data/executor/handlers/write_batch.rs | 14 +- nodedb/src/data/executor/row_shape.rs | 16 +- .../src/data/executor/strict_format/decode.rs | 7 +- nodedb/src/engine/document/store/key.rs | 197 ++++++- nodedb/src/engine/document/store/mod.rs | 2 +- .../wire/cases/sql_typeguard_defaults.rs | 458 ---------------- .../cases/sql_typeguard_defaults/convert.rs | 229 ++++++++ .../cases/sql_typeguard_defaults/defaults.rs | 231 ++++++++ .../wire/cases/sql_typeguard_defaults/mod.rs | 14 + .../cases/sql_typeguard_defaults/validate.rs | 131 +++++ 91 files changed, 2254 insertions(+), 1408 deletions(-) delete mode 100644 nodedb/src/data/executor/handlers/merge_orchestrated/apply.rs create mode 100644 nodedb/src/data/executor/handlers/merge_orchestrated/apply/insert_rows.rs create mode 100644 nodedb/src/data/executor/handlers/merge_orchestrated/apply/mod.rs create mode 100644 nodedb/src/data/executor/handlers/merge_orchestrated/apply/orchestrate.rs create mode 100644 nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs delete mode 100644 nodedb/tests/wire/cases/sql_typeguard_defaults.rs create mode 100644 nodedb/tests/wire/cases/sql_typeguard_defaults/convert.rs create mode 100644 nodedb/tests/wire/cases/sql_typeguard_defaults/defaults.rs create mode 100644 nodedb/tests/wire/cases/sql_typeguard_defaults/mod.rs create mode 100644 nodedb/tests/wire/cases/sql_typeguard_defaults/validate.rs diff --git a/nodedb-physical/src/lib.rs b/nodedb-physical/src/lib.rs index d80c6046a..9b7489305 100644 --- a/nodedb-physical/src/lib.rs +++ b/nodedb-physical/src/lib.rs @@ -18,5 +18,5 @@ pub mod visitor; pub use convert_context::SharedConvertContext; pub use error::ConvertError; -pub use surrogate::{FreshSurrogateKind, SurrogateAssignError, SurrogateAssigner}; +pub use surrogate::{SurrogateAssignError, SurrogateAssigner}; pub use visitor::{PhysicalTaskVisitor, dispatch}; diff --git a/nodedb-physical/src/physical_plan/meta.rs b/nodedb-physical/src/physical_plan/meta.rs index e7244a6c2..297de18ab 100644 --- a/nodedb-physical/src/physical_plan/meta.rs +++ b/nodedb-physical/src/physical_plan/meta.rs @@ -71,10 +71,17 @@ pub enum MetaOp { /// /// `target_type`: "document_schemaless", "document_strict", "kv". /// `schema_json`: for "document_strict"/"kv", JSON-serialized column definitions. + /// `source_storage_mode`: the collection's storage mode BEFORE this + /// conversion, read from the catalog by the Control Plane dispatcher. + /// The Data Plane's own `doc_configs` cache still reflects the OLD mode + /// at dispatch time (the catalog flip and re-register happen after this + /// op returns), so the handler cannot resolve the source format from + /// that cache. It must take it from the plan instead. ConvertCollection { collection: QualifiedCollection, target_type: String, schema_json: String, + source_storage_mode: super::document::StorageMode, }, /// Snapshot a tenant's data from the sparse engine. diff --git a/nodedb-physical/src/surrogate.rs b/nodedb-physical/src/surrogate.rs index d0b6a7624..05bd6c463 100644 --- a/nodedb-physical/src/surrogate.rs +++ b/nodedb-physical/src/surrogate.rs @@ -24,20 +24,6 @@ pub enum SurrogateAssignError { Backend(String), } -/// The identity convention a freshly minted row's surrogate binds under. -/// -/// This enum names the convention. It never formats one. -/// `nodedb`'s allocator turns the variant into the identity string. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum FreshSurrogateKind { - /// The row's identity is its document storage key. Covers a schemaless - /// row with no declared `PRIMARY KEY`, and a timeseries row. - DocumentStorageKey, - /// The row's identity is the strict-schema auto `_rowid` column value. - /// The Data Plane writes that column as an `Int64`. - AutoRowId, -} - /// Allocate stable, cross-engine surrogates for `(collection, pk_bytes)`. /// /// Implementations must be: @@ -72,13 +58,12 @@ pub trait SurrogateAssigner: Send + Sync { /// content-address on, so repeated calls never collapse onto one /// surrogate. /// - /// `kind` picks the convention the binding uses. The returned `String` is - /// the identity. Callers use it verbatim and never re-derive it. + /// The returned `String` is the bound identity. The caller uses it + /// verbatim and never re-derives it. fn assign_fresh( &self, database_id: DatabaseId, tenant_id: TenantId, collection: &str, - kind: FreshSurrogateKind, ) -> Result<(Surrogate, String), SurrogateAssignError>; } diff --git a/nodedb/src/control/planner/rls_injection/context.rs b/nodedb/src/control/planner/rls_injection/context.rs index 1ab113df8..3fb02aea3 100644 --- a/nodedb/src/control/planner/rls_injection/context.rs +++ b/nodedb/src/control/planner/rls_injection/context.rs @@ -110,20 +110,23 @@ impl RlsCtx<'_> { ) } - /// Admit a document write, injecting the row's storage key as `id`. + /// Admit a document write, injecting the row's client-visible identity + /// as `id`. /// /// A schemaless row with no declared `id` column carries its identity - /// only in that key, never in `image`. The read paths inject the same - /// string via `sparse_row_to_doc`, so a policy naming `id` judges the - /// write against the value a later read returns. `inject_str_field` is - /// a no-op when `image` already carries `id`. + /// only in its storage key, never in `image`. The caller decides the + /// encoding: a minted key renders as the surrogate's decimal string, + /// never the hex storage key, matching what the read paths inject via + /// `sparse_row_to_doc` — a policy naming `id` judges the write against + /// the value a later read returns. `inject_str_field` is a no-op when + /// `image` already carries `id`. pub(super) fn admit_document_write_image( &self, collection: &nodedb_types::QualifiedCollection, - row_key: &str, + identity: &crate::engine::document::store::RowIdentity, image: &[u8], ) -> crate::Result<()> { - let with_id = nodedb_query::msgpack_scan::inject_str_field(image, "id", row_key); + let with_id = nodedb_query::msgpack_scan::inject_str_field(image, "id", identity.as_str()); self.admit_write_image(collection, &with_id) } diff --git a/nodedb/src/control/planner/rls_injection/document.rs b/nodedb/src/control/planner/rls_injection/document.rs index 86e8a1729..3533a51a9 100644 --- a/nodedb/src/control/planner/rls_injection/document.rs +++ b/nodedb/src/control/planner/rls_injection/document.rs @@ -4,7 +4,7 @@ use nodedb_physical::physical_plan::DocumentOp; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::{RowIdentity, StorageKey}; use super::context::RlsCtx; @@ -124,8 +124,8 @@ pub(super) fn inject_document(ctx: &RlsCtx<'_>, op: &mut DocumentOp) -> crate::R surrogate, .. } => { - let row_key = surrogate_to_doc_id(*surrogate); - ctx.admit_document_write_image(collection, &row_key, value)?; + let identity = StorageKey::for_surrogate(*surrogate).to_identity(); + ctx.admit_document_write_image(collection, &identity, value)?; ctx.set_post_filters(collection, rls_filters) } @@ -140,10 +140,11 @@ pub(super) fn inject_document(ctx: &RlsCtx<'_>, op: &mut DocumentOp) -> crate::R // falls back to its document id, which a declared key already // carries in the body. for (index, (document_id, value)) in documents.iter().enumerate() { - let row_key = surrogates - .get(index) - .map_or_else(|| document_id.clone(), |s| surrogate_to_doc_id(*s)); - ctx.admit_document_write_image(collection, &row_key, value)?; + let identity = surrogates.get(index).map_or_else( + || RowIdentity::from_user_key(document_id.clone()), + |s| StorageKey::for_surrogate(*s).to_identity(), + ); + ctx.admit_document_write_image(collection, &identity, value)?; } ctx.set_post_filters(collection, rls_filters) } diff --git a/nodedb/src/control/planner/sql_plan_convert/convert.rs b/nodedb/src/control/planner/sql_plan_convert/convert.rs index 087f06bf1..91f6ae37b 100644 --- a/nodedb/src/control/planner/sql_plan_convert/convert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/convert.rs @@ -197,30 +197,25 @@ impl ConvertContext { /// producing executable work. /// /// A metadata plan, or a plan with no wired assigner, returns a - /// `Surrogate::ZERO` placeholder. Its identity string comes from - /// [`fresh_identity_string`](crate::control::surrogate::fresh_identity_string), - /// the same function the allocator calls. + /// `Surrogate::ZERO` placeholder paired with its rendered identity, via + /// [`RowIdentity::for_surrogate`](crate::engine::document::store::RowIdentity::for_surrogate), + /// the same type the allocator renders through. pub fn fresh_surrogate( &self, collection: &str, - kind: nodedb_physical::FreshSurrogateKind, ) -> crate::Result<(nodedb_types::Surrogate, String)> { let placeholder = || { + let zero = nodedb_types::Surrogate::ZERO; Ok(( - nodedb_types::Surrogate::ZERO, - crate::control::surrogate::fresh_identity_string( - kind, - nodedb_types::Surrogate::ZERO, - ), + zero, + crate::engine::document::store::RowIdentity::for_surrogate(zero).into_string(), )) }; if self.is_metadata() { return placeholder(); } match self.surrogate_assigner.as_ref() { - Some(assigner) => { - assigner.assign_fresh(self.database_id, self.tenant_id, collection, kind) - } + Some(assigner) => assigner.assign_fresh(self.database_id, self.tenant_id, collection), None => placeholder(), } } @@ -360,14 +355,7 @@ mod tests { .as_u32(), 0 ); - assert_eq!( - metadata - .fresh_surrogate("users", nodedb_physical::FreshSurrogateKind::AutoRowId) - .unwrap() - .0 - .as_u32(), - 0 - ); + assert_eq!(metadata.fresh_surrogate("users").unwrap().0.as_u32(), 0); assert_eq!( assigner .lookup(DatabaseId::DEFAULT, TenantId::new(1), "users", b"new-user") diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs index 6386d8288..3d89fdbe0 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs @@ -80,11 +80,7 @@ pub(in super::super) fn resolve_doc_identity_with_declared( } if is_auto_rowid_pk(primary_key) { - let (s, pk) = assign_fresh( - ctx, - collection, - nodedb_physical::FreshSurrogateKind::AutoRowId, - )?; + let (s, pk) = assign_fresh(ctx, collection)?; return Ok((pk, s)); } let mint_key: &str = declared.unwrap_or(primary_key); @@ -94,11 +90,7 @@ pub(in super::super) fn resolve_doc_identity_with_declared( Ok((id, s)) } DocId::ExplicitNull | DocId::Absent => { - let (s, pk) = assign_fresh( - ctx, - collection, - nodedb_physical::FreshSurrogateKind::DocumentStorageKey, - )?; + let (s, pk) = assign_fresh(ctx, collection)?; Ok((pk, s)) } } @@ -119,13 +111,12 @@ pub(in super::super) fn assign_for_pk( /// Content-addressing an empty pk collapses every such row onto one /// surrogate, a duplicate-key violation on the second insert. /// -/// Returns the identity string `kind` binds. The caller uses it verbatim. +/// Returns the bound identity string. The caller uses it verbatim. pub(super) fn assign_fresh( ctx: &ConvertContext, collection: &str, - kind: nodedb_physical::FreshSurrogateKind, ) -> crate::Result<(Surrogate, String)> { - ctx.fresh_surrogate(collection, kind) + ctx.fresh_surrogate(collection) } /// Whether a collection's declared primary key is the auto-generated `_rowid` diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs b/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs index 7c4616c99..93ae46a3b 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs @@ -116,10 +116,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_timeseries_ingest( // PK collapses every row onto `Surrogate::ZERO` and merges distinct // rows. Nothing looks a timeseries row up by this binding, so the // identity string is discarded. - let (s, _) = ctx.fresh_surrogate( - collection, - nodedb_physical::FreshSurrogateKind::DocumentStorageKey, - )?; + let (s, _) = ctx.fresh_surrogate(collection)?; surrogates.push(s); } Ok(vec![PhysicalTask { diff --git a/nodedb/src/control/server/response_shape/project.rs b/nodedb/src/control/server/response_shape/project.rs index a4483634e..5ba44ba4e 100644 --- a/nodedb/src/control/server/response_shape/project.rs +++ b/nodedb/src/control/server/response_shape/project.rs @@ -38,11 +38,16 @@ pub fn push_flat_rows( if is_scan_wrapper(&map) && let Some(serde_json::Value::Object(mut inner)) = map.remove("data") { - // The envelope carries the row's storage-key identity. A body - // with no `id` field carries it nowhere else. `or_insert` - // leaves a declared primary key as the authority. - if let Some(id) = map.remove("id") { - inner.entry("id").or_insert(id); + // The envelope carries the row's storage key, which is + // internal. A body with no `id` field carries identity + // nowhere else, so the key renders to an identity at this + // boundary. `or_insert` leaves a declared primary key as the + // authority. + if let Some(serde_json::Value::String(key)) = map.remove("id") { + let identity = crate::engine::document::store::identity_of(&key); + inner + .entry("id") + .or_insert(serde_json::Value::String(identity.into_string())); } out.push(inner); return; diff --git a/nodedb/src/control/server/response_translate/text_hybrid.rs b/nodedb/src/control/server/response_translate/text_hybrid.rs index 6113ed0e8..e736b948c 100644 --- a/nodedb/src/control/server/response_translate/text_hybrid.rs +++ b/nodedb/src/control/server/response_translate/text_hybrid.rs @@ -100,11 +100,11 @@ pub fn translate_text_search_payload( /// (`{doc_id: , : f64, /// vector_rank?, text_rank?}`), resolve each row's `doc_id` surrogate to the /// user PK via the catalog, and inject it as `id` — the field name every -/// `SELECT id` projection looks up. `doc_id` itself is left in place (mirrors -/// the vector translator's `_surrogate` debug field). A `__local_` sentinel -/// or an unresolved surrogate is left untouched: no `id` field is added, so -/// the projection reads NULL rather than a fabricated PK. On any decode -/// failure the payload is returned unchanged. +/// `SELECT id` projection looks up. A storage key must never reach a client, +/// so `doc_id` itself is rewritten to the row's identity too: the catalog PK +/// when one is declared, else the surrogate's decimal string. A `__local_` +/// sentinel (no surrogate binding) passes through untouched in both fields. +/// On any decode failure the payload is returned unchanged. pub fn translate_hybrid_search_payload( payload: &[u8], state: &SharedState, @@ -131,10 +131,10 @@ pub fn translate_hybrid_search_payload( let Some(surrogate) = parse_surrogate_hex(&hex_id) else { continue; }; - if let Some(pk) = resolve_surrogate_pk(state, database_id, tenant_id, collection, surrogate) - { - map.insert("id".to_string(), JsonValue::String(pk)); - } + let identity = resolve_surrogate_pk(state, database_id, tenant_id, collection, surrogate) + .unwrap_or_else(|| surrogate.as_u32().to_string()); + map.insert("id".to_string(), JsonValue::String(identity.clone())); + map.insert("doc_id".to_string(), JsonValue::String(identity)); } match sonic_rs::to_string(&JsonValue::Array(rows)) { diff --git a/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs b/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs index 18ffe1ed8..66fb8f94d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs @@ -67,11 +67,33 @@ pub async fn convert_collection( String::new() }; + // Resolve the SOURCE storage mode from the catalog row read above, before + // this DDL mutates `coll.collection_type`. Mirrors the exhaustive match in + // `build_doc_config_from_stored`, so the Data Plane handler decodes the + // scanned rows the same way the collection's own register path would. + let source_storage_mode = match &coll.collection_type { + nodedb_types::CollectionType::Document(nodedb_types::DocumentMode::Strict(schema)) => { + nodedb_physical::physical_plan::StorageMode::Strict { + schema: schema.clone(), + } + } + nodedb_types::CollectionType::KeyValue(config) => { + nodedb_physical::physical_plan::StorageMode::Strict { + schema: config.schema.clone(), + } + } + nodedb_types::CollectionType::Document(nodedb_types::DocumentMode::Schemaless) + | nodedb_types::CollectionType::Columnar(_) => { + nodedb_physical::physical_plan::StorageMode::Schemaless + } + }; + // Dispatch to Data Plane: re-encode if needed (strict = Binary Tuple). let plan = PhysicalPlan::Meta(MetaOp::ConvertCollection { collection: nodedb_types::QualifiedCollection::new(database_id, &collection), target_type: target_type.clone(), schema_json: schema_json_for_dp, + source_storage_mode, }); dispatch_system( @@ -141,6 +163,15 @@ pub async fn convert_collection( persist_collection_replicated(state, database_id, &coll) .map_err(|e| err("XX000", e.to_string()))?; + // Refresh this node's Data Plane `doc_configs` entry to the NEW storage + // mode. Without this, every later read of the collection resolves its + // body format from the pre-conversion entry until the process restarts. + crate::control::server::shared::ddl::neutral::collection::dispatch_register_from_stored( + state, &coll, + ) + .await + .map_err(|e| err("XX000", e.to_string()))?; + tracing::info!( %collection, target_type, diff --git a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs index 630b3d2b0..c54cd5c71 100644 --- a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs +++ b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs @@ -1696,6 +1696,7 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), target_type: "kv".into(), schema_json: "{}".into(), + source_storage_mode: nodedb_physical::physical_plan::StorageMode::Schemaless, }), PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { tenant_id: 1 }), PhysicalPlan::Meta(MetaOp::RestoreTenantSnapshot { diff --git a/nodedb/src/control/surrogate/assign/core/assign_ops.rs b/nodedb/src/control/surrogate/assign/core/assign_ops.rs index 7eaa70b22..18ebf64ed 100644 --- a/nodedb/src/control/surrogate/assign/core/assign_ops.rs +++ b/nodedb/src/control/surrogate/assign/core/assign_ops.rs @@ -9,24 +9,6 @@ use nodedb_types::Surrogate; use super::types::SurrogateAssigner; -/// Format the identity string a freshly minted surrogate binds under. -/// -/// This is the one site in `nodedb` that formats a minted surrogate. -/// [`assign_fresh`](SurrogateAssigner::assign_fresh) calls it on the -/// allocation path. The SQL-plan converter's placeholder paths call it for -/// `Surrogate::ZERO`, so both stay on one rule. -pub(crate) fn fresh_identity_string( - kind: nodedb_physical::FreshSurrogateKind, - surrogate: Surrogate, -) -> String { - match kind { - nodedb_physical::FreshSurrogateKind::AutoRowId => surrogate.as_u32().to_string(), - nodedb_physical::FreshSurrogateKind::DocumentStorageKey => { - crate::engine::document::store::surrogate_to_doc_id(surrogate) - } - } -} - impl SurrogateAssigner { /// Resolve `(collection, pk_bytes)` to a stable surrogate. /// @@ -129,20 +111,19 @@ impl SurrogateAssigner { /// [`assign`](Self::assign) has a fast-path lookup. This does not. Every /// call allocates a new value, so N rows get N distinct surrogates. /// - /// The surrogate self-binds to the identity string `kind` picks. - /// `AutoRowId` binds the decimal string, matching the `_rowid` value the - /// Data Plane writes, so `WHERE _rowid = N` resolves back to it. - /// `DocumentStorageKey` binds the 8-hex key the row is stored under. + /// The surrogate self-binds under its identity, so a later keyed lookup + /// resolves. The identity is the surrogate's decimal string, matching the + /// `_rowid` value the Data Plane writes for an auto-`_rowid` row, so + /// `WHERE _rowid = N` resolves back to it. /// - /// Both reuse `assign`'s bind/flush machinery, so the hwm advance - /// persists and Raft-proposes identically. Returns the bound identity - /// string, which the caller uses verbatim. + /// Reuses `assign`'s bind/flush machinery, so the hwm advance persists + /// and Raft-proposes identically. Returns the bound identity string, + /// which the caller uses verbatim. pub fn assign_fresh( &self, database_id: DatabaseId, tenant_id: TenantId, collection: &str, - kind: nodedb_physical::FreshSurrogateKind, ) -> crate::Result<(Surrogate, String)> { let catalog = self.credential_store.catalog(); @@ -162,7 +143,8 @@ impl SurrogateAssigner { continue; } }; - let pk = fresh_identity_string(kind, surrogate); + let pk = + crate::engine::document::store::RowIdentity::for_surrogate(surrogate).into_string(); let pk_bytes = pk.as_bytes(); catalog.put_surrogate(database_id, tenant_id, collection, pk_bytes, surrogate)?; self.wal_appender.record_bind_to_wal( diff --git a/nodedb/src/control/surrogate/assign/core/mod.rs b/nodedb/src/control/surrogate/assign/core/mod.rs index 30539977c..3b12349fe 100644 --- a/nodedb/src/control/surrogate/assign/core/mod.rs +++ b/nodedb/src/control/surrogate/assign/core/mod.rs @@ -22,5 +22,4 @@ mod assign_ops; mod flush; mod types; -pub(crate) use assign_ops::fresh_identity_string; pub use types::{SurrogateAssigner, SurrogateRegistryHandle}; diff --git a/nodedb/src/control/surrogate/assign/mod.rs b/nodedb/src/control/surrogate/assign/mod.rs index 226382bce..42aed71ce 100644 --- a/nodedb/src/control/surrogate/assign/mod.rs +++ b/nodedb/src/control/surrogate/assign/mod.rs @@ -6,5 +6,4 @@ pub(super) mod cluster_reserve; pub mod core; -pub(crate) use core::fresh_identity_string; pub use core::{SurrogateAssigner, SurrogateRegistryHandle}; diff --git a/nodedb/src/control/surrogate/mod.rs b/nodedb/src/control/surrogate/mod.rs index 2a2bba6de..755ed73c8 100644 --- a/nodedb/src/control/surrogate/mod.rs +++ b/nodedb/src/control/surrogate/mod.rs @@ -14,7 +14,6 @@ pub mod physical_impl; pub mod registry; pub mod wal_appender; -pub(crate) use assign::fresh_identity_string; pub use assign::{SurrogateAssigner, SurrogateRegistryHandle}; pub use bootstrap::bootstrap_registry; pub use persist::{SURROGATE_HWM, SurrogateHwmPersist, SystemCatalogHwm}; diff --git a/nodedb/src/control/surrogate/physical_impl.rs b/nodedb/src/control/surrogate/physical_impl.rs index 54958d3b8..067c03d63 100644 --- a/nodedb/src/control/surrogate/physical_impl.rs +++ b/nodedb/src/control/surrogate/physical_impl.rs @@ -4,9 +4,7 @@ //! `nodedb_physical::SurrogateAssigner` trait so the shared converter //! can allocate surrogates without depending on Origin internals. -use nodedb_physical::{ - FreshSurrogateKind, SurrogateAssignError, SurrogateAssigner as PhysicalSurrogateAssigner, -}; +use nodedb_physical::{SurrogateAssignError, SurrogateAssigner as PhysicalSurrogateAssigner}; use super::assign::SurrogateAssigner; @@ -31,9 +29,8 @@ impl PhysicalSurrogateAssigner for SurrogateAssigner { database_id: nodedb_types::DatabaseId, tenant_id: nodedb_types::TenantId, collection: &str, - kind: FreshSurrogateKind, ) -> Result<(nodedb_types::Surrogate, String), SurrogateAssignError> { - Self::assign_fresh(self, database_id, tenant_id, collection, kind) + Self::assign_fresh(self, database_id, tenant_id, collection) .map_err(|e| SurrogateAssignError::Backend(e.to_string())) } } diff --git a/nodedb/src/control/target_identity/document_id.rs b/nodedb/src/control/target_identity/document_id.rs index bba51d62f..347ff6d26 100644 --- a/nodedb/src/control/target_identity/document_id.rs +++ b/nodedb/src/control/target_identity/document_id.rs @@ -7,28 +7,25 @@ use nodedb_types::Surrogate; use super::pk::{TargetPk, extract_pk_value}; -use crate::control::surrogate::fresh_identity_string; +use crate::engine::document::store::RowIdentity; /// The user-visible primary key (`document_id`) for a row written on this /// target, mirroring the plain-`INSERT` identity path (`insert.rs`): an /// auto-`_rowid` row's PK is the decimal surrogate the Data Plane also writes /// into `_rowid`. A declared-PK row's PK is the field value extracted from -/// the body. A row with no content key is stored under its document storage -/// key. +/// the body. A row with no content key renders the same decimal identity. /// -/// Both forms come from `fresh_identity_string`, the same formatter +/// Both forms come from [`RowIdentity::for_surrogate`], the same formatter /// `assign_target_surrogate` binds a freshly minted row under. pub(crate) fn derive_document_id( target_pk: &TargetPk, body: &[u8], surrogate: Surrogate, ) -> String { - use nodedb_physical::FreshSurrogateKind; match target_pk { - TargetPk::AutoRowId => fresh_identity_string(FreshSurrogateKind::AutoRowId, surrogate), - TargetPk::Field { name, .. } => extract_pk_value(body, name).unwrap_or_else(|| { - fresh_identity_string(FreshSurrogateKind::DocumentStorageKey, surrogate) - }), + TargetPk::AutoRowId => RowIdentity::for_surrogate(surrogate).into_string(), + TargetPk::Field { name, .. } => extract_pk_value(body, name) + .unwrap_or_else(|| RowIdentity::for_surrogate(surrogate).into_string()), } } diff --git a/nodedb/src/control/target_identity/surrogate.rs b/nodedb/src/control/target_identity/surrogate.rs index 5d52e7c91..35e95f33b 100644 --- a/nodedb/src/control/target_identity/surrogate.rs +++ b/nodedb/src/control/target_identity/surrogate.rs @@ -20,12 +20,10 @@ pub(crate) fn assign_target_surrogate( ) -> crate::Result { match target_pk { TargetPk::AutoRowId => { - let (surrogate, _) = state.surrogate_assigner.assign_fresh( - database_id, - tenant_id, - target_collection, - nodedb_physical::FreshSurrogateKind::AutoRowId, - )?; + let (surrogate, _) = + state + .surrogate_assigner + .assign_fresh(database_id, tenant_id, target_collection)?; Ok(surrogate) } TargetPk::Field { name, declared } => match extract_pk_value(body, name) { @@ -54,7 +52,6 @@ pub(crate) fn assign_target_surrogate( database_id, tenant_id, target_collection, - nodedb_physical::FreshSurrogateKind::DocumentStorageKey, )?; Ok(surrogate) } diff --git a/nodedb/src/data/executor/core_loop/deferred.rs b/nodedb/src/data/executor/core_loop/deferred.rs index c5d36e364..1a2303afa 100644 --- a/nodedb/src/data/executor/core_loop/deferred.rs +++ b/nodedb/src/data/executor/core_loop/deferred.rs @@ -9,13 +9,14 @@ use std::sync::Arc; use super::CoreLoop; +use crate::engine::document::store::RowIdentity; use crate::event::types::{EventSource, RowId, WriteEvent, WriteOp}; /// A write that occurred during a transaction, pending deferred trigger emission. pub(in crate::data::executor) struct DeferredWrite { pub collection: String, pub op: WriteOp, - pub row_id: String, + pub identity: RowIdentity, pub new_value: Option>, pub old_value: Option>, } @@ -49,7 +50,7 @@ impl CoreLoop { sequence: self.event_sequence, collection: Arc::from(write.collection.as_str()), op: write.op, - row_id: RowId::new(write.row_id.as_str()), + row_id: RowId::new(write.identity.as_str()), lsn: self.watermark, database_id, tenant_id, diff --git a/nodedb/src/data/executor/core_loop/event_emit.rs b/nodedb/src/data/executor/core_loop/event_emit.rs index b3d43d0c0..572a20549 100644 --- a/nodedb/src/data/executor/core_loop/event_emit.rs +++ b/nodedb/src/data/executor/core_loop/event_emit.rs @@ -84,7 +84,7 @@ impl CoreLoop { task: &super::super::task::ExecutionTask, tid: u64, collection: &str, - row_id: &str, + identity: crate::engine::document::store::RowIdentity, new_stored: &[u8], prior_stored: Option<&[u8]>, ) { @@ -104,14 +104,18 @@ impl CoreLoop { }; // A schemaless body with no declared `id` column carries its identity - // only in the storage key. Inject `row_id` verbatim — the string every - // read path injects via `sparse_row_to_doc` — so a WHEN filter, CDC, - // or change stream reads the same `id` a query returns. - // `inject_str_field` is a no-op when the body already carries `id`, so - // a declared primary key is never overwritten. + // only in the storage key. The caller decided that identity already; + // this injects it, the same string every read path injects via + // `sparse_row_to_doc`, so a WHEN filter, CDC, or change stream reads + // the same `id` a query returns. This identity also becomes the + // `WriteEvent.row_id` that fans out to CDC serialization, streaming + // materialized views, event-trigger SQL generation, CRDT sync + // packaging, and webhook delivery. `inject_str_field` is a no-op + // when the body already carries `id`, so a declared primary key is + // never overwritten. let doc_id = self .is_schemaless_document_collection(database_id, tid, collection) - .then_some(row_id); + .then_some(identity.as_str()); let new_final: Cow<[u8]> = match (new_converted.as_deref(), doc_id) { (Some(c), _) => Cow::Borrowed(c), (None, Some(id)) => Cow::Owned(msgpack_scan::inject_str_field(new_stored, "id", id)), @@ -127,12 +131,34 @@ impl CoreLoop { task, collection, op, - row_id, + identity, Some(new_final.as_ref()), old_final.as_deref(), ); } + /// Emit a document-row DELETE event to the Event Plane. + /// + /// `identity` is the client-visible identity of the deleted row. The + /// caller decides the encoding, mirroring [`Self::emit_put_event`] on + /// the put side. + pub(in crate::data::executor) fn emit_document_delete_event( + &mut self, + task: &super::super::task::ExecutionTask, + collection: &str, + identity: crate::engine::document::store::RowIdentity, + old_value: Option<&[u8]>, + ) { + self.emit_write_event( + task, + collection, + crate::event::WriteOp::Delete, + identity, + None, + old_value, + ); + } + /// Emit a CDC write event for a node-label mutation on the nameable /// `__graph_node_labels__` stream ([`crate::event::graph_cdc::GRAPH_LABEL_STREAM`]). /// @@ -169,7 +195,8 @@ impl CoreLoop { } else { (Some(value.as_slice()), None) }; - self.emit_write_event(task, stream, op, node_id, new_value, old_value); + let identity = crate::engine::document::store::RowIdentity::from_user_key(node_id); + self.emit_write_event(task, stream, op, identity, new_value, old_value); } /// Emit a CDC write event for a graph edge mutation on the edge's own @@ -186,6 +213,7 @@ impl CoreLoop { edge: GraphEdgeEvent<'_>, ) { let row_id = crate::event::graph_cdc::edge_row_id(edge.src_id, edge.label, edge.dst_id); + let identity = crate::engine::document::store::RowIdentity::from_user_key(row_id); let (new_value, old_value): (Option<&[u8]>, Option<&[u8]>) = if matches!(edge.op, crate::event::WriteOp::Delete) { (None, None) @@ -196,7 +224,7 @@ impl CoreLoop { task, edge.collection, edge.op, - &row_id, + identity, new_value, old_value, ); @@ -227,7 +255,7 @@ impl CoreLoop { task: &super::super::task::ExecutionTask, collection: &str, op: crate::event::WriteOp, - row_id: &str, + identity: crate::engine::document::store::RowIdentity, new_value: Option<&[u8]>, old_value: Option<&[u8]>, ) { @@ -245,7 +273,7 @@ impl CoreLoop { sequence: self.event_sequence, collection: Arc::from(collection), op, - row_id: crate::event::types::RowId::new(row_id), + row_id: crate::event::types::RowId::new(identity.into_string()), lsn: self.watermark, database_id: task.request.database_id, tenant_id: task.request.tenant_id, diff --git a/nodedb/src/data/executor/core_loop/filter_match.rs b/nodedb/src/data/executor/core_loop/filter_match.rs index 16e58a7a6..0325eb510 100644 --- a/nodedb/src/data/executor/core_loop/filter_match.rs +++ b/nodedb/src/data/executor/core_loop/filter_match.rs @@ -43,8 +43,9 @@ use super::CoreLoop; /// body, so the body is matched with `id` injected — the same injection /// [`super::super::row_shape::sparse_row_to_doc`] applies to a materialized /// row, so `WHERE id ...` sees the identity a reader of the same row sees. A -/// strict row already surfaces `id` as a real tuple column, so no injection -/// runs on that arm. +/// minted key injects the client-visible decimal identity, not the hex +/// storage key. A strict row already surfaces `id` as a real tuple column, so +/// no injection runs on that arm. pub(in crate::data::executor) fn matches_with_resolved_schema( strict_schema: Option<&StrictSchema>, filters: &[ScanFilter], @@ -57,7 +58,12 @@ pub(in crate::data::executor) fn matches_with_resolved_schema( None => Ok(false), }, None => { - let with_id = nodedb_query::msgpack_scan::inject_str_field(body, "id", doc_id); + // `doc_id` comes straight off a store iterator, so only a `&str` + // is available here, not a `StorageKey`. A value that fails to + // parse as a minted key is a legacy or user key, taken verbatim. + let identity = crate::engine::document::store::identity_of(doc_id); + let with_id = + nodedb_query::msgpack_scan::inject_str_field(body, "id", identity.as_str()); ScanFilter::all_match_binary(filters, &with_id) } } diff --git a/nodedb/src/data/executor/dispatch/meta.rs b/nodedb/src/data/executor/dispatch/meta.rs index b78010cfd..df6497593 100644 --- a/nodedb/src/data/executor/dispatch/meta.rs +++ b/nodedb/src/data/executor/dispatch/meta.rs @@ -84,12 +84,14 @@ impl CoreLoop { collection, target_type, schema_json, + source_storage_mode, } => self.execute_convert_collection( task, tid, collection.as_str(), target_type, schema_json, + source_storage_mode, ), MetaOp::PurgeTenant { tenant_id } => self.execute_purge_tenant(task, *tenant_id), diff --git a/nodedb/src/data/executor/handlers/bulk_dml/delete.rs b/nodedb/src/data/executor/handlers/bulk_dml/delete.rs index fd516a3cb..418bc5b2f 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/delete.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/delete.rs @@ -179,10 +179,11 @@ impl CoreLoop { Ok(None) => continue, Err(e) => return self.response_error(task, e), }; + let identity = crate::engine::document::store::identity_of(doc_id); if let Err(e) = rls_write_gate::admit_stored_row( rls_write_check, &stored, - doc_id, + &identity, strict_schema.as_ref(), tid, collection, @@ -242,7 +243,13 @@ impl CoreLoop { .flatten() { Some(bytes) => { - match returning_doc::from_stored(&bytes, doc_id, strict_schema.as_ref()) { + // `doc_id` is the storage key from the scan. `RETURNING` + // reports the row's client-visible identity, not the + // storage key. A value that fails to parse as a minted + // key is a legacy or user key, taken verbatim. + let identity = crate::engine::document::store::identity_of(doc_id); + match returning_doc::from_stored(&bytes, &identity, strict_schema.as_ref()) + { Ok(doc) => Some(doc), Err(e) => return self.response_error(task, e), } @@ -417,12 +424,11 @@ impl CoreLoop { collection, deleted_bytes, ); - self.emit_write_event( + let event_identity = crate::engine::document::store::identity_of(doc_id); + self.emit_document_delete_event( task, collection, - crate::event::WriteOp::Delete, - doc_id, - None, + event_identity, Some(old_converted.as_deref().unwrap_or(deleted_bytes)), ); affected += 1; diff --git a/nodedb/src/data/executor/handlers/bulk_dml/scan.rs b/nodedb/src/data/executor/handlers/bulk_dml/scan.rs index b3db843f4..a700d1792 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/scan.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/scan.rs @@ -89,11 +89,7 @@ pub(in crate::data::executor) fn ollp_actual_surrogates(doc_ids: &[String]) -> V let mut surrogates: Vec = doc_ids .iter() .filter_map(|id| { - if id.len() == 8 { - u32::from_str_radix(id, 16).ok() - } else { - None - } + crate::engine::document::store::doc_id_to_surrogate(id).map(|s| s.as_u32()) }) .collect(); surrogates.sort_unstable(); diff --git a/nodedb/src/data/executor/handlers/bulk_dml/update.rs b/nodedb/src/data/executor/handlers/bulk_dml/update.rs index 85877a41b..78a926219 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/update.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/update.rs @@ -342,21 +342,26 @@ impl CoreLoop { // Event Plane's WAL-replay bulk variants are aggregate // metadata reconstructed only when the live per-row events // were lost — the live path always emits per row. + // `doc_id` is the surrogate hex storage key. A value that fails + // to parse as a minted key is a legacy or user key, taken + // verbatim. + let row_identity = crate::engine::document::store::identity_of(doc_id); + // `row_identity` is read again below for `RETURNING`'s `id` field, + // so the event-emit boundary gets a clone rather than the move. self.emit_put_event( task, tid, collection, - doc_id, + row_identity.clone(), &updated_bytes, Some(¤t_bytes), ); affected += 1; if returning.is_some() { - // `doc_id` is the surrogate hex storage key, which only - // stands in as `id` for a row that declares no primary - // key of its own — overwriting a declared key would - // return a value the client never wrote. - returning_doc::attach_row_id(&mut doc, doc_id); + // `row_identity` only stands in as `id` for a row that + // declares no primary key of its own — overwriting a + // declared key would return a value the client never wrote. + returning_doc::attach_row_id(&mut doc, &row_identity); returned_docs.push(doc); } // Carry the surrogate + post-image back for a post-apply diff --git a/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs b/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs index 52e1797fe..770e31885 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs @@ -87,13 +87,25 @@ impl CoreLoop { ) .ok_or_else(|| { crate::diag::strict_row_undecodable(collection, doc_id, "bulk_update_project"); - crate::data::executor::strict_format::undecodable_strict_row(collection, doc_id) + let identity = crate::engine::document::store::identity_of(doc_id); + crate::data::executor::strict_format::undecodable_strict_row( + collection, + identity.as_str(), + ) })?, - None => crate::data::executor::handlers::returning_doc::from_stored( - ¤t_bytes, - doc_id, - None, - )?, + None => { + // `doc_id` is the storage key from the apply set. The + // decoded document's `id` must be the row's client-visible + // identity, not the storage key. A value that fails to + // parse as a minted key is a legacy or user key, taken + // verbatim. + let identity = crate::engine::document::store::identity_of(doc_id); + crate::data::executor::handlers::returning_doc::from_stored( + ¤t_bytes, + &identity, + None, + )? + } }; // Feeds the secondary-index SET diff for values the UPDATE drops. diff --git a/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs b/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs index 9ab679902..4818de70e 100644 --- a/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs +++ b/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs @@ -116,10 +116,11 @@ impl CoreLoop { self.sparse .get(task.request.database_id.as_u64(), tid, collection, doc_id)? { + let identity = crate::engine::document::store::identity_of(doc_id); self.stage_admit_write( rls_write_check, &body, - doc_id, + &identity, task.request.database_id.as_u64(), tid, collection, @@ -217,10 +218,11 @@ impl CoreLoop { )?; // Decide the staged post-image against the write policy: this is // the row the Calvin flush will install. + let identity = crate::engine::document::store::identity_of(&doc_id); self.stage_admit_write( rls_write_check, &new_body, - &doc_id, + &identity, database_id.as_u64(), tid, collection, diff --git a/nodedb/src/data/executor/handlers/control/crdt_doc.rs b/nodedb/src/data/executor/handlers/control/crdt_doc.rs index 4097c863a..0f7ac9c3d 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_doc.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_doc.rs @@ -17,7 +17,7 @@ use crate::data::executor::handlers::point::apply_delete::PointDeleteParams; use crate::data::executor::handlers::returning_doc; use crate::data::executor::handlers::returning_rows; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::{RowIdentity, StorageKey}; use nodedb_physical::physical_plan::ReturningSpec; /// Borrowed arguments for [`CoreLoop::execute_crdt_doc_upsert`], grouped so the @@ -123,7 +123,11 @@ impl CoreLoop { // No strict schema: a CRDT row's stored body is whatever // `encode_crdt_row` materialized from Loro, which is always // MessagePack regardless of the collection's storage mode. - let doc = match returning_doc::from_stored(&bytes, document_id, None) { + let doc = match returning_doc::from_stored( + &bytes, + &RowIdentity::from_user_key(document_id), + None, + ) { Ok(doc) => doc, Err(e) => return self.response_error(task, e), }; @@ -201,7 +205,8 @@ impl CoreLoop { } let tid = tenant_id.as_u64(); - let storage_key = surrogate_to_doc_id(surrogate); + let storage_key = StorageKey::for_surrogate(surrogate); + let row_key = storage_key.to_string(); // The sparse-store removal and its index cascades run in one write txn // this handler owns: on any failure it is dropped un-committed and none // of them land. @@ -222,7 +227,7 @@ impl CoreLoop { database_id: task.request.database_id.as_u64(), tid, collection, - document_id: storage_key.as_str(), + document_id: row_key.as_str(), surrogate, user_roles: &task.request.user_roles, enforce: false, @@ -258,12 +263,10 @@ impl CoreLoop { collection, prior_bytes, ); - self.emit_write_event( + self.emit_document_delete_event( task, collection, - crate::event::WriteOp::Delete, - storage_key.as_str(), - None, + storage_key.to_identity(), Some(old_converted.as_deref().unwrap_or(prior_bytes)), ); } @@ -276,7 +279,11 @@ impl CoreLoop { if let Some(prior_bytes) = outcome.prior_value.as_deref() { // No strict schema — see the upsert path: a CRDT row is // materialized as MessagePack in either storage mode. - let doc = match returning_doc::from_stored(prior_bytes, document_id, None) { + let doc = match returning_doc::from_stored( + prior_bytes, + &RowIdentity::from_user_key(document_id), + None, + ) { Ok(doc) => doc, Err(e) => return self.response_error(task, e), }; diff --git a/nodedb/src/data/executor/handlers/control/crdt_materialize.rs b/nodedb/src/data/executor/handlers/control/crdt_materialize.rs index 71aaa6af0..51160d8dc 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_materialize.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_materialize.rs @@ -36,7 +36,7 @@ use crate::data::executor::handlers::point::apply_put::PointPutParams; use crate::data::executor::task::ExecutionTask; use crate::engine::crdt::tenant_state::TenantCrdtEngine; use crate::engine::document::crdt_store::loro_value_to_json; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::StorageKey; impl CoreLoop { /// Read the merged Loro row back and encode it into the schemaless @@ -106,7 +106,8 @@ impl CoreLoop { index_text: bool, ) { let database_id = task.request.database_id.as_u64(); - let storage_key = surrogate_to_doc_id(surrogate); + let storage_key = StorageKey::for_surrogate(surrogate); + let row_key = storage_key.to_string(); let txn = match self.sparse.begin_write() { Ok(t) => t, @@ -122,7 +123,7 @@ impl CoreLoop { database_id, tid, collection, - document_id: storage_key.as_str(), + document_id: row_key.as_str(), surrogate, value, index_text, @@ -158,7 +159,7 @@ impl CoreLoop { task, tid, collection, - storage_key.as_str(), + storage_key.to_identity(), value, prior.prior_value.as_deref(), ); diff --git a/nodedb/src/data/executor/handlers/control/range_scan_versioned.rs b/nodedb/src/data/executor/handlers/control/range_scan_versioned.rs index 1e0c3ac5a..753247c87 100644 --- a/nodedb/src/data/executor/handlers/control/range_scan_versioned.rs +++ b/nodedb/src/data/executor/handlers/control/range_scan_versioned.rs @@ -134,13 +134,18 @@ impl CoreLoop { // the same encoding every other RLS site filters on — a // strict body is a Binary Tuple until it is decoded here. // A schemaless row's identity lives only in its storage - // key when its body carries no `id` field, so it is - // injected before the RLS check, matching what a reader - // of the same row sees. + // key when its body carries no `id` field, so the + // client-visible identity is injected before the RLS + // check, matching what a reader of the same row sees. Some(filters) => match nodedb_types::json_msgpack::json_to_msgpack(&doc) { Ok(mp) => { let mp = if strict_schema.is_none() { - nodedb_query::msgpack_scan::inject_str_field(&mp, "id", doc_id) + let identity = crate::engine::document::store::identity_of(doc_id); + nodedb_query::msgpack_scan::inject_str_field( + &mp, + "id", + identity.as_str(), + ) } else { mp }; diff --git a/nodedb/src/data/executor/handlers/convert.rs b/nodedb/src/data/executor/handlers/convert.rs index 318598716..ce3678fbb 100644 --- a/nodedb/src/data/executor/handlers/convert.rs +++ b/nodedb/src/data/executor/handlers/convert.rs @@ -3,26 +3,52 @@ //! CONVERT COLLECTION handler: re-encode documents for a new storage mode. //! //! Scans all documents in the collection and re-encodes them in-place. -//! For `TO strict`: validates each doc against the schema and encodes as -//! Binary Tuple via `strict_format::json_to_binary_tuple`. -//! For `TO document` or `TO kv`: no re-encoding needed — sparse engine -//! stores raw bytes regardless of type. +//! For `TO strict`: injects each row's client-visible `id`, validates the +//! result against the schema, and encodes it as a Binary Tuple via +//! `strict_format::bytes_to_binary_tuple`. +//! For `TO document` or `TO kv`: a Binary Tuple source re-encodes to +//! MessagePack. A schemaless source needs no re-encoding — the sparse +//! engine already stores it as MessagePack. +//! A row that fails to convert fails the whole statement: the handler +//! returns an error response instead of a success payload, so the caller +//! never flips the catalog's collection type over partially-converted data. use sonic_rs; +use nodedb_physical::physical_plan::StorageMode; +use nodedb_query::msgpack_scan; use nodedb_types::columnar::{ColumnDef, StrictSchema}; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::response_codec; +use crate::data::executor::scan_normalize::sparse_body_to_msgpack; +use crate::data::executor::sparse_body_format::SparseBodyFormat; use crate::data::executor::task::ExecutionTask; +use crate::engine::document::store::identity_of; + +/// Map the plan's declared source storage mode to the row-decode format. +/// +/// The plan carries this instead of the handler reading `doc_configs`: at +/// dispatch time that cache still describes the mode from BEFORE this +/// conversion, since the catalog flip and Data Plane re-register happen +/// only after this op returns successfully. +fn source_format_of(mode: &StorageMode) -> SparseBodyFormat { + match mode { + StorageMode::Strict { schema } => SparseBodyFormat::Strict(schema.clone()), + StorageMode::Schemaless => SparseBodyFormat::Document, + } +} impl CoreLoop { /// Execute a collection conversion. /// - /// - `TO document` / `TO kv`: no re-encoding needed. Catalog update on Control Plane. - /// - `TO strict`: re-encode each document as a Binary Tuple using the provided - /// schema. Documents that fail validation are skipped and counted as errors. + /// - `TO document` / `TO kv`: re-encodes a Binary Tuple source to + /// MessagePack. A schemaless source is left untouched. Catalog update + /// happens on the Control Plane, after this returns successfully. + /// - `TO strict`: re-encode each document as a Binary Tuple using the + /// provided schema. A document that fails to encode fails the + /// statement. pub(in crate::data::executor) fn execute_convert_collection( &mut self, task: &ExecutionTask, @@ -30,6 +56,7 @@ impl CoreLoop { collection: &str, target_type: &str, schema_json: &str, + source_storage_mode: &StorageMode, ) -> Response { tracing::debug!( core = self.core_id, @@ -38,37 +65,14 @@ impl CoreLoop { "converting collection" ); + let source_format = source_format_of(source_storage_mode); + match target_type { - "document_strict" => self.convert_to_strict(task, tid, collection, schema_json), + "document_strict" => { + self.convert_to_strict(task, tid, collection, schema_json, source_format) + } "document_schemaless" | "kv" => { - // No re-encoding needed — sparse engine stores raw MessagePack bytes - // regardless of collection type. Catalog type update handled by - // Control Plane after this returns. - let count = self - .sparse - .scan_documents( - task.request.database_id.as_u64(), - tid, - collection, - usize::MAX, - ) - .map(|docs| docs.len() as u64) - .unwrap_or(0); - - let result = serde_json::json!({ - "converted": count, - "target_type": target_type, - "collection": collection, - }); - match response_codec::encode_json_as_msgpack(&result) { - Ok(payload) => self.response_with_payload(task, payload), - Err(e) => self.response_error( - task, - ErrorCode::Internal { - detail: e.to_string(), - }, - ), - } + self.convert_from_strict(task, tid, collection, target_type, source_format) } other => self.response_error( task, @@ -80,12 +84,19 @@ impl CoreLoop { } /// Convert to strict mode: re-encode each document as a Binary Tuple. + /// + /// The sparse engine keys a minted row by its storage key, not its + /// client-visible `id`. A `SELECT` synthesizes `id` at read time; this + /// re-encode must do the same before validating and encoding, or a row + /// with no declared primary key loses its identity and the target + /// schema's NOT NULL `id` column rejects it. fn convert_to_strict( &mut self, task: &ExecutionTask, tid: u64, collection: &str, schema_json: &str, + source_format: SparseBodyFormat, ) -> Response { // Parse the target schema from JSON column definitions. let columns: Vec = match sonic_rs::from_str(schema_json) { @@ -133,36 +144,54 @@ impl CoreLoop { } }; - // Re-encode each document as a Binary Tuple. let mut converted = 0u64; - let mut errors = 0u64; for (doc_id, doc_bytes) in &docs { - match super::super::strict_format::bytes_to_binary_tuple(doc_bytes, &schema, collection) - { - Ok(tuple_bytes) => { - if let Err(e) = - self.sparse - .put(database_id, tid, collection, doc_id, &tuple_bytes) - { - tracing::warn!(doc_id, error = %e, "failed to write converted doc"); - errors += 1; - continue; - } - converted += 1; - } + let normalized = sparse_body_to_msgpack(doc_bytes, source_format.as_format_ref()); + let identity = identity_of(doc_id); + let with_id = msgpack_scan::inject_str_field(&normalized, "id", identity.as_str()); + + let tuple_bytes = match super::super::strict_format::bytes_to_binary_tuple( + &with_id, &schema, collection, + ) { + Ok(bytes) => bytes, Err(e) => { - tracing::warn!(doc_id, error = %e, "strict conversion failed"); - errors += 1; + return self.response_error( + task, + ErrorCode::Internal { + detail: format!( + "collection '{collection}': row '{identity}' failed to convert to document_strict: {e}" + ), + }, + ); } + }; + + if let Err(e) = self + .sparse + .put(database_id, tid, collection, doc_id, &tuple_bytes) + { + return self.response_error( + task, + ErrorCode::Internal { + detail: format!( + "collection '{collection}': row '{identity}' failed to write converted document_strict body: {e}" + ), + }, + ); } + // Write-through: a point-get after this statement must see the + // re-encoded bytes, not a stale cache entry from before the + // conversion. `sparse.put` alone never touches this cache. + self.doc_cache + .put(database_id, tid, collection, doc_id, &tuple_bytes); + converted += 1; } - tracing::info!(%collection, converted, errors, "collection converted to document_strict"); + tracing::info!(%collection, converted, "collection converted to document_strict"); let result = serde_json::json!({ "converted": converted, - "errors": errors, "target_type": "document_strict", "collection": collection, }); @@ -176,4 +205,94 @@ impl CoreLoop { ), } } + + /// Convert from strict mode to schemaless document or kv storage. + /// + /// A Binary Tuple source re-encodes to MessagePack against its own + /// strict schema before the catalog flips. A schemaless source needs no + /// re-encoding: the sparse engine already stores it as MessagePack. + fn convert_from_strict( + &mut self, + task: &ExecutionTask, + tid: u64, + collection: &str, + target_type: &str, + source_format: SparseBodyFormat, + ) -> Response { + let database_id = task.request.database_id.as_u64(); + + let docs = match self + .sparse + .scan_documents(database_id, tid, collection, usize::MAX) + { + Ok(d) => d, + Err(e) => { + return self.response_error( + task, + ErrorCode::Internal { + detail: format!("scan failed: {e}"), + }, + ); + } + }; + + let converted = match source_format { + SparseBodyFormat::Strict(schema) => { + let mut converted = 0u64; + for (doc_id, doc_bytes) in &docs { + let identity = identity_of(doc_id); + let Some(mp) = + super::super::strict_format::binary_tuple_to_msgpack(doc_bytes, &schema) + else { + let e = super::super::strict_format::undecodable_strict_row( + collection, + identity.as_str(), + ); + return self.response_error( + task, + ErrorCode::Internal { + detail: format!( + "collection '{collection}': row '{identity}' failed to convert to {target_type}: {e}" + ), + }, + ); + }; + + if let Err(e) = self.sparse.put(database_id, tid, collection, doc_id, &mp) { + return self.response_error( + task, + ErrorCode::Internal { + detail: format!( + "collection '{collection}': row '{identity}' failed to write converted {target_type} body: {e}" + ), + }, + ); + } + // Write-through: a point-get after this statement must see + // the re-encoded bytes, not a stale cache entry from before + // the conversion. `sparse.put` alone never touches this cache. + self.doc_cache + .put(database_id, tid, collection, doc_id, &mp); + converted += 1; + } + converted + } + SparseBodyFormat::Document | SparseBodyFormat::VectorSidecar => docs.len() as u64, + }; + + let result = serde_json::json!({ + "converted": converted, + "target_type": target_type, + "collection": collection, + }); + match response_codec::encode_json_as_msgpack(&result) { + Ok(payload) => self.response_with_payload(task, payload), + Err(e) => self.response_error( + task, + ErrorCode::Internal { + detail: e.to_string(), + }, + ), + } + } } diff --git a/nodedb/src/data/executor/handlers/document/read/scan.rs b/nodedb/src/data/executor/handlers/document/read/scan.rs index 18bc51455..e8f4b329b 100644 --- a/nodedb/src/data/executor/handlers/document/read/scan.rs +++ b/nodedb/src/data/executor/handlers/document/read/scan.rs @@ -224,10 +224,9 @@ impl CoreLoop { if let Some(pf) = prefilter { filtered.retain(|(doc_id, _)| { - if let Ok(n) = u32::from_str_radix(doc_id, 16) { - pf.contains(nodedb_types::Surrogate::new(n)) - } else { - false + match crate::engine::document::store::doc_id_to_surrogate(doc_id) { + Some(surrogate) => pf.contains(surrogate), + None => false, } }); } diff --git a/nodedb/src/data/executor/handlers/document/resolve/apply.rs b/nodedb/src/data/executor/handlers/document/resolve/apply.rs index 19db5c84f..3e1a07086 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/apply.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/apply.rs @@ -50,7 +50,9 @@ impl CoreLoop { && let Err(e) = rls_write_gate::admit_stored_row( rls_write_check, value, - document_id, + &crate::engine::document::store::RowIdentity::from_user_key( + document_id.as_str(), + ), None, tid, collection.as_str(), diff --git a/nodedb/src/data/executor/handlers/document/resolve/apply_row.rs b/nodedb/src/data/executor/handlers/document/resolve/apply_row.rs index 2ad72a4b2..d297345ec 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/apply_row.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/apply_row.rs @@ -15,7 +15,7 @@ use crate::data::executor::enforcement::write_hook::{self, HookCtx, ImageBody, W use crate::data::executor::handlers::point::apply_delete::PointDeleteParams; use crate::data::executor::handlers::point::apply_put::PointPutParams; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::StorageKey; /// One already-decided row write, as the apply loop hands it over. pub(super) struct ApplyResolvedPut<'a> { @@ -56,7 +56,8 @@ impl CoreLoop { resolved_sum_targets, } = put; let database_id = task.request.database_id.as_u64(); - let row_key = surrogate_to_doc_id(surrogate); + let storage_key = StorageKey::for_surrogate(surrogate); + let row_key = storage_key.to_string(); let row_key = row_key.as_str(); let has_vectors = self.collection_has_vectors(database_id, tid, collection); @@ -146,7 +147,14 @@ impl CoreLoop { } let stored_bytes = outcome.stored_value; - self.emit_put_event(task, tid, collection, row_key, &stored_bytes, precondition); + self.emit_put_event( + task, + tid, + collection, + storage_key.to_identity(), + &stored_bytes, + precondition, + ); self.note_surrogate_write_lsn(task, tid, collection, surrogate.as_u32()); let mut write_set = Vec::new(); @@ -244,12 +252,10 @@ impl CoreLoop { } let old_converted = self.resolve_event_payload(database_id, tid, collection, prior_bytes); - self.emit_write_event( + self.emit_document_delete_event( task, collection, - crate::event::WriteOp::Delete, - document_id, - None, + StorageKey::for_surrogate(surrogate).to_identity(), Some(old_converted.as_deref().unwrap_or(prior_bytes)), ); } diff --git a/nodedb/src/data/executor/handlers/document/resolve/bulk.rs b/nodedb/src/data/executor/handlers/document/resolve/bulk.rs index 1a27f7571..8a7e9df6b 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/bulk.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/bulk.rs @@ -24,7 +24,7 @@ use crate::data::executor::handlers::bulk_dml::update_project::{ }; use crate::data::executor::handlers::{returning_rows, rls_write_gate}; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::doc_id_to_surrogate; +use crate::engine::document::store::{RowIdentity, doc_id_to_surrogate}; /// Borrowed arguments for [`CoreLoop::resolve_bulk_update`]. pub(super) struct ResolveBulkUpdate<'a> { @@ -168,17 +168,22 @@ impl CoreLoop { .map_err(ErrorCode::from)?; let mut mutations = Vec::with_capacity(doc_ids.len()); - let mut rows: Vec<(String, Vec)> = Vec::new(); + let mut rows: Vec<(RowIdentity, Vec)> = Vec::new(); for doc_id in doc_ids { // A row that vanished between the scan and this read removes // nothing, so it carries no image for the policy to restrict. let Some(stored) = self.doc_resolve_read(&ctx, collection, &doc_id)? else { continue; }; + // `doc_id` is the storage key from the scan. A value that fails + // to parse as a minted key is a legacy or user key, taken + // verbatim; `RETURNING` reports the client-visible identity + // either way, never the storage key. + let identity = crate::engine::document::store::identity_of(&doc_id); rls_write_gate::admit_stored_row( rls_write_check, &stored, - &doc_id, + &identity, ctx.strict_schema.as_ref(), tid, collection, @@ -194,12 +199,12 @@ impl CoreLoop { Some(stored.clone()), resolved_sum_targets, )); - rows.push((doc_id, stored)); + rows.push((identity, stored)); } - let borrowed: Vec<(&str, &[u8])> = rows + let borrowed: Vec<(&RowIdentity, &[u8])> = rows .iter() - .map(|(id, body)| (id.as_str(), body.as_slice())) + .map(|(id, body)| (id, body.as_slice())) .collect(); let response_payload = resolved_response_payload( returning, diff --git a/nodedb/src/data/executor/handlers/document/resolve/context.rs b/nodedb/src/data/executor/handlers/document/resolve/context.rs index 998c67663..abe188a94 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/context.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/context.rs @@ -15,7 +15,7 @@ use crate::bridge::envelope::ErrorCode; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::returning_rows; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::{RowIdentity, surrogate_to_doc_id}; /// What a resolver returns: the decided mutations and the decided reply, or the /// error the live handler would have returned for the same input. @@ -155,7 +155,7 @@ pub(super) fn resolved_response_payload( returning: Option<&ReturningSpec>, rls_filters: &[u8], strict_schema: Option<&StrictSchema>, - rows: &[(&str, &[u8])], + rows: &[(&RowIdentity, &[u8])], ) -> Result, ErrorCode> { match returning { Some(spec) => { diff --git a/nodedb/src/data/executor/handlers/document/resolve/point.rs b/nodedb/src/data/executor/handlers/document/resolve/point.rs index fa4b1dfc0..2e8ef7310 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/point.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/point.rs @@ -21,6 +21,7 @@ use crate::data::executor::handlers::point::update::post_image::{ }; use crate::data::executor::handlers::rls_write_gate; use crate::data::executor::task::ExecutionTask; +use crate::engine::document::store::{RowIdentity, StorageKey}; /// Borrowed arguments for [`CoreLoop::resolve_point_update`]. pub(super) struct ResolvePointUpdate<'a> { @@ -75,6 +76,8 @@ impl CoreLoop { let ctx = self.doc_resolve_ctx(task, tid, collection); let row_key = row_key_of(surrogate); let row_key = row_key.as_str(); + let row_identity = StorageKey::for_surrogate(surrogate).to_identity(); + let document_identity = RowIdentity::from_user_key(document_id); let config_key = ( task.request.database_id, @@ -139,7 +142,7 @@ impl CoreLoop { rls_write_gate::admit_stored_row( rls_write_check, &stored_image, - document_id, + &row_identity, ctx.strict_schema.as_ref(), tid, collection, @@ -150,7 +153,7 @@ impl CoreLoop { returning, rls_filters, ctx.strict_schema.as_ref(), - &[(document_id, stored_image.as_slice())], + &[(&document_identity, stored_image.as_slice())], )?; Ok(DocumentResolveOutcome { mutations: vec![put_mutation(ResolvedPut { @@ -186,6 +189,8 @@ impl CoreLoop { let ctx = self.doc_resolve_ctx(task, tid, collection); let row_key = row_key_of(surrogate); let row_key = row_key.as_str(); + let row_identity = StorageKey::for_surrogate(surrogate).to_identity(); + let document_identity = RowIdentity::from_user_key(document_id); // A row that is already absent removes nothing, so there is no image for // the policy to restrict — the same admission `gate_point_delete` makes. @@ -204,7 +209,7 @@ impl CoreLoop { rls_write_gate::admit_stored_row( rls_write_check, &prior, - row_key, + &row_identity, ctx.strict_schema.as_ref(), tid, collection, @@ -215,7 +220,7 @@ impl CoreLoop { returning, rls_filters, ctx.strict_schema.as_ref(), - &[(document_id, prior.as_slice())], + &[(&document_identity, prior.as_slice())], )?; Ok(DocumentResolveOutcome { mutations: vec![delete_mutation( diff --git a/nodedb/src/data/executor/handlers/document/resolve/upsert.rs b/nodedb/src/data/executor/handlers/document/resolve/upsert.rs index 9b7ebff83..ccb9745c5 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/upsert.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/upsert.rs @@ -20,6 +20,7 @@ use crate::data::executor::handlers::rls_write_gate; use crate::data::executor::handlers::upsert::merge::{apply_on_conflict_updates, merge_values}; use crate::data::executor::strict_format; use crate::data::executor::task::ExecutionTask; +use crate::engine::document::store::RowIdentity; /// Borrowed arguments for [`CoreLoop::resolve_upsert`]. pub(super) struct ResolveUpsert<'a> { @@ -57,6 +58,9 @@ impl CoreLoop { let ctx = self.doc_resolve_ctx(task, tid, collection); let row_key = row_key_of(surrogate); let row_key = row_key.as_str(); + let row_identity = + crate::engine::document::store::StorageKey::for_surrogate(surrogate).to_identity(); + let document_identity = RowIdentity::from_user_key(document_id); let existing = self.doc_resolve_read(&ctx, collection, row_key)?; let (body, precondition) = match existing { @@ -75,8 +79,15 @@ impl CoreLoop { }; // Both live branches gate the MessagePack body, no schema passed. - rls_write_gate::admit_stored_row(rls_write_check, &body, row_key, None, tid, collection) - .map_err(ErrorCode::from)?; + rls_write_gate::admit_stored_row( + rls_write_check, + &body, + &row_identity, + None, + tid, + collection, + ) + .map_err(ErrorCode::from)?; // `RETURNING` projects the stored image via the same `build_stored_body` // the apply runs, so the resolve reports the row that will actually land. @@ -101,7 +112,7 @@ impl CoreLoop { returning, rls_filters, ctx.strict_schema.as_ref(), - &[(document_id, stored_image.as_slice())], + &[(&document_identity, stored_image.as_slice())], )?; Ok(DocumentResolveOutcome { diff --git a/nodedb/src/data/executor/handlers/document/write/batch_insert.rs b/nodedb/src/data/executor/handlers/document/write/batch_insert.rs index 2d25b9fb2..a04774f17 100644 --- a/nodedb/src/data/executor/handlers/document/write/batch_insert.rs +++ b/nodedb/src/data/executor/handlers/document/write/batch_insert.rs @@ -325,7 +325,8 @@ impl CoreLoop { } for (i, row_key) in applied.iter().enumerate() { - self.emit_put_event(task, tid, collection, row_key, &documents[i].1, None); + let identity = crate::engine::document::store::identity_of(row_key); + self.emit_put_event(task, tid, collection, identity, &documents[i].1, None); } let mut response = if let Some(spec) = returning { @@ -337,10 +338,16 @@ impl CoreLoop { crate::types::TenantId::new(tid), collection, ); - let rows: Vec<(&str, &[u8])> = documents + let identities: Vec = documents + .iter() + .map(|(document_id, _)| { + crate::engine::document::store::RowIdentity::from_user_key(document_id.as_str()) + }) + .collect(); + let rows: Vec<(&crate::engine::document::store::RowIdentity, &[u8])> = identities .iter() .zip(stored_bodies.iter()) - .map(|((document_id, _), stored)| (document_id.as_str(), stored.as_slice())) + .map(|(identity, stored)| (identity, stored.as_slice())) .collect(); self.stored_returning_response(task, spec, rls_filters, strict_schema.as_ref(), &rows) } else { diff --git a/nodedb/src/data/executor/handlers/graph_rag.rs b/nodedb/src/data/executor/handlers/graph_rag.rs index cb5e7b1d5..8f1aa0d31 100644 --- a/nodedb/src/data/executor/handlers/graph_rag.rs +++ b/nodedb/src/data/executor/handlers/graph_rag.rs @@ -200,11 +200,12 @@ impl CoreLoop { // would only hash it straight back to the same node. // // The reporting key is resolved once per hit: the graph node name when - // the surrogate is bound to one, otherwise the document storage key. - // Falling back to the document key rather than to an index-local - // sentinel is what lets a hit fuse with the *text* leg, which keys on - // exactly that; the old `__local_{hnsw_id}` sentinel could match nothing - // and leaked an internal index id into the response's `node_id`. + // the surrogate is bound to one, otherwise the surrogate's + // client-visible decimal identity. Falling back to the identity rather + // than to an index-local sentinel is what lets a hit fuse with the + // *text* leg, which keys on exactly that; the old `__local_{hnsw_id}` + // sentinel could match nothing and leaked an internal index id into + // the response's `node_id`. let csr = self.csr_partition(database_id, tenant_id); let mut vector_scores: HashMap = HashMap::new(); let mut seeds: Vec = Vec::with_capacity(vector_results.len()); @@ -217,7 +218,11 @@ impl CoreLoop { Some(s) => csr .and_then(|c| c.node_id_for_surrogate(s)) .map(str::to_string) - .unwrap_or_else(|| crate::engine::document::store::surrogate_to_doc_id(s)), + .unwrap_or_else(|| { + crate::engine::document::store::RowIdentity::for_surrogate(s) + .as_str() + .to_string() + }), // No surrogate at all: the vector entry predates surrogate // plumbing, so it has no cross-engine identity. It still ranks // in the vector leg under a key that deliberately matches diff --git a/nodedb/src/data/executor/handlers/graph_rag_triple.rs b/nodedb/src/data/executor/handlers/graph_rag_triple.rs index 7a34ad499..a87cd358c 100644 --- a/nodedb/src/data/executor/handlers/graph_rag_triple.rs +++ b/nodedb/src/data/executor/handlers/graph_rag_triple.rs @@ -139,7 +139,9 @@ impl CoreLoop { .iter() .enumerate() .map(|(rank, r)| RankedResult { - document_id: crate::engine::document::store::surrogate_to_doc_id(r.doc_id), + document_id: crate::engine::document::store::RowIdentity::for_surrogate(r.doc_id) + .as_str() + .to_string(), rank, score: r.score, source: "text", diff --git a/nodedb/src/data/executor/handlers/kv/atomic.rs b/nodedb/src/data/executor/handlers/kv/atomic.rs index d3964ed6f..434e6f0ea 100644 --- a/nodedb/src/data/executor/handlers/kv/atomic.rs +++ b/nodedb/src/data/executor/handlers/kv/atomic.rs @@ -81,7 +81,7 @@ impl CoreLoop { task, collection, crate::event::WriteOp::Update, - &key_str, + crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), Some(&new_bytes), None, ); @@ -169,7 +169,7 @@ impl CoreLoop { task, collection, crate::event::WriteOp::Update, - &key_str, + crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), Some(&new_bytes), None, ); @@ -262,7 +262,7 @@ impl CoreLoop { task, collection, crate::event::WriteOp::Update, - &key_str, + crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), Some(new_value), None, ); @@ -342,7 +342,7 @@ impl CoreLoop { task, collection, crate::event::WriteOp::Update, - &key_str, + crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), Some(new_value), old.as_deref(), ); diff --git a/nodedb/src/data/executor/handlers/kv/crud/delete.rs b/nodedb/src/data/executor/handlers/kv/crud/delete.rs index 6aa18984e..6afd2cd08 100644 --- a/nodedb/src/data/executor/handlers/kv/crud/delete.rs +++ b/nodedb/src/data/executor/handlers/kv/crud/delete.rs @@ -59,7 +59,7 @@ impl CoreLoop { task, collection, crate::event::WriteOp::Delete, - &key_str, + crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), None, None, ); diff --git a/nodedb/src/data/executor/handlers/kv/crud/write_basic.rs b/nodedb/src/data/executor/handlers/kv/crud/write_basic.rs index 57bd8e2f5..39594631c 100644 --- a/nodedb/src/data/executor/handlers/kv/crud/write_basic.rs +++ b/nodedb/src/data/executor/handlers/kv/crud/write_basic.rs @@ -64,7 +64,14 @@ impl CoreLoop { Some(o) => (crate::event::WriteOp::Update, Some(o)), None => (crate::event::WriteOp::Insert, None), }; - self.emit_write_event(task, collection, op, &key_str, Some(value), old_slice); + self.emit_write_event( + task, + collection, + op, + crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), + Some(value), + old_slice, + ); self.note_kv_write_lsn(task, did, tid, collection, key); if let Some(spec) = returning { @@ -151,7 +158,7 @@ impl CoreLoop { task, collection, crate::event::WriteOp::Insert, - &key_str, + crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), Some(value), None, ); @@ -232,7 +239,7 @@ impl CoreLoop { task, collection, crate::event::WriteOp::Insert, - &key_str, + crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), Some(value), None, ); diff --git a/nodedb/src/data/executor/handlers/kv/crud/write_upsert.rs b/nodedb/src/data/executor/handlers/kv/crud/write_upsert.rs index 65be85dab..11ba0a53c 100644 --- a/nodedb/src/data/executor/handlers/kv/crud/write_upsert.rs +++ b/nodedb/src/data/executor/handlers/kv/crud/write_upsert.rs @@ -137,7 +137,7 @@ impl CoreLoop { task, collection, op, - &key_str, + crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), Some(&stored_bytes), old_slice, ); diff --git a/nodedb/src/data/executor/handlers/kv/predicate/apply.rs b/nodedb/src/data/executor/handlers/kv/predicate/apply.rs index a36ee0f13..47113afdf 100644 --- a/nodedb/src/data/executor/handlers/kv/predicate/apply.rs +++ b/nodedb/src/data/executor/handlers/kv/predicate/apply.rs @@ -86,7 +86,7 @@ impl CoreLoop { task, collection, crate::event::WriteOp::Update, - &key_str, + crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), Some(new_value), Some(old_body), ); diff --git a/nodedb/src/data/executor/handlers/kv/resolve/apply.rs b/nodedb/src/data/executor/handlers/kv/resolve/apply.rs index e9ce94810..5d1b6b85e 100644 --- a/nodedb/src/data/executor/handlers/kv/resolve/apply.rs +++ b/nodedb/src/data/executor/handlers/kv/resolve/apply.rs @@ -121,7 +121,7 @@ impl CoreLoop { task, collection.as_str(), op, - &key_str, + crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), Some(value), precondition.as_deref(), ); @@ -142,7 +142,7 @@ impl CoreLoop { task, collection.as_str(), crate::event::WriteOp::Delete, - &key_str, + crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), None, precondition.as_deref(), ); diff --git a/nodedb/src/data/executor/handlers/kv/rls.rs b/nodedb/src/data/executor/handlers/kv/rls.rs index 5d0f588df..d3d3e3c5d 100644 --- a/nodedb/src/data/executor/handlers/kv/rls.rs +++ b/nodedb/src/data/executor/handlers/kv/rls.rs @@ -40,7 +40,10 @@ pub(in crate::data::executor) fn admit_kv_row( return Ok(()); } // The key is shown for diagnostics only; a non-UTF-8 key is lossily - // rendered rather than failing a security decision on its encoding. + // rendered rather than failing a security decision on its encoding. A + // KV key is never a document storage key, so it is always the row's + // own identity, taken verbatim. let key_display = String::from_utf8_lossy(key); - rls_write_gate::admit_stored_row(rls_write_check, body, &key_display, None, tid, collection) + let identity = crate::engine::document::store::RowIdentity::from_user_key(key_display.as_ref()); + rls_write_gate::admit_stored_row(rls_write_check, body, &identity, None, tid, collection) } diff --git a/nodedb/src/data/executor/handlers/kv/transfer.rs b/nodedb/src/data/executor/handlers/kv/transfer.rs index 1abad955c..ec93a7efb 100644 --- a/nodedb/src/data/executor/handlers/kv/transfer.rs +++ b/nodedb/src/data/executor/handlers/kv/transfer.rs @@ -198,7 +198,7 @@ impl CoreLoop { task, collection, crate::event::WriteOp::Update, - &src_str, + crate::engine::document::store::RowIdentity::from_user_key(src_str.as_ref()), Some(&new_source), Some(&source_bytes), ); @@ -206,7 +206,7 @@ impl CoreLoop { task, collection, crate::event::WriteOp::Update, - &dst_str, + crate::engine::document::store::RowIdentity::from_user_key(dst_str.as_ref()), Some(&new_dest), if dest_bytes.is_empty() { None @@ -315,7 +315,7 @@ impl CoreLoop { task, source_collection, crate::event::WriteOp::Delete, - &item_str, + crate::engine::document::store::RowIdentity::from_user_key(item_str.as_ref()), None, Some(&item_data), ); @@ -323,7 +323,7 @@ impl CoreLoop { task, dest_collection, crate::event::WriteOp::Insert, - &dest_str, + crate::engine::document::store::RowIdentity::from_user_key(dest_str.as_ref()), Some(&item_data), None, ); diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/apply.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/apply.rs deleted file mode 100644 index 9b7dc16a4..000000000 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/apply.rs +++ /dev/null @@ -1,513 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! MERGE APPLY pass: verify the resolve→apply prediction, then atomically -//! apply every arm's writes with the Control-Plane-pre-assigned surrogates. -//! -//! This file owns the drift verification and the single redb write transaction -//! the UPDATE and INSERT arms share. The two things that cannot live under that -//! transaction have their own files: unwinding a partial apply (`abort`) and -//! the DELETE arms, whose cascade opens transactions of its own and therefore -//! runs after the commit (`delete_arms`). - -use std::collections::HashMap; - -use crate::bridge::envelope::{ErrorCode, Response, WriteSetEntry}; -use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::enforcement::write_hook; -use crate::data::executor::handlers::point::apply_put::PointPutParams; -use crate::data::executor::handlers::transaction::undo::UndoEntry; -use crate::data::executor::response_codec::encode_json_as_msgpack; -use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; -use nodedb_types::Surrogate; - -use super::super::merge::MergeParams; -use super::super::returning_rows; -use super::abort::MergeAbort; -use super::apply_support::{MergePutEvent, gate_merge_arms, record_put_index_undo, returning_doc}; -use super::delete_arms::{MergeDeleteArms, MergeDeleteTally}; - -impl CoreLoop { - /// APPLY pass: verify the resolve→apply prediction, then atomically apply. - pub(in crate::data::executor) fn execute_merge_apply( - &mut self, - task: &ExecutionTask, - tid: u64, - params: MergeParams<'_>, - ) -> Response { - let resolved = match params.resolved_inserts { - Some(r) => r, - None => { - return self.response_error( - task, - ErrorCode::Internal { - detail: "merge apply invoked without resolved inserts".into(), - }, - ); - } - }; - let database_id = task.request.database_id.as_u64(); - - let plan = match self.collect_merge_plan(database_id, tid, task.request.txn_id, ¶ms) { - Ok(p) => p, - Err(e) => return self.response_error(task, e), - }; - - // TOCTOU verification: the recomputed NOT-MATCHED insert-key set must - // still equal the orchestrator's predicted set. Any drift (a target row - // for a predicted-insert key appeared, or a matched row vanished) means - // the pre-assigned surrogates no longer describe the merge — return - // OllpRetryRequired WITHOUT writing so the orchestrator re-resolves. - let mut actual_keys: Vec<&str> = plan.inserts.iter().map(|i| i.join_key.as_str()).collect(); - actual_keys.sort_unstable(); - let mut predicted_keys: Vec<&str> = resolved.iter().map(|(k, _)| k.as_str()).collect(); - predicted_keys.sort_unstable(); - if actual_keys != predicted_keys { - return self.response_error(task, ErrorCode::OllpRetryRequired); - } - let surrogate_for: HashMap<&str, u32> = - resolved.iter().map(|(k, s)| (k.as_str(), *s)).collect(); - - // Whether the target maintains a secondary vector index. Gated ONCE here - // (the schemaless half scans `vector_params` unindexed) and threaded into - // the per-row UPDATE re-index below. - let has_vectors = self.collection_has_vectors(database_id, tid, params.target_collection); - - // Gate every arm on the target's write policy BEFORE the apply - // transaction opens, so a rejected row leaves nothing written and - // nothing to unwind. - if let Err(e) = - gate_merge_arms(&plan, params.rls_write_check, tid, params.target_collection) - { - return self.response_error(task, e); - } - - // One post-apply redo entry per indexed row — a `Put` for each - // UPDATE/INSERT post-image, a `Delete` for each removed row — carried - // back so the Control Plane mints the durable WAL redo the vector index - // needs to survive a WAL-only restart. Empty on non-vector targets. - let mut write_set: Vec = Vec::new(); - - // The whole MERGE is ONE boundary, so its DELETE arms are accounted - // here, before any phase runs: those arms apply in their own - // transactions AFTER the phase-A commit, so entries collected as they - // ran could only report a violation phase A had already made durable. - // Their pre-images are the plan's captured bodies, which the classifier - // already holds — nothing is re-read. - let delete_bodies: Vec<&[u8]> = plan.deletes.iter().map(|d| d.body.as_slice()).collect(); - let mut balanced_entries = self.balanced_entries_for_submitted_deletes( - database_id, - tid, - params.target_collection, - &delete_bodies, - ); - - // Phase A: matched UPDATE + NOT-MATCHED INSERT share ONE redb write - // transaction. Any per-row error (including a UNIQUE violation from - // `apply_point_put`) aborts, dropping the txn and rolling the whole set - // back — the all-or-nothing guarantee the atomicity test pins. - let txn = match self.sparse.begin_write() { - Ok(t) => t, - Err(e) => return self.response_error(task, e), - }; - // Captured for post-commit event emission. The clone into `write_set` - // below is the only owned body copy actually needed, since `plan` - // doesn't outlive the function but does outlive this loop. - let mut put_events: Vec> = Vec::new(); - let mut affected = 0u64; - // Every row key written into `txn`, pushed BEFORE the write so a row that - // fails mid-apply (its cache entry is populated before the UNIQUE check) - // is evicted on abort too — see `rollback_merge_cache`. - let mut applied_keys: Vec = Vec::new(); - // In-memory (HNSW + R-tree) index deltas applied this pass, reversed on - // any abort path — the redb txn drop only reverses store-backed state. - let mut undo_log: Vec = Vec::new(); - // RETURNING rows for THIS apply attempt: post-images for the UPDATE and - // INSERT arms, pre-images for the DELETE arms. Built fresh here rather - // than carried in, because an attempt that ends in `OllpRetryRequired` - // is fully re-resolved and re-applied by the orchestrator — rows from a - // failed attempt describe a snapshot that never committed. - let mut returned_docs: Vec = Vec::new(); - - for upd in &plan.updates { - match upd.surrogate { - Some(surrogate) => { - let row_key = surrogate_to_doc_id(surrogate); - applied_keys.push(row_key.clone()); - // `apply_point_put`'s vector step APPENDS (it never replaces), - // so an in-place UPDATE must first soft-delete the surrogate's - // prior embedding or the stale vector keeps scoring in KNN - // search. Push each removal as a `DeleteVector` undo BEFORE the - // put's `InsertVector` undos so an abort undeletes the old - // vector after removing the new one (reverse order). - if has_vectors { - for d in self.remove_document_vector_indexes( - database_id, - tid, - params.target_collection, - &row_key, - ) { - undo_log.push(UndoEntry::DeleteVector { - index_key: d.index_key, - vector_id: d.vector_id, - collection: d.collection, - field: d.field, - doc_id: d.doc_id, - }); - } - } - match self.apply_point_put( - &txn, - PointPutParams { - database_id, - tid, - collection: params.target_collection, - document_id: &row_key, - surrogate, - value: &upd.body, - index_text: true, - user_roles: &task.request.user_roles, - enforce: true, - wal_lsn: task.wal_lsn(), - }, - ) { - Ok(mut outcome) => { - record_put_index_undo(&mut undo_log, &mut outcome); - // The arm's materialized-sum delta is folded inside - // the SAME transaction the arm's row lands in, so a - // moved total rolls back with the row that moved it. - // Both images come from the plan: the classifier held - // the pre-image already, so nothing is re-read. - match write_hook::run( - self, - &txn, - &write_hook::HookCtx { - database_id, - tid, - collection: params.target_collection, - resolved_targets: params.resolved_sum_targets, - deferred_sum_targets: &[], - wal_lsn: task.wal_lsn(), - }, - write_hook::WriteImages::Update { - old: write_hook::ImageBody::Submitted(&upd.old_body), - new: write_hook::ImageBody::Submitted(&upd.body), - }, - ) { - Ok(enforcement) => { - write_set.extend(write_hook::target_write_set( - &enforcement.target_writes, - )); - balanced_entries.extend(enforcement.balanced_entries); - } - Err(e) => { - return self.abort_merge_apply(MergeAbort { - task, - database_id, - tid, - collection: params.target_collection, - applied_keys: &applied_keys, - undo_log, - err: e.into(), - }); - } - } - if has_vectors { - write_set.push(WriteSetEntry { - surrogate: surrogate.as_u32(), - is_delete: false, - value: upd.body.clone(), - collection: None, - }); - } - if params.returning.is_some() { - match returning_doc(&upd.body, &row_key) { - Ok(doc) => returned_docs.push(doc), - Err(e) => { - return self.abort_merge_apply(MergeAbort { - task, - database_id, - tid, - collection: params.target_collection, - applied_keys: &applied_keys, - undo_log, - err: e.into(), - }); - } - } - } - put_events.push((row_key, upd.body.as_slice(), outcome.prior_value)); - affected += 1; - } - Err(e) => { - return self.abort_merge_apply(MergeAbort { - task, - database_id, - tid, - collection: params.target_collection, - applied_keys: &applied_keys, - undo_log, - err: e.into(), - }); - } - } - } - None => { - // Legacy non-surrogate target row: raw in-txn body rewrite - // (no cross-engine index — these rows predate surrogate - // keying and were never indexed). - applied_keys.push(upd.doc_id.clone()); - if let Err(e) = self.sparse.put_in_txn( - &txn, - database_id, - tid, - params.target_collection, - &upd.doc_id, - &upd.body, - ) { - return self.abort_merge_apply(MergeAbort { - task, - database_id, - tid, - collection: params.target_collection, - applied_keys: &applied_keys, - undo_log, - err: e.into(), - }); - } - if params.returning.is_some() { - match returning_doc(&upd.body, &upd.doc_id) { - Ok(doc) => returned_docs.push(doc), - Err(e) => { - return self.abort_merge_apply(MergeAbort { - task, - database_id, - tid, - collection: params.target_collection, - applied_keys: &applied_keys, - undo_log, - err: e.into(), - }); - } - } - } - affected += 1; - } - } - } - - for ins in &plan.inserts { - // The verify above proved every insert key has a pre-assigned - // surrogate; the lookup cannot miss, but a missing entry is treated - // as drift rather than unwrapped. - let surrogate = match surrogate_for.get(ins.join_key.as_str()) { - Some(s) => Surrogate(*s), - None => { - return self.abort_merge_apply(MergeAbort { - task, - database_id, - tid, - collection: params.target_collection, - applied_keys: &applied_keys, - undo_log, - err: ErrorCode::OllpRetryRequired, - }); - } - }; - let row_key = surrogate_to_doc_id(surrogate); - applied_keys.push(row_key.clone()); - match self.apply_point_put( - &txn, - PointPutParams { - database_id, - tid, - collection: params.target_collection, - document_id: &row_key, - surrogate, - value: &ins.body, - index_text: true, - user_roles: &task.request.user_roles, - enforce: true, - wal_lsn: task.wal_lsn(), - }, - ) { - Ok(mut outcome) => { - record_put_index_undo(&mut undo_log, &mut outcome); - // A NOT-MATCHED INSERT arm credits its target with the whole - // new row — post-image only, which is exactly what - // `RowImages::Insert` expresses. - match write_hook::run( - self, - &txn, - &write_hook::HookCtx { - database_id, - tid, - collection: params.target_collection, - resolved_targets: params.resolved_sum_targets, - deferred_sum_targets: &[], - wal_lsn: task.wal_lsn(), - }, - write_hook::WriteImages::Insert { - new: write_hook::ImageBody::Submitted(&ins.body), - }, - ) { - Ok(enforcement) => { - write_set - .extend(write_hook::target_write_set(&enforcement.target_writes)); - balanced_entries.extend(enforcement.balanced_entries); - } - Err(e) => { - return self.abort_merge_apply(MergeAbort { - task, - database_id, - tid, - collection: params.target_collection, - applied_keys: &applied_keys, - undo_log, - err: e.into(), - }); - } - } - if has_vectors { - write_set.push(WriteSetEntry { - surrogate: surrogate.as_u32(), - is_delete: false, - value: ins.body.clone(), - collection: None, - }); - } - if params.returning.is_some() { - match returning_doc(&ins.body, &row_key) { - Ok(doc) => returned_docs.push(doc), - Err(e) => { - return self.abort_merge_apply(MergeAbort { - task, - database_id, - tid, - collection: params.target_collection, - applied_keys: &applied_keys, - undo_log, - err: e.into(), - }); - } - } - } - put_events.push((row_key, ins.body.as_slice(), None)); - affected += 1; - } - Err(e) => { - return self.abort_merge_apply(MergeAbort { - task, - database_id, - tid, - collection: params.target_collection, - applied_keys: &applied_keys, - undo_log, - err: e.into(), - }); - } - } - } - - // Every arm of the statement — the UPDATE and INSERT arms folded above - // and the DELETE arms accounted before phase A — is judged once here, - // before the phase-A commit, so a MERGE that leaves a journal group - // unbalanced writes nothing at all. - if let Err(e) = self.settle_balanced_entries( - database_id, - tid, - params.target_collection, - balanced_entries, - ) { - return self.abort_merge_apply(MergeAbort { - task, - database_id, - tid, - collection: params.target_collection, - applied_keys: &applied_keys, - undo_log, - err: e.into(), - }); - } - - if let Err(e) = txn.commit() { - return self.abort_merge_apply(MergeAbort { - task, - database_id, - tid, - collection: params.target_collection, - applied_keys: &applied_keys, - undo_log, - err: ErrorCode::Internal { - detail: format!("merge apply commit: {e}"), - }, - }); - } - self.checkpoint_coordinator - .mark_dirty("sparse", put_events.len()); - - for (row_key, body, prior) in &put_events { - self.emit_put_event( - task, - tid, - params.target_collection, - row_key, - body, - prior.as_deref(), - ); - } - - // Phase B: DELETE arms, applied after the put commit because their - // cascade opens its own transactions. - if let Err(response) = self.apply_merge_delete_arms( - MergeDeleteArms { - task, - database_id, - tid, - collection: params.target_collection, - deletes: &plan.deletes, - has_vectors, - returning: params.returning.is_some(), - resolved_targets: params.resolved_sum_targets, - }, - MergeDeleteTally { - affected: &mut affected, - write_set: &mut write_set, - returned_docs: &mut returned_docs, - }, - ) { - return response; - } - - let mut response = if let Some(spec) = params.returning { - match returning_rows::build_rows_payload(spec, params.rls_filters, &returned_docs) { - Ok(payload) => self.response_with_payload(task, payload), - Err(e) => { - return self.response_error( - task, - ErrorCode::Internal { - detail: format!("RETURNING encode: {e}"), - }, - ); - } - } - } else { - let result = serde_json::json!({ "affected": affected }); - match encode_json_as_msgpack(&result) { - Ok(payload) => self.response_with_payload(task, payload), - Err(e) => { - return self.response_error( - task, - ErrorCode::Internal { - detail: e.to_string(), - }, - ); - } - } - }; - if !write_set.is_empty() { - response.write_set = write_set; - } - response - } -} diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/insert_rows.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/insert_rows.rs new file mode 100644 index 000000000..50cf7cb81 --- /dev/null +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/insert_rows.rs @@ -0,0 +1,198 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The MERGE NOT-MATCHED INSERT arm, applied inside the phase-A transaction +//! shared with the UPDATE arm. + +use std::collections::HashMap; + +use redb::WriteTransaction; + +use crate::bridge::envelope::{ErrorCode, Response, WriteSetEntry}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::enforcement::balanced::BalancedEntry; +use crate::data::executor::enforcement::write_hook; +use crate::data::executor::handlers::point::apply_put::PointPutParams; +use crate::data::executor::handlers::transaction::undo::UndoEntry; +use crate::data::executor::task::ExecutionTask; +use crate::engine::document::store::surrogate_to_doc_id; +use nodedb_types::Surrogate; + +use super::super::abort::MergeAbort; +use super::super::apply_support::{MergePutEvent, record_put_index_undo, returning_doc}; +use super::super::plan::MergeInsert; + +/// Read-only context the INSERT arm needs from the shared apply pass. +pub(super) struct InsertRowsCtx<'a> { + pub(super) task: &'a ExecutionTask, + pub(super) database_id: u64, + pub(super) tid: u64, + pub(super) collection: &'a str, + /// Whether the target maintains a secondary vector index. + pub(super) has_vectors: bool, + /// Whether the statement carries a `RETURNING` projection. + pub(super) returning: bool, + pub(super) resolved_sum_targets: &'a [nodedb_physical::physical_plan::ResolvedSumTarget], + /// Source join value → Control-Plane-pre-assigned surrogate, verified + /// against `inserts` by the caller before this arm runs. + pub(super) surrogate_for: &'a HashMap<&'a str, u32>, +} + +/// Mutable accumulators the INSERT arm folds into. Owned by the caller for +/// the whole apply pass and borrowed here rather than cloned. +pub(super) struct InsertRowsTally<'a, 'p> { + pub(super) affected: &'a mut u64, + pub(super) applied_keys: &'a mut Vec, + pub(super) undo_log: &'a mut Vec, + pub(super) put_events: &'a mut Vec>, + pub(super) write_set: &'a mut Vec, + pub(super) balanced_entries: &'a mut Vec, + pub(super) returned_docs: &'a mut Vec, +} + +impl CoreLoop { + /// Apply every NOT-MATCHED INSERT arm inside the caller's shared write + /// transaction. `Err(response)` is the terminating error response the + /// caller must return as-is — the transaction is left uncommitted for the + /// caller to abort. + pub(super) fn apply_merge_insert_arm<'p>( + &mut self, + txn: &WriteTransaction, + inserts: &'p [MergeInsert], + ctx: InsertRowsCtx<'_>, + tally: InsertRowsTally<'_, 'p>, + ) -> Result<(), Response> { + let InsertRowsCtx { + task, + database_id, + tid, + collection, + has_vectors, + returning, + resolved_sum_targets, + surrogate_for, + } = ctx; + let InsertRowsTally { + affected, + applied_keys, + undo_log, + put_events, + write_set, + balanced_entries, + returned_docs, + } = tally; + + for ins in inserts { + // The verify above proved every insert key has a pre-assigned + // surrogate; the lookup cannot miss, but a missing entry is treated + // as drift rather than unwrapped. + let surrogate = match surrogate_for.get(ins.join_key.as_str()) { + Some(s) => Surrogate(*s), + None => { + return Err(self.abort_merge_apply(MergeAbort { + task, + database_id, + tid, + collection, + applied_keys: applied_keys.as_slice(), + undo_log: std::mem::take(undo_log), + err: ErrorCode::OllpRetryRequired, + })); + } + }; + let row_key = surrogate_to_doc_id(surrogate); + applied_keys.push(row_key.clone()); + match self.apply_point_put( + txn, + PointPutParams { + database_id, + tid, + collection, + document_id: &row_key, + surrogate, + value: &ins.body, + index_text: true, + user_roles: &task.request.user_roles, + enforce: true, + wal_lsn: task.wal_lsn(), + }, + ) { + Ok(mut outcome) => { + record_put_index_undo(undo_log, &mut outcome); + // A NOT-MATCHED INSERT arm credits its target with the whole + // new row — post-image only, which is exactly what + // `RowImages::Insert` expresses. + match write_hook::run( + self, + txn, + &write_hook::HookCtx { + database_id, + tid, + collection, + resolved_targets: resolved_sum_targets, + deferred_sum_targets: &[], + wal_lsn: task.wal_lsn(), + }, + write_hook::WriteImages::Insert { + new: write_hook::ImageBody::Submitted(&ins.body), + }, + ) { + Ok(enforcement) => { + write_set + .extend(write_hook::target_write_set(&enforcement.target_writes)); + balanced_entries.extend(enforcement.balanced_entries); + } + Err(e) => { + return Err(self.abort_merge_apply(MergeAbort { + task, + database_id, + tid, + collection, + applied_keys: applied_keys.as_slice(), + undo_log: std::mem::take(undo_log), + err: e.into(), + })); + } + } + if has_vectors { + write_set.push(WriteSetEntry { + surrogate: surrogate.as_u32(), + is_delete: false, + value: ins.body.clone(), + collection: None, + }); + } + if returning { + match returning_doc(&ins.body, &row_key) { + Ok(doc) => returned_docs.push(doc), + Err(e) => { + return Err(self.abort_merge_apply(MergeAbort { + task, + database_id, + tid, + collection, + applied_keys: applied_keys.as_slice(), + undo_log: std::mem::take(undo_log), + err: e.into(), + })); + } + } + } + put_events.push((row_key, ins.body.as_slice(), None)); + *affected += 1; + } + Err(e) => { + return Err(self.abort_merge_apply(MergeAbort { + task, + database_id, + tid, + collection, + applied_keys: applied_keys.as_slice(), + undo_log: std::mem::take(undo_log), + err: e.into(), + })); + } + } + } + Ok(()) + } +} diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/mod.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/mod.rs new file mode 100644 index 000000000..f71734991 --- /dev/null +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/mod.rs @@ -0,0 +1,17 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! MERGE APPLY pass: verify the resolve→apply prediction, then atomically +//! apply every arm's writes with the Control-Plane-pre-assigned surrogates. +//! +//! This directory owns the drift verification and the single redb write +//! transaction the UPDATE and INSERT arms share. The two things that cannot +//! live under that transaction have their own sibling files: unwinding a +//! partial apply (`abort`) and the DELETE arms, whose cascade opens +//! transactions of its own and therefore runs after the commit +//! (`delete_arms`). `orchestrate` owns the setup and the commit, and calls +//! `update_rows` and `insert_rows` for the two arms that share the +//! transaction. + +mod insert_rows; +mod orchestrate; +mod update_rows; diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/orchestrate.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/orchestrate.rs new file mode 100644 index 000000000..0ba3a6027 --- /dev/null +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/orchestrate.rs @@ -0,0 +1,278 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `execute_merge_apply`: verify the resolve→apply prediction, run the +//! UPDATE and INSERT arms inside one shared write transaction, commit, then +//! run the DELETE arms and build the response. + +use std::collections::HashMap; + +use crate::bridge::envelope::{ErrorCode, Response, WriteSetEntry}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::merge::MergeParams; +use crate::data::executor::handlers::returning_rows; +use crate::data::executor::handlers::transaction::undo::UndoEntry; +use crate::data::executor::response_codec::encode_json_as_msgpack; +use crate::data::executor::task::ExecutionTask; + +use super::super::abort::MergeAbort; +use super::super::apply_support::{MergePutEvent, gate_merge_arms}; +use super::super::delete_arms::{MergeDeleteArms, MergeDeleteTally}; +use super::insert_rows::{InsertRowsCtx, InsertRowsTally}; +use super::update_rows::{UpdateRowsCtx, UpdateRowsTally}; + +impl CoreLoop { + /// APPLY pass: verify the resolve→apply prediction, then atomically apply. + pub(in crate::data::executor) fn execute_merge_apply( + &mut self, + task: &ExecutionTask, + tid: u64, + params: MergeParams<'_>, + ) -> Response { + let resolved = match params.resolved_inserts { + Some(r) => r, + None => { + return self.response_error( + task, + ErrorCode::Internal { + detail: "merge apply invoked without resolved inserts".into(), + }, + ); + } + }; + let database_id = task.request.database_id.as_u64(); + + let plan = match self.collect_merge_plan(database_id, tid, task.request.txn_id, ¶ms) { + Ok(p) => p, + Err(e) => return self.response_error(task, e), + }; + + // TOCTOU verification: the recomputed NOT-MATCHED insert-key set must + // still equal the orchestrator's predicted set. Any drift (a target row + // for a predicted-insert key appeared, or a matched row vanished) means + // the pre-assigned surrogates no longer describe the merge — return + // OllpRetryRequired WITHOUT writing so the orchestrator re-resolves. + let mut actual_keys: Vec<&str> = plan.inserts.iter().map(|i| i.join_key.as_str()).collect(); + actual_keys.sort_unstable(); + let mut predicted_keys: Vec<&str> = resolved.iter().map(|(k, _)| k.as_str()).collect(); + predicted_keys.sort_unstable(); + if actual_keys != predicted_keys { + return self.response_error(task, ErrorCode::OllpRetryRequired); + } + let surrogate_for: HashMap<&str, u32> = + resolved.iter().map(|(k, s)| (k.as_str(), *s)).collect(); + + // Whether the target maintains a secondary vector index. Gated ONCE here + // (the schemaless half scans `vector_params` unindexed) and threaded into + // the per-row UPDATE re-index below. + let has_vectors = self.collection_has_vectors(database_id, tid, params.target_collection); + + // Gate every arm on the target's write policy BEFORE the apply + // transaction opens, so a rejected row leaves nothing written and + // nothing to unwind. + if let Err(e) = + gate_merge_arms(&plan, params.rls_write_check, tid, params.target_collection) + { + return self.response_error(task, e); + } + + // One post-apply redo entry per indexed row — a `Put` for each + // UPDATE/INSERT post-image, a `Delete` for each removed row — carried + // back so the Control Plane mints the durable WAL redo the vector index + // needs to survive a WAL-only restart. Empty on non-vector targets. + let mut write_set: Vec = Vec::new(); + + // The whole MERGE is ONE boundary, so its DELETE arms are accounted + // here, before any phase runs: those arms apply in their own + // transactions AFTER the phase-A commit, so entries collected as they + // ran could only report a violation phase A had already made durable. + // Their pre-images are the plan's captured bodies, which the classifier + // already holds — nothing is re-read. + let delete_bodies: Vec<&[u8]> = plan.deletes.iter().map(|d| d.body.as_slice()).collect(); + let mut balanced_entries = self.balanced_entries_for_submitted_deletes( + database_id, + tid, + params.target_collection, + &delete_bodies, + ); + + // Phase A: matched UPDATE + NOT-MATCHED INSERT share ONE redb write + // transaction. Any per-row error (including a UNIQUE violation from + // `apply_point_put`) aborts, dropping the txn and rolling the whole set + // back — the all-or-nothing guarantee the atomicity test pins. + let txn = match self.sparse.begin_write() { + Ok(t) => t, + Err(e) => return self.response_error(task, e), + }; + // Captured for post-commit event emission. The clone into `write_set` + // below is the only owned body copy actually needed, since `plan` + // doesn't outlive the function but does outlive this loop. + let mut put_events: Vec> = Vec::new(); + let mut affected = 0u64; + // Every row key written into `txn`, pushed BEFORE the write so a row that + // fails mid-apply (its cache entry is populated before the UNIQUE check) + // is evicted on abort too — see `rollback_merge_cache`. + let mut applied_keys: Vec = Vec::new(); + // In-memory (HNSW + R-tree) index deltas applied this pass, reversed on + // any abort path — the redb txn drop only reverses store-backed state. + let mut undo_log: Vec = Vec::new(); + // RETURNING rows for THIS apply attempt: post-images for the UPDATE and + // INSERT arms, pre-images for the DELETE arms. Built fresh here rather + // than carried in, because an attempt that ends in `OllpRetryRequired` + // is fully re-resolved and re-applied by the orchestrator — rows from a + // failed attempt describe a snapshot that never committed. + let mut returned_docs: Vec = Vec::new(); + + if let Err(response) = self.apply_merge_update_arm( + &txn, + &plan.updates, + UpdateRowsCtx { + task, + database_id, + tid, + collection: params.target_collection, + has_vectors, + returning: params.returning.is_some(), + resolved_sum_targets: params.resolved_sum_targets, + }, + UpdateRowsTally { + affected: &mut affected, + applied_keys: &mut applied_keys, + undo_log: &mut undo_log, + put_events: &mut put_events, + write_set: &mut write_set, + balanced_entries: &mut balanced_entries, + returned_docs: &mut returned_docs, + }, + ) { + return response; + } + + if let Err(response) = self.apply_merge_insert_arm( + &txn, + &plan.inserts, + InsertRowsCtx { + task, + database_id, + tid, + collection: params.target_collection, + has_vectors, + returning: params.returning.is_some(), + resolved_sum_targets: params.resolved_sum_targets, + surrogate_for: &surrogate_for, + }, + InsertRowsTally { + affected: &mut affected, + applied_keys: &mut applied_keys, + undo_log: &mut undo_log, + put_events: &mut put_events, + write_set: &mut write_set, + balanced_entries: &mut balanced_entries, + returned_docs: &mut returned_docs, + }, + ) { + return response; + } + + // Every arm of the statement — the UPDATE and INSERT arms folded above + // and the DELETE arms accounted before phase A — is judged once here, + // before the phase-A commit, so a MERGE that leaves a journal group + // unbalanced writes nothing at all. + if let Err(e) = self.settle_balanced_entries( + database_id, + tid, + params.target_collection, + balanced_entries, + ) { + return self.abort_merge_apply(MergeAbort { + task, + database_id, + tid, + collection: params.target_collection, + applied_keys: &applied_keys, + undo_log, + err: e.into(), + }); + } + + if let Err(e) = txn.commit() { + return self.abort_merge_apply(MergeAbort { + task, + database_id, + tid, + collection: params.target_collection, + applied_keys: &applied_keys, + undo_log, + err: ErrorCode::Internal { + detail: format!("merge apply commit: {e}"), + }, + }); + } + self.checkpoint_coordinator + .mark_dirty("sparse", put_events.len()); + + for (row_key, body, prior) in &put_events { + let identity = crate::engine::document::store::identity_of(row_key); + self.emit_put_event( + task, + tid, + params.target_collection, + identity, + body, + prior.as_deref(), + ); + } + + // Phase B: DELETE arms, applied after the put commit because their + // cascade opens its own transactions. + if let Err(response) = self.apply_merge_delete_arms( + MergeDeleteArms { + task, + database_id, + tid, + collection: params.target_collection, + deletes: &plan.deletes, + has_vectors, + returning: params.returning.is_some(), + resolved_targets: params.resolved_sum_targets, + }, + MergeDeleteTally { + affected: &mut affected, + write_set: &mut write_set, + returned_docs: &mut returned_docs, + }, + ) { + return response; + } + + let mut response = if let Some(spec) = params.returning { + match returning_rows::build_rows_payload(spec, params.rls_filters, &returned_docs) { + Ok(payload) => self.response_with_payload(task, payload), + Err(e) => { + return self.response_error( + task, + ErrorCode::Internal { + detail: format!("RETURNING encode: {e}"), + }, + ); + } + } + } else { + let result = serde_json::json!({ "affected": affected }); + match encode_json_as_msgpack(&result) { + Ok(payload) => self.response_with_payload(task, payload), + Err(e) => { + return self.response_error( + task, + ErrorCode::Internal { + detail: e.to_string(), + }, + ); + } + } + }; + if !write_set.is_empty() { + response.write_set = write_set; + } + response + } +} diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs new file mode 100644 index 000000000..e402160a9 --- /dev/null +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs @@ -0,0 +1,246 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The MERGE UPDATE arm (matched + not-matched-by-source), applied inside the +//! phase-A transaction shared with the INSERT arm. + +use redb::WriteTransaction; + +use crate::bridge::envelope::{Response, WriteSetEntry}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::enforcement::balanced::BalancedEntry; +use crate::data::executor::enforcement::write_hook; +use crate::data::executor::handlers::point::apply_put::PointPutParams; +use crate::data::executor::handlers::transaction::undo::UndoEntry; +use crate::data::executor::task::ExecutionTask; +use crate::engine::document::store::surrogate_to_doc_id; + +use super::super::abort::MergeAbort; +use super::super::apply_support::{MergePutEvent, record_put_index_undo, returning_doc}; +use super::super::plan::MergeUpdate; + +/// Read-only context the UPDATE arm needs from the shared apply pass. +pub(super) struct UpdateRowsCtx<'a> { + pub(super) task: &'a ExecutionTask, + pub(super) database_id: u64, + pub(super) tid: u64, + pub(super) collection: &'a str, + /// Whether the target maintains a secondary vector index. Gated once by + /// the caller and threaded into the per-row re-index below. + pub(super) has_vectors: bool, + /// Whether the statement carries a `RETURNING` projection. + pub(super) returning: bool, + pub(super) resolved_sum_targets: &'a [nodedb_physical::physical_plan::ResolvedSumTarget], +} + +/// Mutable accumulators the UPDATE arm folds into. Owned by the caller for +/// the whole apply pass and borrowed here rather than cloned. +pub(super) struct UpdateRowsTally<'a, 'p> { + pub(super) affected: &'a mut u64, + pub(super) applied_keys: &'a mut Vec, + pub(super) undo_log: &'a mut Vec, + pub(super) put_events: &'a mut Vec>, + pub(super) write_set: &'a mut Vec, + pub(super) balanced_entries: &'a mut Vec, + pub(super) returned_docs: &'a mut Vec, +} + +impl CoreLoop { + /// Apply every UPDATE arm (matched + not-matched-by-source) inside the + /// caller's shared write transaction. `Err(response)` is the terminating + /// error response the caller must return as-is — the transaction is left + /// uncommitted for the caller to abort. + pub(super) fn apply_merge_update_arm<'p>( + &mut self, + txn: &WriteTransaction, + updates: &'p [MergeUpdate], + ctx: UpdateRowsCtx<'_>, + tally: UpdateRowsTally<'_, 'p>, + ) -> Result<(), Response> { + let UpdateRowsCtx { + task, + database_id, + tid, + collection, + has_vectors, + returning, + resolved_sum_targets, + } = ctx; + let UpdateRowsTally { + affected, + applied_keys, + undo_log, + put_events, + write_set, + balanced_entries, + returned_docs, + } = tally; + + for upd in updates { + match upd.surrogate { + Some(surrogate) => { + let row_key = surrogate_to_doc_id(surrogate); + applied_keys.push(row_key.clone()); + // `apply_point_put`'s vector step APPENDS (it never replaces), + // so an in-place UPDATE must first soft-delete the surrogate's + // prior embedding or the stale vector keeps scoring in KNN + // search. Push each removal as a `DeleteVector` undo BEFORE the + // put's `InsertVector` undos so an abort undeletes the old + // vector after removing the new one (reverse order). + if has_vectors { + for d in self.remove_document_vector_indexes( + database_id, + tid, + collection, + &row_key, + ) { + undo_log.push(UndoEntry::DeleteVector { + index_key: d.index_key, + vector_id: d.vector_id, + collection: d.collection, + field: d.field, + doc_id: d.doc_id, + }); + } + } + match self.apply_point_put( + txn, + PointPutParams { + database_id, + tid, + collection, + document_id: &row_key, + surrogate, + value: &upd.body, + index_text: true, + user_roles: &task.request.user_roles, + enforce: true, + wal_lsn: task.wal_lsn(), + }, + ) { + Ok(mut outcome) => { + record_put_index_undo(undo_log, &mut outcome); + // The arm's materialized-sum delta is folded inside + // the SAME transaction the arm's row lands in, so a + // moved total rolls back with the row that moved it. + // Both images come from the plan: the classifier held + // the pre-image already, so nothing is re-read. + match write_hook::run( + self, + txn, + &write_hook::HookCtx { + database_id, + tid, + collection, + resolved_targets: resolved_sum_targets, + deferred_sum_targets: &[], + wal_lsn: task.wal_lsn(), + }, + write_hook::WriteImages::Update { + old: write_hook::ImageBody::Submitted(&upd.old_body), + new: write_hook::ImageBody::Submitted(&upd.body), + }, + ) { + Ok(enforcement) => { + write_set.extend(write_hook::target_write_set( + &enforcement.target_writes, + )); + balanced_entries.extend(enforcement.balanced_entries); + } + Err(e) => { + return Err(self.abort_merge_apply(MergeAbort { + task, + database_id, + tid, + collection, + applied_keys: applied_keys.as_slice(), + undo_log: std::mem::take(undo_log), + err: e.into(), + })); + } + } + if has_vectors { + write_set.push(WriteSetEntry { + surrogate: surrogate.as_u32(), + is_delete: false, + value: upd.body.clone(), + collection: None, + }); + } + if returning { + match returning_doc(&upd.body, &row_key) { + Ok(doc) => returned_docs.push(doc), + Err(e) => { + return Err(self.abort_merge_apply(MergeAbort { + task, + database_id, + tid, + collection, + applied_keys: applied_keys.as_slice(), + undo_log: std::mem::take(undo_log), + err: e.into(), + })); + } + } + } + put_events.push((row_key, upd.body.as_slice(), outcome.prior_value)); + *affected += 1; + } + Err(e) => { + return Err(self.abort_merge_apply(MergeAbort { + task, + database_id, + tid, + collection, + applied_keys: applied_keys.as_slice(), + undo_log: std::mem::take(undo_log), + err: e.into(), + })); + } + } + } + None => { + // Legacy non-surrogate target row: raw in-txn body rewrite + // (no cross-engine index — these rows predate surrogate + // keying and were never indexed). + applied_keys.push(upd.doc_id.clone()); + if let Err(e) = self.sparse.put_in_txn( + txn, + database_id, + tid, + collection, + &upd.doc_id, + &upd.body, + ) { + return Err(self.abort_merge_apply(MergeAbort { + task, + database_id, + tid, + collection, + applied_keys: applied_keys.as_slice(), + undo_log: std::mem::take(undo_log), + err: e.into(), + })); + } + if returning { + match returning_doc(&upd.body, &upd.doc_id) { + Ok(doc) => returned_docs.push(doc), + Err(e) => { + return Err(self.abort_merge_apply(MergeAbort { + task, + database_id, + tid, + collection, + applied_keys: applied_keys.as_slice(), + undo_log: std::mem::take(undo_log), + err: e.into(), + })); + } + } + } + *affected += 1; + } + } + } + Ok(()) + } +} diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/apply_support.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/apply_support.rs index 75f11ee6e..63e81af1e 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/apply_support.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/apply_support.rs @@ -62,7 +62,10 @@ pub(super) fn gate_merge_arms( ) { return Ok(()); } - let arms = plan + // Updates and deletes name their row by its storage key; inserts name + // theirs by the source join value, an engine-native key that is never a + // document storage key. + let doc_arms = plan .updates .iter() .map(|u| (u.body.as_slice(), u.doc_id.as_str())) @@ -70,14 +73,22 @@ pub(super) fn gate_merge_arms( plan.deletes .iter() .map(|d| (d.body.as_slice(), d.doc_id.as_str())), - ) - .chain( - plan.inserts - .iter() - .map(|i| (i.body.as_slice(), i.join_key.as_str())), ); - for (body, row_id) in arms { - rls_write_gate::admit_stored_row(rls_write_check, body, row_id, None, tid, collection)?; + for (body, doc_id) in doc_arms { + let identity = crate::engine::document::store::identity_of(doc_id); + rls_write_gate::admit_stored_row(rls_write_check, body, &identity, None, tid, collection)?; + } + for insert in &plan.inserts { + let identity = + crate::engine::document::store::RowIdentity::from_user_key(insert.join_key.as_str()); + rls_write_gate::admit_stored_row( + rls_write_check, + &insert.body, + &identity, + None, + tid, + collection, + )?; } Ok(()) } @@ -86,10 +97,15 @@ pub(super) fn gate_merge_arms( /// reads. Same shape the point and bulk DML RETURNING paths emit, so a MERGE /// row projects identically. /// +/// `doc_id` is the row's storage key, every caller's `MergeUpdate::doc_id`, +/// `MergeDelete::doc_id`, or a minted insert key from `surrogate_to_doc_id`. +/// This function converts it to the client-visible identity before decoding. +/// /// The schema argument is `None` unconditionally: a merge plan's captured /// bodies are MessagePack for BOTH storage modes (`collect_merge_plan` decodes /// a strict target's Binary Tuple and re-encodes the resolved row before the /// apply pass ever sees it), so the strict decoder would have nothing to read. pub(super) fn returning_doc(body: &[u8], doc_id: &str) -> crate::Result { - super::super::returning_doc::from_stored(body, doc_id, None) + let identity = crate::engine::document::store::identity_of(doc_id); + super::super::returning_doc::from_stored(body, &identity, None) } diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs index c963a41ca..79c33a69f 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs @@ -17,7 +17,7 @@ use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::enforcement::write_hook; use crate::data::executor::handlers::point::apply_delete::PointDeleteParams; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::StorageKey; use super::apply_support::returning_doc; use super::plan::MergeDelete; @@ -160,13 +160,10 @@ impl CoreLoop { }); } } - let row_key = surrogate_to_doc_id(surrogate); - self.emit_write_event( + self.emit_document_delete_event( task, collection, - crate::event::WriteOp::Delete, - &row_key, - None, + StorageKey::for_surrogate(surrogate).to_identity(), outcome.prior_value.as_deref(), ); } diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/plan.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/plan.rs index cde8b600b..be17625b1 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/plan.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/plan.rs @@ -9,7 +9,7 @@ use std::collections::HashSet; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::doc_format; use crate::data::executor::doc_format::encode_resolved_wire_body as encode_doc_body; -use crate::engine::document::store::doc_id_to_surrogate; +use crate::engine::document::store::{RowIdentity, doc_id_to_surrogate}; use nodedb_physical::physical_plan::document::merge_types::{ MergeActionOp, MergeClauseKind as MergeClauseKindOp, }; @@ -66,12 +66,13 @@ pub(super) struct MergePlanActions { /// duplicate of a row that already exists. /// /// A schemaless collection with no declared `id` field carries its identity -/// only in `doc_id` (the storage key), never in the body — so a MERGE arm's -/// `AND id ...` condition, matched by [`find_arm`], must see the identity -/// injected here. A strict row already surfaces `id` as a real tuple column, -/// so injection only runs on the schemaless arm. +/// only in the row's storage key, never in the body — so a MERGE arm's +/// `AND id ...` condition, matched by [`find_arm`], must see the row's +/// client-visible identity injected here, never the raw storage key. A +/// strict row already surfaces `id` as a real tuple column, so injection +/// only runs on the schemaless arm. fn decode_target( - doc_id: &str, + identity: &RowIdentity, bytes: &[u8], strict_schema: &Option, ) -> crate::Result { @@ -86,7 +87,7 @@ fn decode_target( { obj.insert( "id".to_string(), - serde_json::Value::String(doc_id.to_string()), + serde_json::Value::String(identity.as_str().to_string()), ); } Ok(doc) @@ -123,12 +124,16 @@ impl CoreLoop { let null_source = serde_json::Value::Null; for (doc_id, bytes) in &target_docs { - let target_doc = decode_target(doc_id, bytes, &strict_schema)?; + let surrogate = doc_id_to_surrogate(doc_id); + let identity = match surrogate { + Some(s) => RowIdentity::for_surrogate(s), + None => RowIdentity::from_user_key(doc_id.clone()), + }; + let target_doc = decode_target(&identity, bytes, &strict_schema)?; let join_val = target_doc .get(params.target_join_col) .map(json_to_str) .unwrap_or_default(); - let surrogate = doc_id_to_surrogate(doc_id); let (arm_kind, source_doc): (MergeClauseKindOp, &serde_json::Value) = if let Some(source_doc) = source_map.get(&join_val) { diff --git a/nodedb/src/data/executor/handlers/point/delete.rs b/nodedb/src/data/executor/handlers/point/delete.rs index 923c57170..96e9d681f 100644 --- a/nodedb/src/data/executor/handlers/point/delete.rs +++ b/nodedb/src/data/executor/handlers/point/delete.rs @@ -173,6 +173,8 @@ impl CoreLoop { // through so CDC/trigger consumers see the pre-delete state as // `old_value`. A delete against a non-existent key is a true // no-op and emits nothing. + let document_identity = + crate::engine::document::store::RowIdentity::from_user_key(document_id); if let Some(prior_bytes) = prior.as_deref() { let old_converted = self.resolve_event_payload( task.request.database_id.as_u64(), @@ -180,12 +182,13 @@ impl CoreLoop { collection, prior_bytes, ); - self.emit_write_event( + // `document_identity` is read again below for `RETURNING`'s `id` + // field, so the event-emit boundary gets a clone rather than the + // move. + self.emit_document_delete_event( task, collection, - crate::event::WriteOp::Delete, - document_id, - None, + document_identity.clone(), Some(old_converted.as_deref().unwrap_or(prior_bytes)), ); } @@ -208,7 +211,7 @@ impl CoreLoop { StorageMode::Strict { schema } => Some(schema), StorageMode::Schemaless => None, }); - returning_doc::from_stored(prior_bytes, document_id, strict_schema) + returning_doc::from_stored(prior_bytes, &document_identity, strict_schema) }; let doc = match doc { Ok(doc) => doc, @@ -274,7 +277,8 @@ impl CoreLoop { return Ok(()); } let database_id = task.request.database_id.as_u64(); - let row_key = crate::engine::document::store::surrogate_to_doc_id(surrogate); + let storage_key = crate::engine::document::store::StorageKey::for_surrogate(surrogate); + let row_key = storage_key.to_string(); let row_key = row_key.as_str(); let stored = if self.is_bitemporal(database_id, tid, collection) { self.sparse @@ -299,7 +303,7 @@ impl CoreLoop { rls_write_gate::admit_stored_row( rls_write_check, &body, - row_key, + &storage_key.to_identity(), strict_schema, tid, collection, diff --git a/nodedb/src/data/executor/handlers/point/get.rs b/nodedb/src/data/executor/handlers/point/get.rs index 9b682a3d8..75108c956 100644 --- a/nodedb/src/data/executor/handlers/point/get.rs +++ b/nodedb/src/data/executor/handlers/point/get.rs @@ -149,7 +149,7 @@ impl CoreLoop { let transcoded = { let normalized = sparse_body_to_msgpack(&data, body_format.as_format_ref()); if !rls_filters.is_empty() { - let (_, gated) = sparse_row_to_doc(document_id, &data, body_format.as_format_ref()); + let (_, gated) = sparse_row_to_doc(row_key, &data, body_format.as_format_ref()); if !super::super::rls_eval::rls_check_msgpack_bytes(rls_filters, &gated) { return self.response_with_payload(task, Vec::new()); } diff --git a/nodedb/src/data/executor/handlers/point/insert.rs b/nodedb/src/data/executor/handlers/point/insert.rs index a90619f2c..7e4b1a9f3 100644 --- a/nodedb/src/data/executor/handlers/point/insert.rs +++ b/nodedb/src/data/executor/handlers/point/insert.rs @@ -16,7 +16,7 @@ use crate::data::executor::enforcement::chain_guard::{self, ChainGuard}; use crate::data::executor::enforcement::write_hook::{self, HookCtx, ImageBody, WriteImages}; use crate::data::executor::handlers::point::apply_put::PointPutParams; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::{RowIdentity, StorageKey}; use nodedb_physical::physical_plan::{ResolvedSumTarget, ReturningSpec}; use nodedb_types::Surrogate; @@ -65,8 +65,10 @@ impl CoreLoop { resolved_sum_targets, deferred_sum_targets, } = p; - let row_key = surrogate_to_doc_id(surrogate); + let storage_key = StorageKey::for_surrogate(surrogate); + let row_key = storage_key.to_string(); let row_key = row_key.as_str(); + let document_identity = RowIdentity::from_user_key(document_id); debug!( core = self.core_id, %collection, %document_id, if_absent, @@ -267,7 +269,14 @@ impl CoreLoop { // only writes the document; it no longer derives edges (which mis-homed // cross-shard edges by the document's vShard). - self.emit_put_event(task, tid, collection, row_key, value, None); + self.emit_put_event( + task, + tid, + collection, + storage_key.to_identity(), + value, + None, + ); let mut response = if let Some(spec) = returning { let strict_schema = self.strict_schema_for( @@ -280,7 +289,7 @@ impl CoreLoop { spec, rls_filters, strict_schema.as_ref(), - &[(document_id, stored_value.as_slice())], + &[(&document_identity, stored_value.as_slice())], ) } else { // The row was inserted: exactly one row affected. @@ -299,7 +308,7 @@ mod tests { use crate::bridge::envelope::Status; use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; use crate::data::executor::doc_format; - use crate::engine::document::store::CollectionConfig; + use crate::engine::document::store::{CollectionConfig, surrogate_to_doc_id}; use crate::types::{DatabaseId, TenantId}; const DB: u64 = 0; diff --git a/nodedb/src/data/executor/handlers/point/put.rs b/nodedb/src/data/executor/handlers/point/put.rs index 3878bc274..01684d98c 100644 --- a/nodedb/src/data/executor/handlers/point/put.rs +++ b/nodedb/src/data/executor/handlers/point/put.rs @@ -11,7 +11,7 @@ use crate::data::executor::enforcement::chain_guard::{self, ChainGuard}; use crate::data::executor::enforcement::write_hook::{self, HookCtx, ImageBody, WriteImages}; use crate::data::executor::handlers::point::apply_put::PointPutParams; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::{RowIdentity, StorageKey}; use nodedb_physical::physical_plan::{ResolvedSumTarget, ReturningSpec}; use nodedb_types::Surrogate; @@ -48,8 +48,10 @@ impl CoreLoop { rls_filters, resolved_sum_targets, } = params; - let row_key = surrogate_to_doc_id(surrogate); + let storage_key = StorageKey::for_surrogate(surrogate); + let row_key = storage_key.to_string(); let row_key = row_key.as_str(); + let document_identity = RowIdentity::from_user_key(document_id); debug!(core = self.core_id, %collection, %document_id, "point put"); let database_id = task.request.database_id.as_u64(); @@ -189,7 +191,7 @@ impl CoreLoop { task, tid, collection, - row_key, + storage_key.to_identity(), value, prior.prior_value.as_deref(), ); @@ -205,7 +207,7 @@ impl CoreLoop { spec, rls_filters, strict_schema.as_ref(), - &[(document_id, prior.stored_value.as_slice())], + &[(&document_identity, prior.stored_value.as_slice())], ) } else { // An upsert always writes the row, whether or not one was there before. @@ -224,7 +226,7 @@ mod tests { use crate::bridge::envelope::Status; use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; use crate::data::executor::doc_format; - use crate::engine::document::store::CollectionConfig; + use crate::engine::document::store::{CollectionConfig, surrogate_to_doc_id}; use crate::types::{DatabaseId, TenantId}; const DB: u64 = 0; diff --git a/nodedb/src/data/executor/handlers/point/update/exec.rs b/nodedb/src/data/executor/handlers/point/update/exec.rs index c1ac1d1e0..27ca912f9 100644 --- a/nodedb/src/data/executor/handlers/point/update/exec.rs +++ b/nodedb/src/data/executor/handlers/point/update/exec.rs @@ -18,7 +18,7 @@ use crate::data::executor::handlers::returning_doc; use crate::data::executor::handlers::returning_rows; use crate::data::executor::handlers::rls_write_gate; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::{RowIdentity, StorageKey}; use nodedb_physical::physical_plan::{ResolvedSumTarget, ReturningSpec, StorageMode, UpdateValue}; use nodedb_types::Surrogate; @@ -67,8 +67,10 @@ impl CoreLoop { resolved_sum_targets, declared_primary_key, } = params; - let row_key = surrogate_to_doc_id(surrogate); + let storage_key = StorageKey::for_surrogate(surrogate); + let row_key = storage_key.to_string(); let row_key = row_key.as_str(); + let document_identity = RowIdentity::from_user_key(document_id); debug!( core = self.core_id, %collection, @@ -173,7 +175,7 @@ impl CoreLoop { if let Err(e) = rls_write_gate::admit_stored_row( rls_write_check, &updated_bytes, - document_id, + &document_identity, strict_schema.as_ref(), tid, collection, @@ -229,7 +231,7 @@ impl CoreLoop { task, tid, collection, - row_key, + storage_key.to_identity(), &updated_bytes, Some(¤t_bytes), ); @@ -249,7 +251,7 @@ impl CoreLoop { // as `id` when the row declares none of its own. let doc = match returning_doc::from_stored( &updated_bytes, - document_id, + &document_identity, strict_schema.as_ref(), ) { Ok(doc) => doc, diff --git a/nodedb/src/data/executor/handlers/returning_doc.rs b/nodedb/src/data/executor/handlers/returning_doc.rs index cfed28aa5..0a01f12c3 100644 --- a/nodedb/src/data/executor/handlers/returning_doc.rs +++ b/nodedb/src/data/executor/handlers/returning_doc.rs @@ -20,18 +20,23 @@ use nodedb_types::columnar::StrictSchema; use crate::data::executor::doc_format; use crate::data::executor::strict_format; +use crate::engine::document::store::RowIdentity; -/// Set `id` to the row's storage key unless the document already carries one. +/// Set `id` to the row's client-visible identity unless the document already +/// carries one. /// /// For callers that already hold the decoded document (the update paths /// re-project the image they just built rather than re-reading storage). -pub(in crate::data::executor) fn attach_row_id(doc: &mut serde_json::Value, doc_id: &str) { +pub(in crate::data::executor) fn attach_row_id( + doc: &mut serde_json::Value, + identity: &RowIdentity, +) { if let Some(obj) = doc.as_object_mut() && !obj.contains_key("id") { obj.insert( "id".to_string(), - serde_json::Value::String(doc_id.to_string()), + serde_json::Value::String(identity.as_str().to_string()), ); } } @@ -49,7 +54,7 @@ pub(in crate::data::executor) fn attach_row_id(doc: &mut serde_json::Value, doc_ /// row count as the truth. pub(in crate::data::executor) fn from_stored( body: &[u8], - doc_id: &str, + identity: &RowIdentity, strict_schema: Option<&StrictSchema>, ) -> crate::Result { let mut doc = match strict_schema { @@ -57,7 +62,7 @@ pub(in crate::data::executor) fn from_stored( crate::Error::Serialization { format: "binary_tuple".to_string(), detail: format!( - "RETURNING row {doc_id}: stored body ({} bytes) is not a Binary Tuple \ + "RETURNING row {identity}: stored body ({} bytes) is not a Binary Tuple \ readable under the collection's strict schema", body.len() ), @@ -67,10 +72,11 @@ pub(in crate::data::executor) fn from_stored( // non-map body as `{id, value}`, which is the shape schemaless callers // have always emitted for a body that is not a document map. None => { - let with_id = nodedb_query::msgpack_scan::inject_str_field(body, "id", doc_id); + let with_id = + nodedb_query::msgpack_scan::inject_str_field(body, "id", identity.as_str()); doc_format::decode_document(&with_id)? } }; - attach_row_id(&mut doc, doc_id); + attach_row_id(&mut doc, identity); Ok(doc) } diff --git a/nodedb/src/data/executor/handlers/returning_rows.rs b/nodedb/src/data/executor/handlers/returning_rows.rs index 6fb6e8621..43505e0c2 100644 --- a/nodedb/src/data/executor/handlers/returning_rows.rs +++ b/nodedb/src/data/executor/handlers/returning_rows.rs @@ -14,12 +14,13 @@ use crate::data::executor::response_codec::RowsPayload; use crate::data::executor::scan_normalize::{kv_row_to_doc, sparse_row_to_doc}; use crate::data::executor::sparse_body_format::SparseBodyFormatRef; use crate::data::executor::task::ExecutionTask; +use crate::engine::document::store::RowIdentity; use nodedb_physical::physical_plan::{ReturningColumns, ReturningSpec}; use nodedb_types::columnar::StrictSchema; -/// Rows a write path hands back to a `RETURNING` projection: the user-facing -/// document id paired with the exact bytes stored for it. -pub(in crate::data::executor) type StoredRow<'a> = (&'a str, &'a [u8]); +/// Rows a write path hands back to a `RETURNING` projection: the row's +/// client-visible identity paired with the exact bytes stored for it. +pub(in crate::data::executor) type StoredRow<'a> = (&'a RowIdentity, &'a [u8]); impl CoreLoop { /// Build this task's `RETURNING` response from the rows it just stored. diff --git a/nodedb/src/data/executor/handlers/rls_write_gate.rs b/nodedb/src/data/executor/handlers/rls_write_gate.rs index 04500eac5..bba77abdd 100644 --- a/nodedb/src/data/executor/handlers/rls_write_gate.rs +++ b/nodedb/src/data/executor/handlers/rls_write_gate.rs @@ -79,10 +79,13 @@ pub(in crate::data::executor) fn admit_row( /// A body that does not decode at all is refused rather than written /// unchecked — an image the policy could not be evaluated against is not an /// image the policy admitted. +/// +/// `identity` is the row's client-visible identity. The caller decides the +/// encoding: a caller holding a document storage key converts it first. pub(in crate::data::executor) fn admit_stored_row( rls_write_check: &RlsWriteCheck, body: &[u8], - doc_id: &str, + identity: &crate::engine::document::store::RowIdentity, strict_schema: Option<&StrictSchema>, tid: u64, collection: &str, @@ -96,13 +99,13 @@ pub(in crate::data::executor) fn admit_stored_row( ), }), WriteGateDecision::Evaluate(_) => { - match returning_doc::from_stored(body, doc_id, strict_schema) { + match returning_doc::from_stored(body, identity, strict_schema) { Ok(image) => admit_row(rls_write_check, &image, tid, collection), Err(e) => Err(crate::Error::RejectedAuthz { tenant_id: crate::types::TenantId::new(tid), resource: format!( - "RLS write policy on '{collection}': row '{doc_id}' did not decode, so the \ - policy could not be evaluated against it: {e}" + "RLS write policy on '{collection}': row '{identity}' did not decode, so \ + the policy could not be evaluated against it: {e}" ), }), } diff --git a/nodedb/src/data/executor/handlers/transaction/batch.rs b/nodedb/src/data/executor/handlers/transaction/batch.rs index 93487da7e..57566910a 100644 --- a/nodedb/src/data/executor/handlers/transaction/batch.rs +++ b/nodedb/src/data/executor/handlers/transaction/batch.rs @@ -430,7 +430,11 @@ impl CoreLoop { } /// Emit deferred trigger events for every write recorded in the - /// committed transaction's undo log. + /// committed transaction's undo log. `UndoEntry::{PutDocument, + /// DeleteDocument}.document_id` is the row's storage key, so a deferred + /// trigger converts it to the client-visible identity here — the same + /// conversion an immediate trigger sees via `emit_put_event` / + /// `emit_document_delete_event`. fn emit_deferred_writes(&mut self, task: &ExecutionTask, undo_log: Vec) { use crate::data::executor::core_loop::deferred::DeferredWrite; let deferred_writes: Vec = undo_log @@ -448,7 +452,7 @@ impl CoreLoop { } else { crate::event::WriteOp::Insert }, - row_id: document_id, + identity: crate::engine::document::store::identity_of(&document_id), new_value: None, old_value, }), @@ -460,7 +464,7 @@ impl CoreLoop { } => Some(DeferredWrite { collection, op: crate::event::WriteOp::Delete, - row_id: document_id, + identity: crate::engine::document::store::identity_of(&document_id), new_value: None, old_value: Some(old_value), }), diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/merge.rs b/nodedb/src/data/executor/handlers/transaction/overlay/merge.rs index ff323fd19..32b454393 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/merge.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/merge.rs @@ -97,14 +97,18 @@ impl CoreLoop { // retain pass has just superseded in place). let mut seen: HashSet = rows .iter() - .filter_map(|(k, _)| u32::from_str_radix(k, 16).ok()) + .filter_map(|(k, _)| { + crate::engine::document::store::doc_id_to_surrogate(k).map(|s| s.as_u32()) + }) .collect(); // Base-minus-superseded: a single in-place pass. Drop tombstoned rows, // replace put-superseded bodies and re-check the predicate, keep the // rest untouched. rows.retain_mut(|(row_key, body)| { - let Ok(surrogate) = u32::from_str_radix(row_key, 16) else { + let Some(surrogate) = + crate::engine::document::store::doc_id_to_surrogate(row_key).map(|s| s.as_u32()) + else { return true; }; match overlay.get(coll_key, surrogate) { @@ -328,7 +332,9 @@ impl CoreLoop { // don't re-append a row the base index lookup already returned. let mut seen: HashSet = doc_ids .iter() - .filter_map(|id| u32::from_str_radix(id, 16).ok()) + .filter_map(|id| { + crate::engine::document::store::doc_id_to_surrogate(id).map(|s| s.as_u32()) + }) .collect(); // Base-minus-superseded: resolve each base hex doc_id to its surrogate @@ -337,7 +343,9 @@ impl CoreLoop { // have moved the row off the indexed value); no overlay entry — or an // unparseable key — keeps it as-is. doc_ids.retain(|doc_id| { - let Ok(surrogate) = u32::from_str_radix(doc_id, 16) else { + let Some(surrogate) = + crate::engine::document::store::doc_id_to_surrogate(doc_id).map(|s| s.as_u32()) + else { return true; }; match overlay.get(coll_key, surrogate) { @@ -399,7 +407,8 @@ impl CoreLoop { // Read-your-own-writes refreshes the lease (see the reaper). self.touch_overlay(txn_id); if let Some(overlay) = self.txn_overlays.get(&txn_id) - && let Ok(surrogate) = u32::from_str_radix(doc_id, 16) + && let Some(surrogate) = + crate::engine::document::store::doc_id_to_surrogate(doc_id).map(|s| s.as_u32()) { match overlay.get(coll_key, surrogate) { Some(Staged::Put(body)) => return Ok(Some(body.clone())), diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/dispatch.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/dispatch.rs index 79ddfb91d..f05737b98 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/dispatch.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/dispatch.rs @@ -384,7 +384,7 @@ impl CoreLoop { &self, rls_write_check: &nodedb_types::RlsWriteCheck, body: &[u8], - doc_id: &str, + identity: &crate::engine::document::store::RowIdentity, database_id: u64, tid: u64, collection: &str, @@ -399,7 +399,7 @@ impl CoreLoop { crate::data::executor::handlers::rls_write_gate::admit_stored_row( rls_write_check, body, - doc_id, + identity, schema.as_ref(), tid, collection, diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_delete.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_delete.rs index 549b11cd7..402576d56 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_delete.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_delete.rs @@ -115,10 +115,11 @@ impl CoreLoop { nodedb_types::WriteGateDecision::AdmitAll ) { for (row_key, body) in &rows { + let identity = crate::engine::document::store::identity_of(row_key); if let Err(e) = self.stage_admit_write( rls_write_check, body, - row_key, + &identity, database_id.as_u64(), tid, collection, diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs index 76cc69bbd..d0783719f 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs @@ -149,10 +149,11 @@ impl CoreLoop { // policy. A rejected row fails the statement rather than being // skipped: skipping would under-report `affected` while the rest of // the predicate's matches were still rewritten. + let identity = crate::engine::document::store::identity_of(row_key); if let Err(e) = self.stage_admit_write( rls_write_check, &new_body, - row_key, + &identity, database_id.as_u64(), tid, collection, diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs index 6ce03183c..d42faaee1 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs @@ -242,7 +242,7 @@ impl CoreLoop { self.stage_admit_write( rls_write_check, image, - &ctx.document_id, + &crate::engine::document::store::RowIdentity::from_user_key(ctx.document_id.as_ref()), ctx.database_id, ctx.tid, ctx.collection, diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_conflict.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_conflict.rs index 9883e8ed1..d0a9e8ef7 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_conflict.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_conflict.rs @@ -94,7 +94,7 @@ impl CoreLoop { if let Err(e) = self.stage_admit_write( rls_write_check, &stored_bytes, - &ctx.document_id, + &crate::engine::document::store::RowIdentity::from_user_key(ctx.document_id.as_ref()), ctx.database_id, ctx.tid, ctx.collection, diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_delete.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_delete.rs index 3b48ae9c2..ce34fded4 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_delete.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_delete.rs @@ -116,7 +116,9 @@ impl CoreLoop { && let Err(e) = self.stage_admit_write( rls_write_check, &body, - &doc_id, + &crate::engine::document::store::RowIdentity::from_user_key( + doc_id.as_str(), + ), did.as_u64(), tid, collection, diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_point_document.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_point_document.rs index 9ff8358f9..4ddfdf41f 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_point_document.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_point_document.rs @@ -25,7 +25,7 @@ use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::doc_format; use crate::data::executor::handlers::generated; use crate::data::executor::handlers::transaction::overlay::Staged; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::StorageKey; use crate::types::TenantId; impl CoreLoop { @@ -35,7 +35,7 @@ impl CoreLoop { value: &[u8], if_absent: bool, ) -> Response { - let row_key = surrogate_to_doc_id(ctx.surrogate); + let row_key = StorageKey::for_surrogate(ctx.surrogate).to_string(); let bitemporal = self.is_bitemporal(ctx.database_id, ctx.tid, ctx.collection); let overlay_pk = self.stage_overlay_pk(ctx); @@ -97,7 +97,8 @@ impl CoreLoop { // still exists (a surrogate outlives its row so a re-insert keeps it), // and an earlier statement in this transaction may already have // tombstoned it. - let row_key = surrogate_to_doc_id(ctx.surrogate); + let storage_key = StorageKey::for_surrogate(ctx.surrogate); + let row_key = storage_key.to_string(); let bitemporal = self.is_bitemporal(ctx.database_id, ctx.tid, ctx.collection); let overlay_pk = self.stage_overlay_pk(ctx); let present = match self.stage_pk_present( @@ -130,7 +131,7 @@ impl CoreLoop { && let Err(e) = self.stage_admit_write( rls_write_check, &body, - row_key.as_str(), + &storage_key.to_identity(), ctx.database_id, ctx.tid, ctx.collection, @@ -160,7 +161,8 @@ impl CoreLoop { TenantId::new(ctx.tid), ctx.collection.to_string(), ); - let row_key = surrogate_to_doc_id(ctx.surrogate); + let storage_key = StorageKey::for_surrogate(ctx.surrogate); + let row_key = storage_key.to_string(); // Reject direct updates to generated columns (matches the durable path). if let Some(config) = self.doc_configs.get(&config_key) @@ -218,7 +220,7 @@ impl CoreLoop { if let Err(e) = self.stage_admit_write( rls_write_check, &body, - row_key.as_str(), + &storage_key.to_identity(), ctx.database_id, ctx.tid, ctx.collection, diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs index 7e7e4936f..022a5520e 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs @@ -67,7 +67,7 @@ impl CoreLoop { if let Err(e) = self.stage_admit_write( rls_write_check, &stored_bytes, - &ctx.document_id, + &crate::engine::document::store::RowIdentity::from_user_key(ctx.document_id.as_ref()), ctx.database_id, ctx.tid, ctx.collection, diff --git a/nodedb/src/data/executor/handlers/truncate.rs b/nodedb/src/data/executor/handlers/truncate.rs index 96b2bbd0b..53b98c37e 100644 --- a/nodedb/src/data/executor/handlers/truncate.rs +++ b/nodedb/src/data/executor/handlers/truncate.rs @@ -227,12 +227,11 @@ impl CoreLoop { collection, deleted_bytes, ); - self.emit_write_event( + let identity = crate::engine::document::store::identity_of(doc_id); + self.emit_document_delete_event( task, collection, - crate::event::WriteOp::Delete, - doc_id, - None, + identity, Some(old_converted.as_deref().unwrap_or(deleted_bytes)), ); truncated += 1; diff --git a/nodedb/src/data/executor/handlers/update_from_join_collect.rs b/nodedb/src/data/executor/handlers/update_from_join_collect.rs index 2e36ad9ef..2a33db855 100644 --- a/nodedb/src/data/executor/handlers/update_from_join_collect.rs +++ b/nodedb/src/data/executor/handlers/update_from_join_collect.rs @@ -116,9 +116,10 @@ impl CoreLoop { &doc_id, "update_from_join_collect", ); + let identity = crate::engine::document::store::identity_of(&doc_id); super::super::strict_format::undecodable_strict_row( target_collection, - &doc_id, + identity.as_str(), ) })? } else { diff --git a/nodedb/src/data/executor/handlers/update_from_join_write.rs b/nodedb/src/data/executor/handlers/update_from_join_write.rs index 1dde67061..b10c2bf68 100644 --- a/nodedb/src/data/executor/handlers/update_from_join_write.rs +++ b/nodedb/src/data/executor/handlers/update_from_join_write.rs @@ -141,11 +141,15 @@ impl CoreLoop { // `collect_update_from_join_rows`; `emit_put_event` derives // `WriteOp::Update` from the Some prior + Some new pair and // handles strict->msgpack conversion on both sides. + let row_identity = crate::engine::document::store::identity_of(&doc_id); + // `row_identity` is read again below for `RETURNING`'s `id` + // field, so the event-emit boundary gets a clone rather than + // the move. self.emit_put_event( task, tid, target_collection, - &doc_id, + row_identity.clone(), &updated_bytes, Some(&old_body), ); @@ -177,11 +181,10 @@ impl CoreLoop { } affected += 1; if want_returning { - // `doc_id` is the surrogate hex storage key, which only - // stands in as `id` for a row that declares no primary key - // of its own — overwriting a declared key would return a - // value the client never wrote. - returning_doc::attach_row_id(&mut doc, &doc_id); + // `row_identity` only stands in as `id` for a row that + // declares no primary key of its own — overwriting a + // declared key would return a value the client never wrote. + returning_doc::attach_row_id(&mut doc, &row_identity); returned_docs.push(doc); } } diff --git a/nodedb/src/data/executor/handlers/upsert/exec/insert.rs b/nodedb/src/data/executor/handlers/upsert/exec/insert.rs index f981371d8..cf185d1fa 100644 --- a/nodedb/src/data/executor/handlers/upsert/exec/insert.rs +++ b/nodedb/src/data/executor/handlers/upsert/exec/insert.rs @@ -10,6 +10,7 @@ use crate::data::executor::enforcement::write_hook::{self, HookCtx, ImageBody, W use crate::data::executor::handlers::point::apply_put::PointPutParams; use crate::data::executor::handlers::rls_write_gate; use crate::data::executor::task::ExecutionTask; +use crate::engine::document::store::{RowIdentity, StorageKey}; use nodedb_types::Surrogate; use nodedb_types::columnar::StrictSchema; @@ -56,14 +57,22 @@ impl CoreLoop { strict_schema, } = ctx; + let row_identity = StorageKey::for_surrogate(surrogate).to_identity(); + let document_identity = RowIdentity::from_user_key(document_id); + // Insert: document doesn't exist, create new (same as PointPut). // The incoming body IS the post-image here, and the planner // emits it as MessagePack for both storage modes (the strict // tuple is encoded on the way to disk), so it is decoded // without a schema. - if let Err(e) = - rls_write_gate::admit_stored_row(rls_write_check, value, row_key, None, tid, collection) - { + if let Err(e) = rls_write_gate::admit_stored_row( + rls_write_check, + value, + &row_identity, + None, + tid, + collection, + ) { return self.response_error(task, e); } @@ -160,7 +169,7 @@ impl CoreLoop { task, tid, collection, - row_key, + row_identity, value, prior.prior_value.as_deref(), ); @@ -177,7 +186,7 @@ impl CoreLoop { spec, rls_filters, strict_schema, - &[(document_id, prior.stored_value.as_slice())], + &[(&document_identity, prior.stored_value.as_slice())], ), None => self.response_affected(task, 1), }; diff --git a/nodedb/src/data/executor/handlers/upsert/exec/overwrite.rs b/nodedb/src/data/executor/handlers/upsert/exec/overwrite.rs index bfac7403c..f0c94236b 100644 --- a/nodedb/src/data/executor/handlers/upsert/exec/overwrite.rs +++ b/nodedb/src/data/executor/handlers/upsert/exec/overwrite.rs @@ -11,6 +11,7 @@ use crate::data::executor::handlers::point::apply_put::PointPutParams; use crate::data::executor::handlers::rls_write_gate; use crate::data::executor::handlers::upsert::merge::{apply_on_conflict_updates, merge_values}; use crate::data::executor::task::ExecutionTask; +use crate::engine::document::store::{RowIdentity, StorageKey}; use nodedb_types::Surrogate; use nodedb_types::columnar::StrictSchema; @@ -63,6 +64,9 @@ impl CoreLoop { current_bytes, } = ctx; + let row_identity = StorageKey::for_surrogate(surrogate).to_identity(); + let document_identity = RowIdentity::from_user_key(document_id); + // Decode existing document to nodedb_types::Value. let existing_val = if let Some(schema) = strict_schema { // Strict: binary tuple → Value via schema. @@ -153,7 +157,7 @@ impl CoreLoop { if let Err(e) = rls_write_gate::admit_stored_row( rls_write_check, &merged_body, - row_key, + &row_identity, None, tid, collection, @@ -265,7 +269,7 @@ impl CoreLoop { task, tid, collection, - row_key, + row_identity, &stored_bytes, Some(¤t_bytes), ); @@ -285,7 +289,7 @@ impl CoreLoop { spec, rls_filters, strict_schema, - &[(document_id, stored_bytes.as_slice())], + &[(&document_identity, stored_bytes.as_slice())], ), None => self.response_affected(task, 1), }; diff --git a/nodedb/src/data/executor/handlers/write_batch.rs b/nodedb/src/data/executor/handlers/write_batch.rs index 419042b02..25ab08868 100644 --- a/nodedb/src/data/executor/handlers/write_batch.rs +++ b/nodedb/src/data/executor/handlers/write_batch.rs @@ -165,7 +165,7 @@ impl CoreLoop { // bytes captured per row above. if let PhysicalPlan::Document(DocumentOp::PointPut { collection, - document_id, + surrogate, value, .. }) = task.plan() @@ -175,14 +175,10 @@ impl CoreLoop { Ok(p) => p.prior_value.as_deref(), Err(_) => None, }; - self.emit_put_event( - task, - tid, - collection.as_str(), - document_id, - value, - prior, - ); + let identity = + crate::engine::document::store::StorageKey::for_surrogate(*surrogate) + .to_identity(); + self.emit_put_event(task, tid, collection.as_str(), identity, value, prior); } self.response_ok(task) } diff --git a/nodedb/src/data/executor/row_shape.rs b/nodedb/src/data/executor/row_shape.rs index c02432f5b..48f9d1e8b 100644 --- a/nodedb/src/data/executor/row_shape.rs +++ b/nodedb/src/data/executor/row_shape.rs @@ -66,15 +66,25 @@ pub(in crate::data::executor) fn sparse_body_to_msgpack<'a>( /// field. Injection is a no-op when the body already carries an `id` — a /// vector-primary sidecar stores the user's declared primary key, and its /// sparse key is the internal surrogate-hex, which must not displace it. -/// Shared by the materializing scan and the streaming scan so both paths -/// produce byte-identical output. +/// +/// `id` is the row's storage key. When it is a minted key, the client-visible +/// identity is the surrogate's decimal string, not the hex storage key; a +/// non-minted key (a user-supplied or DDL-declared primary key) passes +/// through unchanged. Shared by the materializing scan and the streaming scan +/// so both paths produce byte-identical output. +/// +/// `id` comes off a store iterator as a plain `&str`, so this is one of the +/// few sites that reinterprets that shape: a value that parses as 8 lowercase +/// hex characters is a minted storage key, everything else is a legacy or +/// user key taken verbatim. pub(in crate::data::executor) fn sparse_row_to_doc( id: &str, raw: &[u8], format: SparseBodyFormatRef<'_>, ) -> (String, Vec) { + let identity = crate::engine::document::store::identity_of(id); let mp = sparse_body_to_msgpack(raw, format); - let mp = msgpack_scan::inject_str_field(&mp, "id", id); + let mp = msgpack_scan::inject_str_field(&mp, "id", identity.as_str()); (id.to_string(), mp) } diff --git a/nodedb/src/data/executor/strict_format/decode.rs b/nodedb/src/data/executor/strict_format/decode.rs index 600eb00fc..3dfa174e5 100644 --- a/nodedb/src/data/executor/strict_format/decode.rs +++ b/nodedb/src/data/executor/strict_format/decode.rs @@ -72,11 +72,12 @@ pub fn binary_tuple_to_msgpack(tuple_bytes: &[u8], schema: &StrictSchema) -> Opt } /// The error for a stored Binary Tuple that does not decode against the -/// collection's strict schema. Names the row so an operator can find it. -pub fn undecodable_strict_row(collection: &str, doc_id: &str) -> crate::Error { +/// collection's strict schema. Names the row by its client-visible identity, +/// never its storage key, so the error text never leaks the internal hex key. +pub fn undecodable_strict_row(collection: &str, identity: &str) -> crate::Error { crate::Error::Serialization { format: "binary_tuple".into(), - detail: format!("document \"{doc_id}\" of collection \"{collection}\" does not decode"), + detail: format!("document \"{identity}\" of collection \"{collection}\" does not decode"), } } diff --git a/nodedb/src/engine/document/store/key.rs b/nodedb/src/engine/document/store/key.rs index a9b488386..e004d6aa1 100644 --- a/nodedb/src/engine/document/store/key.rs +++ b/nodedb/src/engine/document/store/key.rs @@ -1,33 +1,157 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Document storage keys are derived from the row's stable surrogate identity. +//! This module owns both encodings of a row's surrogate identity. //! -//! The substrate redb tables (`DOCUMENTS`, `INDEXES`) use string keys; the -//! Document engine encodes each surrogate as a fixed-width 8-character -//! lowercase hexadecimal string (e.g. `Surrogate(42)` → `"0000002a"`). +//! [`StorageKey`] is internal: the substrate redb tables (`DOCUMENTS`, +//! `INDEXES`) use string keys, and the Document engine encodes each +//! surrogate as a fixed-width 8-character lowercase hexadecimal string +//! (e.g. `Surrogate(42)` -> `"0000002a"`). It must never reach a client. //! //! The format is intentionally fixed width: lexicographic ordering of the //! hex string matches numeric ordering of the underlying surrogate, so //! redb range scans can iterate rows in surrogate order without any -//! additional index. The user-visible primary key is bound to the -//! surrogate via the `_system.surrogate_pk{,_rev}` catalog tables. +//! additional index. +//! +//! [`RowIdentity`] is what a client sees: a predicate matches it, and the +//! catalog binds it. For a minted row it is the surrogate's decimal string. +//! A row with a user-supplied or DDL-declared primary key carries a +//! different identity, the key's body value, bound via the +//! `_system.surrogate_pk{,_rev}` catalog tables. +//! +//! The two types exist so the compiler rejects passing one where the other +//! belongs. A storage key and a user-declared identity can share the same +//! 8-hex-character shape (a primary key `deadbeef` parses as a storage key), +//! so a loose `String` cannot tell them apart. `StorageKey::parse` is the +//! only place that shape gets reinterpreted as a surrogate. use nodedb_types::Surrogate; +/// The redb key a document row is stored under. +/// +/// Fixed-width lowercase hex, so lexicographic order matches surrogate order +/// and a range scan iterates rows in surrogate order with no extra index. +/// Internal: a storage key never reaches a client. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct StorageKey(Surrogate); + +impl StorageKey { + /// Wrap `surrogate` as the key it is stored under. Allocation-free: the + /// hex text is a rendering, produced only by `Display` or `to_identity`. + pub fn for_surrogate(surrogate: Surrogate) -> Self { + Self(surrogate) + } + + /// Parse a redb key back into a `StorageKey`. + /// + /// Returns `None` unless `key` is exactly 8 lowercase hex characters. + /// This handles legacy non-surrogate document IDs gracefully: a value + /// that fails to parse is not a minted storage key. Allocation-free. + pub fn parse(key: &str) -> Option { + if key.len() != 8 + || !key + .bytes() + .all(|b| b.is_ascii_hexdigit() && !b.is_ascii_uppercase()) + { + return None; + } + let raw = u32::from_str_radix(key, 16).ok()?; + Some(Self(Surrogate::new(raw))) + } + + /// Recover the surrogate this key encodes. Infallible and free: the + /// surrogate IS the key, not something re-derived from stored text. + pub fn surrogate(&self) -> Surrogate { + self.0 + } + + /// The client-visible identity of the row stored under this key. + /// + /// Allocates the decimal string — unavoidable, since a client never sees + /// the hex encoding. + pub fn to_identity(&self) -> RowIdentity { + RowIdentity::for_surrogate(self.0) + } +} + +impl std::fmt::Display for StorageKey { + /// The only `{:08x}` surrogate-to-hex formatting in the workspace. + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "{:08x}", self.0.as_u32()) + } +} + +/// The identity a client sees for a row. +/// +/// A minted row renders its surrogate in decimal. A row with a declared +/// `PRIMARY KEY` carries the user's own value. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct RowIdentity(String); + +impl RowIdentity { + /// The decimal identity of a minted row. + pub fn for_surrogate(surrogate: Surrogate) -> Self { + Self(surrogate.as_u32().to_string()) + } + + /// Wrap a declared or client-supplied key as the row's identity. + /// + /// The value is taken verbatim and never interpreted as a storage key + /// or a surrogate. This is what makes it safe to call with a KV key, a + /// user-declared primary key, or any other engine-native identifier. + /// + /// Takes `impl Into` so an owned `String` moves in without a + /// copy. A `&str` caller still pays one copy, which is unavoidable. + pub fn from_user_key(key: impl Into) -> Self { + Self(key.into()) + } + + pub fn as_str(&self) -> &str { + &self.0 + } + + /// Consume this identity and return its inner `String` without copying. + pub fn into_string(self) -> String { + self.0 + } +} + +impl std::fmt::Display for RowIdentity { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(&self.0) + } +} + /// Format a surrogate as the 8-character zero-padded lowercase hex string /// used as the document's redb key. +/// +/// Thin wrapper over [`StorageKey::for_surrogate`], kept because 242 +/// call sites across the workspace hold the result as a plain `String` +/// (redb key params, msgpack field injection, WAL replay) rather than a +/// `StorageKey`. Converting all of them is a separate ripple from this one. +/// One allocation: the `Display` format. pub fn surrogate_to_doc_id(surrogate: Surrogate) -> String { - format!("{:08x}", surrogate.as_u32()) + StorageKey::for_surrogate(surrogate).to_string() } /// Parse a hex-encoded document storage key back to a `Surrogate`. /// /// Returns `None` if the key is not exactly 8 lowercase hex characters — /// this handles legacy non-surrogate document IDs gracefully. +/// +/// Thin wrapper over [`StorageKey::parse`], kept for the same reason as +/// [`surrogate_to_doc_id`]: 35 call sites hold a plain `&str` doc ID. +/// Allocation-free. pub fn doc_id_to_surrogate(doc_id: &str) -> Option { - if doc_id.len() != 8 { - return None; - } - u32::from_str_radix(doc_id, 16).ok().map(Surrogate::new) + StorageKey::parse(doc_id).map(|key| key.surrogate()) +} + +/// The client-visible identity of a row stored under `doc_id`. +/// +/// A minted key renders its surrogate in decimal. Any other key is a user's +/// own value and passes through verbatim. +pub fn identity_of(doc_id: &str) -> RowIdentity { + StorageKey::parse(doc_id) + .map(|key| key.to_identity()) + .unwrap_or_else(|| RowIdentity::from_user_key(doc_id)) } #[cfg(test)] @@ -36,15 +160,56 @@ mod tests { #[test] fn formats_zero_padded_lowercase() { - assert_eq!(surrogate_to_doc_id(Surrogate::new(0)), "00000000"); - assert_eq!(surrogate_to_doc_id(Surrogate::new(42)), "0000002a"); - assert_eq!(surrogate_to_doc_id(Surrogate::new(0xDEAD_BEEF)), "deadbeef"); + assert_eq!( + StorageKey::for_surrogate(Surrogate::new(0)).to_string(), + "00000000" + ); + assert_eq!( + StorageKey::for_surrogate(Surrogate::new(42)).to_string(), + "0000002a" + ); + assert_eq!( + StorageKey::for_surrogate(Surrogate::new(0xDEAD_BEEF)).to_string(), + "deadbeef" + ); } #[test] fn lex_order_matches_numeric() { - let a = surrogate_to_doc_id(Surrogate::new(0x10)); - let b = surrogate_to_doc_id(Surrogate::new(0x100)); + let a = StorageKey::for_surrogate(Surrogate::new(0x10)).to_string(); + let b = StorageKey::for_surrogate(Surrogate::new(0x100)).to_string(); assert!(a < b); } + + #[test] + fn parse_rejects_wrong_shape() { + assert!(StorageKey::parse("").is_none()); + assert!(StorageKey::parse("abc").is_none()); + assert!(StorageKey::parse("DEADBEEF").is_none()); + assert!(StorageKey::parse("zzzzzzzz").is_none()); + assert!(StorageKey::parse("123456789").is_none()); + } + + #[test] + fn parse_roundtrips_surrogate() { + let key = StorageKey::for_surrogate(Surrogate::new(0x2a)); + let parsed = StorageKey::parse(&key.to_string()).expect("valid storage key"); + assert_eq!(parsed.surrogate(), Surrogate::new(0x2a)); + } + + #[test] + fn to_identity_is_decimal() { + let key = StorageKey::for_surrogate(Surrogate::new(42)); + assert_eq!(key.to_identity().as_str(), "42"); + } + + #[test] + fn identity_of_minted_key_is_decimal() { + assert_eq!(identity_of("0000002a").as_str(), "42"); + } + + #[test] + fn identity_of_user_key_passes_through() { + assert_eq!(identity_of("user-declared-id").as_str(), "user-declared-id"); + } } diff --git a/nodedb/src/engine/document/store/mod.rs b/nodedb/src/engine/document/store/mod.rs index c0d6a3a8d..29dc00e98 100644 --- a/nodedb/src/engine/document/store/mod.rs +++ b/nodedb/src/engine/document/store/mod.rs @@ -10,4 +10,4 @@ pub use config::CollectionConfig; pub use engine::DocumentEngine; pub use extract::{extract_index_values, json_to_msgpack}; pub use index_path::IndexPath; -pub use key::{doc_id_to_surrogate, surrogate_to_doc_id}; +pub use key::{RowIdentity, StorageKey, doc_id_to_surrogate, identity_of, surrogate_to_doc_id}; diff --git a/nodedb/tests/wire/cases/sql_typeguard_defaults.rs b/nodedb/tests/wire/cases/sql_typeguard_defaults.rs deleted file mode 100644 index 726716db5..000000000 --- a/nodedb/tests/wire/cases/sql_typeguard_defaults.rs +++ /dev/null @@ -1,458 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Integration tests for typeguard DEFAULT/VALUE expressions and VALIDATE TYPEGUARD. -//! -//! Verifies that: -//! - DEFAULT injects a value when the field is absent -//! - DEFAULT does not overwrite user-provided values -//! - VALUE always overwrites, even when user provides a value -//! - REQUIRED + DEFAULT = field is always present -//! - Cross-field VALUE expressions resolve other document fields - -use crate::harness::TestServer; - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn typeguard_default_injects_when_absent() { - let server = TestServer::start().await; - - server.exec("CREATE COLLECTION tg_defaults").await.unwrap(); - - server - .exec( - "CREATE TYPEGUARD ON tg_defaults (\ - status STRING DEFAULT 'draft'\ - )", - ) - .await - .unwrap(); - - // Insert without status — DEFAULT should fill it. - server - .exec("INSERT INTO tg_defaults { id: 'd1', name: 'Alice' }") - .await - .unwrap(); - - let rows = server - .query_text_joined("SELECT * FROM tg_defaults WHERE id = 'd1'") - .await - .unwrap(); - assert_eq!(rows.len(), 1); - assert!( - rows[0].contains("draft"), - "DEFAULT should inject 'draft': {rows:?}" - ); -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn typeguard_default_does_not_overwrite() { - let server = TestServer::start().await; - - server - .exec("CREATE COLLECTION tg_no_overwrite") - .await - .unwrap(); - - server - .exec( - "CREATE TYPEGUARD ON tg_no_overwrite (\ - status STRING DEFAULT 'draft'\ - )", - ) - .await - .unwrap(); - - // Insert with explicit status — DEFAULT should NOT overwrite. - server - .exec("INSERT INTO tg_no_overwrite { id: 'd1', status: 'active' }") - .await - .unwrap(); - - let rows = server - .query_text_joined("SELECT * FROM tg_no_overwrite WHERE id = 'd1'") - .await - .unwrap(); - assert_eq!(rows.len(), 1); - assert!( - rows[0].contains("active"), - "DEFAULT should not overwrite user value: {rows:?}" - ); - assert!( - !rows[0].contains("draft"), - "should NOT contain default: {rows:?}" - ); -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn typeguard_value_always_overwrites() { - let server = TestServer::start().await; - - server - .exec("CREATE COLLECTION tg_value_overwrite") - .await - .unwrap(); - - server - .exec( - "CREATE TYPEGUARD ON tg_value_overwrite (\ - computed STRING VALUE 'server_computed'\ - )", - ) - .await - .unwrap(); - - // Insert with user-provided value — VALUE should overwrite. - server - .exec("INSERT INTO tg_value_overwrite { id: 'v1', computed: 'user_input' }") - .await - .unwrap(); - - let rows = server - .query_text_joined("SELECT * FROM tg_value_overwrite WHERE id = 'v1'") - .await - .unwrap(); - assert_eq!(rows.len(), 1); - assert!( - rows[0].contains("server_computed"), - "VALUE should overwrite user input: {rows:?}" - ); -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn typeguard_required_plus_default() { - let server = TestServer::start().await; - - server - .exec("CREATE COLLECTION tg_req_default") - .await - .unwrap(); - - server - .exec( - "CREATE TYPEGUARD ON tg_req_default (\ - version INT REQUIRED DEFAULT 1\ - )", - ) - .await - .unwrap(); - - // Insert without version — DEFAULT fills before REQUIRED check. - server - .exec("INSERT INTO tg_req_default { id: 'r1', name: 'test' }") - .await - .unwrap(); - - let rows = server - .query_text_joined("SELECT * FROM tg_req_default WHERE id = 'r1'") - .await - .unwrap(); - assert_eq!(rows.len(), 1); - assert!( - rows[0].contains('1'), - "REQUIRED + DEFAULT should inject version=1: {rows:?}" - ); -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn typeguard_default_integer() { - let server = TestServer::start().await; - - server - .exec("CREATE COLLECTION tg_int_default") - .await - .unwrap(); - - server - .exec( - "CREATE TYPEGUARD ON tg_int_default (\ - priority INT DEFAULT 0 CHECK (priority >= 0)\ - )", - ) - .await - .unwrap(); - - // Insert without priority — DEFAULT 0 should be injected and pass CHECK. - server - .exec("INSERT INTO tg_int_default { id: 'p1', name: 'test' }") - .await - .unwrap(); - - let rows = server - .query_text_joined("SELECT * FROM tg_int_default WHERE id = 'p1'") - .await - .unwrap(); - assert_eq!(rows.len(), 1); -} - -// ── VALIDATE TYPEGUARD ── - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn validate_typeguard_no_violations() { - let server = TestServer::start().await; - - server.exec("CREATE COLLECTION val_clean").await.unwrap(); - - // Insert valid data first. - server - .exec("INSERT INTO val_clean { id: 'v1', name: 'Alice', age: 25 }") - .await - .unwrap(); - - // Add type guard after data. - server - .exec( - "CREATE TYPEGUARD ON val_clean (\ - name STRING,\ - age INT\ - )", - ) - .await - .unwrap(); - - // Validate — all docs should pass. - let rows = server - .query_text("VALIDATE TYPEGUARD ON val_clean") - .await - .unwrap(); - assert_eq!(rows.len(), 0, "no violations expected: {rows:?}"); -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn validate_typeguard_finds_violations() { - let server = TestServer::start().await; - - server.exec("CREATE COLLECTION val_dirty").await.unwrap(); - - // Insert data that will violate a future type guard. - server - .exec("INSERT INTO val_dirty { id: 'd1', name: 'Alice', score: 42 }") - .await - .unwrap(); - server - .exec("INSERT INTO val_dirty { id: 'd2', name: 123, score: 99 }") - .await - .unwrap(); - - // Add type guard — name must be STRING. - server - .exec( - "CREATE TYPEGUARD ON val_dirty (\ - name STRING\ - )", - ) - .await - .unwrap(); - - // Validate — d2 has name=123 (INT, not STRING). - let rows = server - .query_text("VALIDATE TYPEGUARD ON val_dirty") - .await - .unwrap(); - assert!( - !rows.is_empty(), - "should find at least one violation: {rows:?}" - ); - // First column is document_id — should be d2. - assert!( - rows.iter().any(|r| r.contains("d2")), - "violation should reference d2: {rows:?}" - ); -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn validate_typeguard_no_guards() { - let server = TestServer::start().await; - - server.exec("CREATE COLLECTION val_noguard").await.unwrap(); - - server - .exec("INSERT INTO val_noguard { id: 'n1', x: 1 }") - .await - .unwrap(); - - // No typeguard — should return empty result. - let rows = server - .query_text("VALIDATE TYPEGUARD ON val_noguard") - .await - .unwrap(); - assert_eq!(rows.len(), 0); -} - -// ── CONVERT TO strict ── - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn convert_to_strict_from_typeguards() { - let server = TestServer::start().await; - - server.exec("CREATE COLLECTION conv_tg").await.unwrap(); - - // Add typeguards with types and CHECK. - server - .exec( - "CREATE TYPEGUARD ON conv_tg (\ - name STRING REQUIRED,\ - age INT CHECK (age >= 0)\ - )", - ) - .await - .unwrap(); - - // Insert valid data. - server - .exec("INSERT INTO conv_tg { id: 'c1', name: 'Alice', age: 25 }") - .await - .unwrap(); - - // Convert to strict WITHOUT explicit column defs — should infer from typeguards. - server - .exec("CONVERT COLLECTION conv_tg TO document_strict") - .await - .unwrap(); - - // Typeguards should be gone. - let tg_rows = server - .query_text("SHOW TYPEGUARD ON conv_tg") - .await - .unwrap(); - assert_eq!( - tg_rows.len(), - 0, - "typeguards should be cleared: {tg_rows:?}" - ); - - // CHECK constraints should be carried over. - let constraint_rows = server - .query_text("SHOW CONSTRAINTS ON conv_tg") - .await - .unwrap(); - assert!( - constraint_rows.iter().any(|r| r.contains("_guard_age")), - "CHECK from typeguard should carry over: {constraint_rows:?}" - ); -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn convert_to_strict_no_typeguards_no_cols_errors() { - let server = TestServer::start().await; - - server.exec("CREATE COLLECTION conv_empty").await.unwrap(); - - // No typeguards, no column defs — should fail. - let err = server - .exec("CONVERT COLLECTION conv_empty TO document_strict") - .await; - assert!( - err.is_err(), - "should fail without typeguards or column defs: {err:?}" - ); -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn convert_to_strict_with_explicit_cols() { - let server = TestServer::start().await; - - server - .exec("CREATE COLLECTION conv_explicit") - .await - .unwrap(); - - server - .exec("INSERT INTO conv_explicit { id: 'e1', val: 42 }") - .await - .unwrap(); - - // Convert with explicit column defs (should still work as before). - let result = server - .exec("CONVERT COLLECTION conv_explicit TO document_strict (id TEXT, val INT)") - .await; - assert!(result.is_ok(), "explicit convert should work"); -} - -// ── DEFAULT gen_uuid_v7() / now() on strict schema ── - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn strict_default_gen_uuid_v7() { - let server = TestServer::start().await; - - server - .exec( - "CREATE COLLECTION strict_uuid (\ - id TEXT PRIMARY KEY DEFAULT gen_uuid_v7(),\ - name TEXT\ - ) WITH (engine='document_strict')", - ) - .await - .unwrap(); - - // Insert without id — DEFAULT gen_uuid_v7() should fill it. - let result = server - .exec("INSERT INTO strict_uuid (name) VALUES ('Alice')") - .await; - result.unwrap(); - - let rows = server - .query_text_joined("SELECT * FROM strict_uuid") - .await - .unwrap(); - assert_eq!(rows.len(), 1, "should have 1 row: {rows:?}"); -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn strict_default_now() { - let server = TestServer::start().await; - - server - .exec( - "CREATE COLLECTION strict_ts (\ - id TEXT PRIMARY KEY,\ - created_at TEXT DEFAULT now()\ - ) WITH (engine='document_strict')", - ) - .await - .unwrap(); - - // Insert without created_at — DEFAULT now() should fill it. - server - .exec("INSERT INTO strict_ts (id) VALUES ('t1')") - .await - .unwrap(); - - let rows = server - .query_text_joined("SELECT * FROM strict_ts WHERE id = 't1'") - .await - .unwrap(); - assert_eq!(rows.len(), 1, "should have 1 row: {rows:?}"); -} - -// ── Unresolvable declared type ── - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn typeguard_unresolvable_type_is_refused_at_declaration() { - let server = TestServer::start().await; - - server.exec("CREATE COLLECTION tg_bad_type").await.unwrap(); - - // A type name the engine resolves to nothing. - server - .expect_error("CREATE TYPEGUARD ON tg_bad_type (gadget WIDGET)", "42601") - .await; - - // A trailing word that reading the leading token alone would ignore. - server - .expect_error( - "CREATE TYPEGUARD ON tg_bad_type (at TIMESTAMP GARBAGE)", - "42601", - ) - .await; - - // ALTER carries the same refusal. - server - .expect_error("ALTER TYPEGUARD ON tg_bad_type ADD gadget WIDGET", "42601") - .await; - - // A refused declaration reaches no storage. - let rows = server - .query_text("SHOW TYPEGUARD ON tg_bad_type") - .await - .unwrap(); - assert_eq!(rows.len(), 0, "refused guard must not be stored: {rows:?}"); -} diff --git a/nodedb/tests/wire/cases/sql_typeguard_defaults/convert.rs b/nodedb/tests/wire/cases/sql_typeguard_defaults/convert.rs new file mode 100644 index 000000000..92800fefa --- /dev/null +++ b/nodedb/tests/wire/cases/sql_typeguard_defaults/convert.rs @@ -0,0 +1,229 @@ +// SPDX-License-Identifier: BUSL-1.1 + +use crate::harness::TestServer; + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn convert_to_strict_from_typeguards() { + let server = TestServer::start().await; + + server.exec("CREATE COLLECTION conv_tg").await.unwrap(); + + // Add typeguards with types and CHECK. + server + .exec( + "CREATE TYPEGUARD ON conv_tg (\ + name STRING REQUIRED,\ + age INT CHECK (age >= 0)\ + )", + ) + .await + .unwrap(); + + // Insert valid data. + server + .exec("INSERT INTO conv_tg { id: 'c1', name: 'Alice', age: 25 }") + .await + .unwrap(); + + // Convert to strict WITHOUT explicit column defs — should infer from typeguards. + server + .exec("CONVERT COLLECTION conv_tg TO document_strict") + .await + .unwrap(); + + // Typeguards should be gone. + let tg_rows = server + .query_text("SHOW TYPEGUARD ON conv_tg") + .await + .unwrap(); + assert_eq!( + tg_rows.len(), + 0, + "typeguards should be cleared: {tg_rows:?}" + ); + + // CHECK constraints should be carried over. + let constraint_rows = server + .query_text("SHOW CONSTRAINTS ON conv_tg") + .await + .unwrap(); + assert!( + constraint_rows.iter().any(|r| r.contains("_guard_age")), + "CHECK from typeguard should carry over: {constraint_rows:?}" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn convert_to_strict_no_typeguards_no_cols_errors() { + let server = TestServer::start().await; + + server.exec("CREATE COLLECTION conv_empty").await.unwrap(); + + // No typeguards, no column defs — should fail. + let err = server + .exec("CONVERT COLLECTION conv_empty TO document_strict") + .await; + assert!( + err.is_err(), + "should fail without typeguards or column defs: {err:?}" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn convert_to_strict_with_explicit_cols() { + let server = TestServer::start().await; + + server + .exec("CREATE COLLECTION conv_explicit") + .await + .unwrap(); + + server + .exec("INSERT INTO conv_explicit { id: 'e1', val: 42 }") + .await + .unwrap(); + + // Convert with explicit column defs (should still work as before). + let result = server + .exec("CONVERT COLLECTION conv_explicit TO document_strict (id TEXT, val INT)") + .await; + assert!(result.is_ok(), "explicit convert should work"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn convert_to_strict_preserves_a_minted_rows_identity() { + let server = TestServer::start().await; + + server.exec("CREATE COLLECTION conv_minted").await.unwrap(); + + server + .exec( + "CREATE TYPEGUARD ON conv_minted (\ + name STRING REQUIRED\ + )", + ) + .await + .unwrap(); + + // No declared id — the row's identity lives only in its storage key. + server + .exec("INSERT INTO conv_minted { name: 'bob' }") + .await + .unwrap(); + + let before = server + .query_text("SELECT id FROM conv_minted") + .await + .unwrap(); + assert_eq!(before.len(), 1, "insert must produce one row: {before:?}"); + let identity = before[0].clone(); + + server + .exec("CONVERT COLLECTION conv_minted TO document_strict") + .await + .unwrap(); + + let after = server + .query_text("SELECT id FROM conv_minted") + .await + .unwrap(); + assert_eq!( + after.len(), + 1, + "the minted row must survive conversion: {after:?}" + ); + assert_eq!( + after[0], identity, + "conversion must not change the row's client-visible identity" + ); + + let names = server + .query_text("SELECT name FROM conv_minted") + .await + .unwrap(); + assert_eq!(names, vec!["bob".to_string()]); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn convert_to_schemaless_reencodes_rows_from_a_strict_source() { + let server = TestServer::start().await; + + server + .exec( + "CREATE COLLECTION conv_strict_src (\ + id TEXT PRIMARY KEY,\ + name TEXT,\ + age INT\ + ) WITH (engine='document_strict')", + ) + .await + .unwrap(); + + server + .exec("INSERT INTO conv_strict_src (id, name, age) VALUES ('s1', 'carol', 30)") + .await + .unwrap(); + + server + .exec("CONVERT COLLECTION conv_strict_src TO document_schemaless") + .await + .unwrap(); + + let rows = server + .query_named_rows("SELECT * FROM conv_strict_src WHERE id = 's1'") + .await + .unwrap(); + assert_eq!( + rows.len(), + 1, + "the Binary Tuple row must survive conversion: {rows:?}" + ); + assert_eq!( + rows[0].get("name").map(String::as_str), + Some("carol"), + "row: {:?}", + rows[0] + ); + assert_eq!( + rows[0].get("age").map(String::as_str), + Some("30"), + "row: {:?}", + rows[0] + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn convert_to_strict_fails_the_statement_when_a_row_cannot_encode() { + let server = TestServer::start().await; + + server + .exec("CREATE COLLECTION conv_fail_encode") + .await + .unwrap(); + + // No 'name' field — the target schema below requires it. + server + .exec("INSERT INTO conv_fail_encode { id: 'f1' }") + .await + .unwrap(); + + server + .expect_error( + "CONVERT COLLECTION conv_fail_encode TO document_strict \ + (id TEXT, name TEXT NOT NULL)", + "NOT NULL", + ) + .await; + + // The catalog must still show the collection as schemaless: a failed + // conversion must not flip the stored type over unconverted data. + let rows = server + .query_text_joined("SELECT * FROM conv_fail_encode WHERE id = 'f1'") + .await + .unwrap(); + assert_eq!( + rows.len(), + 1, + "the row must remain readable under the original type: {rows:?}" + ); +} diff --git a/nodedb/tests/wire/cases/sql_typeguard_defaults/defaults.rs b/nodedb/tests/wire/cases/sql_typeguard_defaults/defaults.rs new file mode 100644 index 000000000..9c91f220b --- /dev/null +++ b/nodedb/tests/wire/cases/sql_typeguard_defaults/defaults.rs @@ -0,0 +1,231 @@ +// SPDX-License-Identifier: BUSL-1.1 + +use crate::harness::TestServer; + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn typeguard_default_injects_when_absent() { + let server = TestServer::start().await; + + server.exec("CREATE COLLECTION tg_defaults").await.unwrap(); + + server + .exec( + "CREATE TYPEGUARD ON tg_defaults (\ + status STRING DEFAULT 'draft'\ + )", + ) + .await + .unwrap(); + + // Insert without status — DEFAULT should fill it. + server + .exec("INSERT INTO tg_defaults { id: 'd1', name: 'Alice' }") + .await + .unwrap(); + + let rows = server + .query_text_joined("SELECT * FROM tg_defaults WHERE id = 'd1'") + .await + .unwrap(); + assert_eq!(rows.len(), 1); + assert!( + rows[0].contains("draft"), + "DEFAULT should inject 'draft': {rows:?}" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn typeguard_default_does_not_overwrite() { + let server = TestServer::start().await; + + server + .exec("CREATE COLLECTION tg_no_overwrite") + .await + .unwrap(); + + server + .exec( + "CREATE TYPEGUARD ON tg_no_overwrite (\ + status STRING DEFAULT 'draft'\ + )", + ) + .await + .unwrap(); + + // Insert with explicit status — DEFAULT should NOT overwrite. + server + .exec("INSERT INTO tg_no_overwrite { id: 'd1', status: 'active' }") + .await + .unwrap(); + + let rows = server + .query_text_joined("SELECT * FROM tg_no_overwrite WHERE id = 'd1'") + .await + .unwrap(); + assert_eq!(rows.len(), 1); + assert!( + rows[0].contains("active"), + "DEFAULT should not overwrite user value: {rows:?}" + ); + assert!( + !rows[0].contains("draft"), + "should NOT contain default: {rows:?}" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn typeguard_value_always_overwrites() { + let server = TestServer::start().await; + + server + .exec("CREATE COLLECTION tg_value_overwrite") + .await + .unwrap(); + + server + .exec( + "CREATE TYPEGUARD ON tg_value_overwrite (\ + computed STRING VALUE 'server_computed'\ + )", + ) + .await + .unwrap(); + + // Insert with user-provided value — VALUE should overwrite. + server + .exec("INSERT INTO tg_value_overwrite { id: 'v1', computed: 'user_input' }") + .await + .unwrap(); + + let rows = server + .query_text_joined("SELECT * FROM tg_value_overwrite WHERE id = 'v1'") + .await + .unwrap(); + assert_eq!(rows.len(), 1); + assert!( + rows[0].contains("server_computed"), + "VALUE should overwrite user input: {rows:?}" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn typeguard_required_plus_default() { + let server = TestServer::start().await; + + server + .exec("CREATE COLLECTION tg_req_default") + .await + .unwrap(); + + server + .exec( + "CREATE TYPEGUARD ON tg_req_default (\ + version INT REQUIRED DEFAULT 1\ + )", + ) + .await + .unwrap(); + + // Insert without version — DEFAULT fills before REQUIRED check. + server + .exec("INSERT INTO tg_req_default { id: 'r1', name: 'test' }") + .await + .unwrap(); + + let rows = server + .query_text_joined("SELECT * FROM tg_req_default WHERE id = 'r1'") + .await + .unwrap(); + assert_eq!(rows.len(), 1); + assert!( + rows[0].contains('1'), + "REQUIRED + DEFAULT should inject version=1: {rows:?}" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn typeguard_default_integer() { + let server = TestServer::start().await; + + server + .exec("CREATE COLLECTION tg_int_default") + .await + .unwrap(); + + server + .exec( + "CREATE TYPEGUARD ON tg_int_default (\ + priority INT DEFAULT 0 CHECK (priority >= 0)\ + )", + ) + .await + .unwrap(); + + // Insert without priority — DEFAULT 0 should be injected and pass CHECK. + server + .exec("INSERT INTO tg_int_default { id: 'p1', name: 'test' }") + .await + .unwrap(); + + let rows = server + .query_text_joined("SELECT * FROM tg_int_default WHERE id = 'p1'") + .await + .unwrap(); + assert_eq!(rows.len(), 1); +} + +// ── DEFAULT gen_uuid_v7() / now() on strict schema ── + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn strict_default_gen_uuid_v7() { + let server = TestServer::start().await; + + server + .exec( + "CREATE COLLECTION strict_uuid (\ + id TEXT PRIMARY KEY DEFAULT gen_uuid_v7(),\ + name TEXT\ + ) WITH (engine='document_strict')", + ) + .await + .unwrap(); + + // Insert without id — DEFAULT gen_uuid_v7() should fill it. + let result = server + .exec("INSERT INTO strict_uuid (name) VALUES ('Alice')") + .await; + result.unwrap(); + + let rows = server + .query_text_joined("SELECT * FROM strict_uuid") + .await + .unwrap(); + assert_eq!(rows.len(), 1, "should have 1 row: {rows:?}"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn strict_default_now() { + let server = TestServer::start().await; + + server + .exec( + "CREATE COLLECTION strict_ts (\ + id TEXT PRIMARY KEY,\ + created_at TEXT DEFAULT now()\ + ) WITH (engine='document_strict')", + ) + .await + .unwrap(); + + // Insert without created_at — DEFAULT now() should fill it. + server + .exec("INSERT INTO strict_ts (id) VALUES ('t1')") + .await + .unwrap(); + + let rows = server + .query_text_joined("SELECT * FROM strict_ts WHERE id = 't1'") + .await + .unwrap(); + assert_eq!(rows.len(), 1, "should have 1 row: {rows:?}"); +} diff --git a/nodedb/tests/wire/cases/sql_typeguard_defaults/mod.rs b/nodedb/tests/wire/cases/sql_typeguard_defaults/mod.rs new file mode 100644 index 000000000..9b0360706 --- /dev/null +++ b/nodedb/tests/wire/cases/sql_typeguard_defaults/mod.rs @@ -0,0 +1,14 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Integration tests for typeguard DEFAULT/VALUE expressions and VALIDATE TYPEGUARD. +//! +//! Verifies that: +//! - DEFAULT injects a value when the field is absent +//! - DEFAULT does not overwrite user-provided values +//! - VALUE always overwrites, even when user provides a value +//! - REQUIRED + DEFAULT = field is always present +//! - Cross-field VALUE expressions resolve other document fields + +mod convert; +mod defaults; +mod validate; diff --git a/nodedb/tests/wire/cases/sql_typeguard_defaults/validate.rs b/nodedb/tests/wire/cases/sql_typeguard_defaults/validate.rs new file mode 100644 index 000000000..07625e6c2 --- /dev/null +++ b/nodedb/tests/wire/cases/sql_typeguard_defaults/validate.rs @@ -0,0 +1,131 @@ +// SPDX-License-Identifier: BUSL-1.1 + +use crate::harness::TestServer; + +// ── VALIDATE TYPEGUARD ── + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn validate_typeguard_no_violations() { + let server = TestServer::start().await; + + server.exec("CREATE COLLECTION val_clean").await.unwrap(); + + // Insert valid data first. + server + .exec("INSERT INTO val_clean { id: 'v1', name: 'Alice', age: 25 }") + .await + .unwrap(); + + // Add type guard after data. + server + .exec( + "CREATE TYPEGUARD ON val_clean (\ + name STRING,\ + age INT\ + )", + ) + .await + .unwrap(); + + // Validate — all docs should pass. + let rows = server + .query_text("VALIDATE TYPEGUARD ON val_clean") + .await + .unwrap(); + assert_eq!(rows.len(), 0, "no violations expected: {rows:?}"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn validate_typeguard_finds_violations() { + let server = TestServer::start().await; + + server.exec("CREATE COLLECTION val_dirty").await.unwrap(); + + // Insert data that will violate a future type guard. + server + .exec("INSERT INTO val_dirty { id: 'd1', name: 'Alice', score: 42 }") + .await + .unwrap(); + server + .exec("INSERT INTO val_dirty { id: 'd2', name: 123, score: 99 }") + .await + .unwrap(); + + // Add type guard — name must be STRING. + server + .exec( + "CREATE TYPEGUARD ON val_dirty (\ + name STRING\ + )", + ) + .await + .unwrap(); + + // Validate — d2 has name=123 (INT, not STRING). + let rows = server + .query_text("VALIDATE TYPEGUARD ON val_dirty") + .await + .unwrap(); + assert!( + !rows.is_empty(), + "should find at least one violation: {rows:?}" + ); + // First column is document_id — should be d2. + assert!( + rows.iter().any(|r| r.contains("d2")), + "violation should reference d2: {rows:?}" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn validate_typeguard_no_guards() { + let server = TestServer::start().await; + + server.exec("CREATE COLLECTION val_noguard").await.unwrap(); + + server + .exec("INSERT INTO val_noguard { id: 'n1', x: 1 }") + .await + .unwrap(); + + // No typeguard — should return empty result. + let rows = server + .query_text("VALIDATE TYPEGUARD ON val_noguard") + .await + .unwrap(); + assert_eq!(rows.len(), 0); +} + +// ── Unresolvable declared type ── + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn typeguard_unresolvable_type_is_refused_at_declaration() { + let server = TestServer::start().await; + + server.exec("CREATE COLLECTION tg_bad_type").await.unwrap(); + + // A type name the engine resolves to nothing. + server + .expect_error("CREATE TYPEGUARD ON tg_bad_type (gadget WIDGET)", "42601") + .await; + + // A trailing word that reading the leading token alone would ignore. + server + .expect_error( + "CREATE TYPEGUARD ON tg_bad_type (at TIMESTAMP GARBAGE)", + "42601", + ) + .await; + + // ALTER carries the same refusal. + server + .expect_error("ALTER TYPEGUARD ON tg_bad_type ADD gadget WIDGET", "42601") + .await; + + // A refused declaration reaches no storage. + let rows = server + .query_text("SHOW TYPEGUARD ON tg_bad_type") + .await + .unwrap(); + assert_eq!(rows.len(), 0, "refused guard must not be stored: {rows:?}"); +} From e30b3d05209f5aea363784b638ec5a0ff7a59c11 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 11 Sep 2026 22:33:19 +0800 Subject: [PATCH 05/17] fix(row-identity): resolve rows by surrogate storage key everywhere MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Row identity was resolved inconsistently across the write and read paths — some call sites treated a document's storage key as an opaque string, others re-derived it from user-supplied fields, letting a period-lock or materialized-sum reference resolve to a row that looked right but did not match the surrogate the Control Plane actually bound. Introduce nodedb_types::{StorageKey, RowIdentity} as the single encoding boundary: StorageKey is the internal fixed-width hex key a document is stored under, RowIdentity is what a client sees. Thread StorageKey through the document store, sparse btree, bulk DML, upsert/merge/transaction handlers, WAL replay, and diagnostics so every read and write resolves the same row the planner resolved. Extend period-lock enforcement to resolve the reference row through the Control Plane's resolved_targets surrogate binding instead of a raw key lookup, and add PeriodLockMisconfigured (SQLSTATE 23609) to distinguish a reference row missing its configured status column from an actually locked period. --- .../src/rpc_codec/data_plane_error.rs | 9 + nodedb-types/src/error/code.rs | 4 + nodedb-types/src/error/code_table.rs | 5 + nodedb-types/src/error/ctors/write_path.rs | 28 ++ nodedb-types/src/error/details.rs | 6 + nodedb-types/src/error/msgpack/constants.rs | 2 + .../error/msgpack/decode/from_messagepack.rs | 19 + nodedb-types/src/error/msgpack/encode.rs | 11 + nodedb-types/src/error/sqlstate.rs | 5 + nodedb-types/src/lib.rs | 4 + nodedb-types/src/row_identity.rs | 215 ++++++++++ nodedb/src/bridge/envelope/error_code.rs | 20 + .../control/cluster/data_plane_error_wire.rs | 22 ++ .../control/insert_select/expand_staged.rs | 16 +- .../src/control/insert_select/orchestrator.rs | 17 +- .../merge_orchestrator/orchestrator.rs | 17 +- .../control/planner/materialized_sum/mod.rs | 2 + .../planner/materialized_sum/predicate.rs | 28 +- .../control/planner/materialized_sum/recon.rs | 6 +- .../planner/materialized_sum/resolve.rs | 36 +- .../materialized_sum/resolve_target.rs | 35 ++ .../planner/materialized_sum/stored.rs | 40 +- nodedb/src/control/planner/mod.rs | 1 + .../src/control/planner/period_lock/gate.rs | 92 +++++ .../src/control/planner/period_lock/lookup.rs | 55 +++ nodedb/src/control/planner/period_lock/mod.rs | 16 + .../planner/period_lock/point_update.rs | 72 ++++ .../control/planner/period_lock/predicate.rs | 87 +++++ .../control/planner/period_lock/resolve.rs | 258 ++++++++++++ .../control/planner/period_lock/singular.rs | 108 +++++ .../server/dispatch_utils/write_abort.rs | 1 + .../server/native/dispatch/direct_ops.rs | 14 + .../server/pgwire/handler/routing/execute.rs | 13 + .../neutral/collection/dml/parse/dispatch.rs | 13 + .../src/control/server/shared/ddl/sqlstate.rs | 13 + .../control/server/shared/plan_admission.rs | 12 + .../orchestrator.rs | 40 +- .../executor/core_loop/doc_config_seed.rs | 4 +- nodedb/src/data/executor/core_loop/tick.rs | 17 +- .../core_loop/vector_index_rebuild.rs | 15 +- .../data/executor/enforcement/chain_guard.rs | 2 +- .../data/executor/enforcement/hash_chain.rs | 4 +- .../enforcement/materialized_sum/apply.rs | 47 ++- .../materialized_sum/divergence.rs | 8 +- .../enforcement/materialized_sum/rmw.rs | 10 +- .../data/executor/enforcement/period_lock.rs | 218 ++++++++++- .../data/executor/enforcement/statement.rs | 9 +- .../executor/handlers/bulk_dml/admission.rs | 11 +- .../data/executor/handlers/bulk_dml/delete.rs | 196 ++++------ .../handlers/bulk_dml/delete_cascade.rs | 183 +++++++++ .../data/executor/handlers/bulk_dml/mod.rs | 1 + .../data/executor/handlers/bulk_dml/update.rs | 33 +- .../handlers/bulk_dml/update_project.rs | 13 +- .../data/executor/handlers/control/calvin.rs | 5 +- .../control/calvin_overlay_stage_bulk.rs | 10 +- .../handlers/control/calvin_resolve.rs | 3 +- .../executor/handlers/control/crdt_doc.rs | 1 + .../handlers/control/crdt_materialize.rs | 1 + .../executor/handlers/control/crdt_preview.rs | 17 +- nodedb/src/data/executor/handlers/convert.rs | 5 +- .../handlers/document/apply_balance_delta.rs | 3 +- .../executor/handlers/document/index_fetch.rs | 9 +- .../handlers/document/index_maintenance.rs | 7 +- .../executor/handlers/document/read/fetch.rs | 229 +++++------ .../handlers/document/read/fetch_types.rs | 81 ++++ .../document/read/materialize_scan.rs | 7 +- .../executor/handlers/document/read/mod.rs | 1 + .../executor/handlers/document/read/scan.rs | 58 ++- .../handlers/document/resolve/apply.rs | 12 +- .../handlers/document/resolve/apply_row.rs | 8 +- .../handlers/document/resolve/bulk.rs | 33 +- .../handlers/document/resolve/context.rs | 12 +- .../handlers/document/resolve/point.rs | 6 +- .../handlers/document/resolve/upsert.rs | 3 +- .../handlers/document/write/batch_insert.rs | 73 ++-- nodedb/src/data/executor/handlers/facet.rs | 9 +- .../handlers/merge_orchestrated/abort.rs | 9 +- .../merge_orchestrated/apply/insert_rows.rs | 1 + .../merge_orchestrated/apply/update_rows.rs | 55 +-- .../merge_orchestrated/delete_arms.rs | 35 +- .../executor/handlers/point/apply_delete.rs | 31 +- .../executor/handlers/point/apply_put/core.rs | 21 +- .../handlers/point/apply_put/enforce.rs | 19 +- .../handlers/point/apply_put/types.rs | 17 + .../data/executor/handlers/point/delete.rs | 22 +- .../src/data/executor/handlers/point/get.rs | 10 +- .../data/executor/handlers/point/insert.rs | 57 ++- .../src/data/executor/handlers/point/put.rs | 50 ++- .../executor/handlers/point/update/exec.rs | 65 +++- .../executor/handlers/point/update/persist.rs | 14 +- .../executor/handlers/point/update_reindex.rs | 4 +- .../src/data/executor/handlers/recursive.rs | 4 +- .../data/executor/handlers/returning_rows.rs | 2 +- nodedb/src/data/executor/handlers/spatial.rs | 35 +- .../data/executor/handlers/spatial_sync.rs | 14 +- .../src/data/executor/handlers/text_search.rs | 3 +- .../executor/handlers/text_search_hybrid.rs | 16 +- .../executor/handlers/text_search_scan.rs | 7 +- .../executor/handlers/text_search_triple.rs | 16 +- .../executor/handlers/transaction/batch.rs | 11 +- .../handlers/transaction/resolve/entry.rs | 73 ++-- .../transaction/stage_write/constraint.rs | 20 +- .../stage_write/stage_bulk_update.rs | 8 +- .../stage_write/stage_point_document.rs | 15 +- .../transaction/stage_write/stage_upsert.rs | 5 +- .../transaction/sub_plan_doc/delete.rs | 1 + .../handlers/transaction/sub_plan_doc/put.rs | 42 +- .../handlers/transaction/undo/document.rs | 59 ++- .../handlers/transaction/undo/graph_node.rs | 1 + .../handlers/transaction/undo/rollback.rs | 2 + nodedb/src/data/executor/handlers/truncate.rs | 46 ++- .../handlers/update_from_join_write.rs | 67 +++- .../executor/handlers/upsert/exec/dispatch.rs | 12 +- .../executor/handlers/upsert/exec/insert.rs | 40 +- .../handlers/upsert/exec/overwrite.rs | 10 +- .../executor/handlers/vector_search_exec.rs | 5 +- .../data/executor/handlers/vector_upsert.rs | 7 +- .../data/executor/handlers/vector_write.rs | 3 +- .../src/data/executor/handlers/write_batch.rs | 2 + nodedb/src/data/executor/row_shape.rs | 20 +- nodedb/src/data/executor/scan_normalize.rs | 36 +- nodedb/src/data/executor/scan_versioned.rs | 11 +- nodedb/src/data/executor/wal_replay/crdt.rs | 5 +- .../data/executor/wal_replay_redo_document.rs | 26 +- .../src/data/executor/wal_replay_spatial.rs | 6 +- nodedb/src/diag/context/mod.rs | 4 +- nodedb/src/diag/context/write_path.rs | 51 +++ nodedb/src/diag/mod.rs | 11 +- nodedb/src/diag/recording/mod.rs | 5 +- nodedb/src/diag/recording/recovery.rs | 25 ++ .../src/engine/document/store/engine/batch.rs | 27 +- .../engine/document/store/engine/delete.rs | 26 +- .../src/engine/document/store/engine/get.rs | 35 +- .../src/engine/document/store/engine/put.rs | 36 +- nodedb/src/engine/document/store/key.rs | 223 +---------- .../graph/pattern/executor/core/triple.rs | 11 +- .../graph/pattern/executor/predicates.rs | 2 +- nodedb/src/engine/sparse/btree/document.rs | 67 ++-- nodedb/src/engine/sparse/btree/keys.rs | 8 +- nodedb/src/engine/sparse/btree/mod.rs | 2 +- nodedb/src/engine/sparse/btree/rename.rs | 8 +- nodedb/src/engine/sparse/btree/tables.rs | 16 + nodedb/src/engine/sparse/btree_scan.rs | 87 +++-- nodedb/src/engine/sparse/doc_cache.rs | 111 +++--- nodedb/src/error/types.rs | 14 + nodedb/src/error_classify.rs | 11 + nodedb/src/error_from_data_plane.rs | 11 + .../cases/collection_purge_persistent.rs | 35 +- .../inproc/cases/document_bitemporal_dml.rs | 59 +-- .../inproc/cases/document_bitemporal_store.rs | 39 +- .../wire/cases/enforcement_period_lock.rs | 368 ++++++++++++++++++ nodedb/tests/wire/cases/mod.rs | 1 + 152 files changed, 3817 insertions(+), 1222 deletions(-) create mode 100644 nodedb-types/src/row_identity.rs create mode 100644 nodedb/src/control/planner/materialized_sum/resolve_target.rs create mode 100644 nodedb/src/control/planner/period_lock/gate.rs create mode 100644 nodedb/src/control/planner/period_lock/lookup.rs create mode 100644 nodedb/src/control/planner/period_lock/mod.rs create mode 100644 nodedb/src/control/planner/period_lock/point_update.rs create mode 100644 nodedb/src/control/planner/period_lock/predicate.rs create mode 100644 nodedb/src/control/planner/period_lock/resolve.rs create mode 100644 nodedb/src/control/planner/period_lock/singular.rs create mode 100644 nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs create mode 100644 nodedb/src/data/executor/handlers/document/read/fetch_types.rs create mode 100644 nodedb/tests/wire/cases/enforcement_period_lock.rs diff --git a/nodedb-cluster/src/rpc_codec/data_plane_error.rs b/nodedb-cluster/src/rpc_codec/data_plane_error.rs index 2309d072a..45927be57 100644 --- a/nodedb-cluster/src/rpc_codec/data_plane_error.rs +++ b/nodedb-cluster/src/rpc_codec/data_plane_error.rs @@ -112,4 +112,13 @@ pub enum DataPlaneErrorCode { limit: u64, }, DivisionByZero, + /// A period-lock reference row exists but does not carry the + /// configured `status_column` — a misconfigured column name, not a + /// locked period. + PeriodLockMisconfigured { + collection: String, + ref_table: String, + status_column: String, + row_identity: String, + }, } diff --git a/nodedb-types/src/error/code.rs b/nodedb-types/src/error/code.rs index e689c0263..2ff0461b7 100644 --- a/nodedb-types/src/error/code.rs +++ b/nodedb-types/src/error/code.rs @@ -23,6 +23,10 @@ impl ErrorCode { pub const TRANSITION_CHECK_VIOLATION: Self = Self(1014); pub const RETENTION_VIOLATION: Self = Self(1015); pub const LEGAL_HOLD_ACTIVE: Self = Self(1016); + /// A period-lock reference row exists but does not carry the + /// configured `status_column` — a misconfigured column name, not a + /// locked period. + pub const PERIOD_LOCK_MISCONFIGURED: Self = Self(1017); pub const TYPE_MISMATCH: Self = Self(1020); pub const OVERFLOW: Self = Self(1021); pub const INSUFFICIENT_BALANCE: Self = Self(1022); diff --git a/nodedb-types/src/error/code_table.rs b/nodedb-types/src/error/code_table.rs index e319e5a59..f9fdfdc14 100644 --- a/nodedb-types/src/error/code_table.rs +++ b/nodedb-types/src/error/code_table.rs @@ -57,6 +57,11 @@ error_code_table! { APPEND_ONLY_VIOLATION => AppendOnlyViolation { collection: String::new() }, BALANCE_VIOLATION => BalanceViolation { collection: String::new() }, PERIOD_LOCKED => PeriodLocked { collection: String::new() }, + PERIOD_LOCK_MISCONFIGURED => PeriodLockMisconfigured { + collection: String::new(), + ref_table: String::new(), + status_column: String::new(), + }, STATE_TRANSITION_VIOLATION => StateTransitionViolation { collection: String::new() }, TRANSITION_CHECK_VIOLATION => TransitionCheckViolation { collection: String::new() }, TYPE_GUARD_VIOLATION => TypeGuardViolation { collection: String::new() }, diff --git a/nodedb-types/src/error/ctors/write_path.rs b/nodedb-types/src/error/ctors/write_path.rs index 130b9d338..0970ccacd 100644 --- a/nodedb-types/src/error/ctors/write_path.rs +++ b/nodedb-types/src/error/ctors/write_path.rs @@ -99,6 +99,34 @@ impl NodeDbError { } } + /// A period-lock reference row exists but does not carry the configured + /// `status_column` — a misconfigured column name, refused as a config + /// error rather than admitted or treated as locked. `row_identity` + /// names the reference row that is missing the column. + pub fn period_lock_misconfigured( + collection: impl Into, + ref_table: impl Into, + status_column: impl Into, + row_identity: impl fmt::Display, + ) -> Self { + let collection = collection.into(); + let ref_table = ref_table.into(); + let status_column = status_column.into(); + Self { + code: ErrorCode::PERIOD_LOCK_MISCONFIGURED, + message: format!( + "period lock on {collection} misconfigured: reference table \ + '{ref_table}' row '{row_identity}' has no column '{status_column}'" + ), + details: ErrorDetails::PeriodLockMisconfigured { + collection, + ref_table, + status_column, + }, + cause: None, + } + } + pub fn state_transition_violation( collection: impl Into, detail: impl fmt::Display, diff --git a/nodedb-types/src/error/details.rs b/nodedb-types/src/error/details.rs index c24285c3b..fbf0e4fc4 100644 --- a/nodedb-types/src/error/details.rs +++ b/nodedb-types/src/error/details.rs @@ -38,6 +38,12 @@ pub enum ErrorDetails { BalanceViolation { collection: String }, #[serde(rename = "period_locked")] PeriodLocked { collection: String }, + #[serde(rename = "period_lock_misconfigured")] + PeriodLockMisconfigured { + collection: String, + ref_table: String, + status_column: String, + }, #[serde(rename = "state_transition_violation")] StateTransitionViolation { collection: String }, #[serde(rename = "transition_check_violation")] diff --git a/nodedb-types/src/error/msgpack/constants.rs b/nodedb-types/src/error/msgpack/constants.rs index 2de2601e2..9627de4b7 100644 --- a/nodedb-types/src/error/msgpack/constants.rs +++ b/nodedb-types/src/error/msgpack/constants.rs @@ -84,6 +84,7 @@ // | 78 | InvalidLimitValue | // | 79 | UndefinedColumn | // | 80 | AmbiguousColumn | +// | 81 | PeriodLockMisconfigured | pub(super) const TAG_CONSTRAINT_VIOLATION: u16 = 1; pub(super) const TAG_WRITE_CONFLICT: u16 = 2; @@ -165,3 +166,4 @@ pub(super) const TAG_CANNOT_DROP_DEFAULT_DATABASE: u16 = 77; pub(super) const TAG_INVALID_LIMIT_VALUE: u16 = 78; pub(super) const TAG_UNDEFINED_COLUMN: u16 = 79; pub(super) const TAG_AMBIGUOUS_COLUMN: u16 = 80; +pub(super) const TAG_PERIOD_LOCK_MISCONFIGURED: u16 = 81; diff --git a/nodedb-types/src/error/msgpack/decode/from_messagepack.rs b/nodedb-types/src/error/msgpack/decode/from_messagepack.rs index e0704fa86..5666fcd4d 100644 --- a/nodedb-types/src/error/msgpack/decode/from_messagepack.rs +++ b/nodedb-types/src/error/msgpack/decode/from_messagepack.rs @@ -51,6 +51,15 @@ impl<'a> FromMessagePack<'a> for ErrorDetails { let (collection,) = read1_str(reader, field_count)?; Ok(ErrorDetails::PeriodLocked { collection }) } + TAG_PERIOD_LOCK_MISCONFIGURED => { + let (collection, ref_table, status_column) = + read3_str_tolerant(reader, field_count)?; + Ok(ErrorDetails::PeriodLockMisconfigured { + collection, + ref_table, + status_column, + }) + } TAG_STATE_TRANSITION_VIOLATION => { let (collection,) = read1_str(reader, field_count)?; Ok(ErrorDetails::StateTransitionViolation { collection }) @@ -766,4 +775,14 @@ mod tests { }; assert_eq!(roundtrip(&v), v); } + + #[test] + fn period_lock_misconfigured_roundtrip() { + let v = ErrorDetails::PeriodLockMisconfigured { + collection: "journal_entries".into(), + ref_table: "fiscal_periods".into(), + status_column: "status".into(), + }; + assert_eq!(roundtrip(&v), v); + } } diff --git a/nodedb-types/src/error/msgpack/encode.rs b/nodedb-types/src/error/msgpack/encode.rs index ab06a7882..dfb983f48 100644 --- a/nodedb-types/src/error/msgpack/encode.rs +++ b/nodedb-types/src/error/msgpack/encode.rs @@ -102,6 +102,17 @@ impl ToMessagePack for ErrorDetails { ErrorDetails::PeriodLocked { collection } => { write1(writer, TAG_PERIOD_LOCKED, collection) } + ErrorDetails::PeriodLockMisconfigured { + collection, + ref_table, + status_column, + } => write3( + writer, + TAG_PERIOD_LOCK_MISCONFIGURED, + collection, + ref_table, + status_column, + ), ErrorDetails::StateTransitionViolation { collection } => { write1(writer, TAG_STATE_TRANSITION_VIOLATION, collection) } diff --git a/nodedb-types/src/error/sqlstate.rs b/nodedb-types/src/error/sqlstate.rs index 021e979ac..076dc0758 100644 --- a/nodedb-types/src/error/sqlstate.rs +++ b/nodedb-types/src/error/sqlstate.rs @@ -105,6 +105,10 @@ pub const LEGAL_HOLD_ACTIVE: &str = "23607"; /// `23608` — NodeDB extension: type-guard constraint violated. pub const TYPE_GUARD_VIOLATION: &str = "23608"; +/// `23609` — NodeDB extension: period-lock reference row has no configured +/// status column; a misconfigured column name, not a locked period. +pub const PERIOD_LOCK_MISCONFIGURED: &str = "23609"; + // ── Class 28 — Invalid Authorization Specification ─────────────────────────── /// `28000` — `invalid_authorization_specification` (no valid credentials) @@ -346,6 +350,7 @@ mod tests { APPEND_ONLY_VIOLATION, BALANCE_VIOLATION, PERIOD_LOCKED, + PERIOD_LOCK_MISCONFIGURED, STATE_TRANSITION_VIOLATION, TRANSITION_CHECK_VIOLATION, RETENTION_VIOLATION, diff --git a/nodedb-types/src/lib.rs b/nodedb-types/src/lib.rs index afcad8014..2c8587839 100644 --- a/nodedb-types/src/lib.rs +++ b/nodedb-types/src/lib.rs @@ -52,6 +52,7 @@ pub mod pg_compat; pub mod protocol; pub mod result; pub mod rls_write_check; +pub mod row_identity; pub mod sparse_vector; pub mod sql_quote; pub mod surrogate; @@ -118,6 +119,9 @@ pub use quota::{ }; pub use result::{QueryResult, SearchResult, SubGraph}; pub use rls_write_check::{RlsWriteCheck, WriteGateDecision}; +pub use row_identity::{ + RowIdentity, StorageKey, doc_id_to_surrogate, identity_of, surrogate_to_doc_id, +}; pub use sparse_vector::{SparseVector, SparseVectorError}; pub use sql_quote::{quote_ident, quote_literal}; pub use surrogate::Surrogate; diff --git a/nodedb-types/src/row_identity.rs b/nodedb-types/src/row_identity.rs new file mode 100644 index 000000000..e12e69893 --- /dev/null +++ b/nodedb-types/src/row_identity.rs @@ -0,0 +1,215 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! This module owns both encodings of a row's surrogate identity. +//! +//! [`StorageKey`] is internal: the substrate redb tables (`DOCUMENTS`, +//! `INDEXES`) use string keys, and the Document engine encodes each +//! surrogate as a fixed-width 8-character lowercase hexadecimal string +//! (e.g. `Surrogate(42)` -> `"0000002a"`). It must never reach a client. +//! +//! The format is intentionally fixed width: lexicographic ordering of the +//! hex string matches numeric ordering of the underlying surrogate, so +//! redb range scans can iterate rows in surrogate order without any +//! additional index. +//! +//! [`RowIdentity`] is what a client sees: a predicate matches it, and the +//! catalog binds it. For a minted row it is the surrogate's decimal string. +//! A row with a user-supplied or DDL-declared primary key carries a +//! different identity, the key's body value, bound via the +//! `_system.surrogate_pk{,_rev}` catalog tables. +//! +//! The two types exist so the compiler rejects passing one where the other +//! belongs. A storage key and a user-declared identity can share the same +//! 8-hex-character shape (a primary key `deadbeef` parses as a storage key), +//! so a loose `String` cannot tell them apart. `StorageKey::parse` is the +//! only place that shape gets reinterpreted as a surrogate. +use crate::Surrogate; + +/// The redb key a document row is stored under. +/// +/// Fixed-width lowercase hex, so lexicographic order matches surrogate order +/// and a range scan iterates rows in surrogate order with no extra index. +/// Internal: a storage key never reaches a client. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct StorageKey(Surrogate); + +impl StorageKey { + /// Wrap `surrogate` as the key it is stored under. Allocation-free: the + /// hex text is a rendering, produced only by `Display` or `to_identity`. + pub fn for_surrogate(surrogate: Surrogate) -> Self { + Self(surrogate) + } + + /// Parse a redb key back into a `StorageKey`. + /// + /// Returns `None` unless `key` is exactly 8 lowercase hex characters. + /// This handles legacy non-surrogate document IDs gracefully: a value + /// that fails to parse is not a minted storage key. Allocation-free. + pub fn parse(key: &str) -> Option { + if key.len() != 8 + || !key + .bytes() + .all(|b| b.is_ascii_hexdigit() && !b.is_ascii_uppercase()) + { + return None; + } + let raw = u32::from_str_radix(key, 16).ok()?; + Some(Self(Surrogate::new(raw))) + } + + /// Recover the surrogate this key encodes. Infallible and free: the + /// surrogate IS the key, not something re-derived from stored text. + pub fn surrogate(&self) -> Surrogate { + self.0 + } + + /// The client-visible identity of the row stored under this key. + /// + /// Allocates the decimal string — unavoidable, since a client never sees + /// the hex encoding. + pub fn to_identity(&self) -> RowIdentity { + RowIdentity::for_surrogate(self.0) + } +} + +impl std::fmt::Display for StorageKey { + /// The only place a surrogate becomes a storage key. + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "{:08x}", self.0.as_u32()) + } +} + +/// The identity a client sees for a row. +/// +/// A minted row renders its surrogate in decimal. A row with a declared +/// `PRIMARY KEY` carries the user's own value. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct RowIdentity(String); + +impl RowIdentity { + /// The decimal identity of a minted row. + pub fn for_surrogate(surrogate: Surrogate) -> Self { + Self(surrogate.as_u32().to_string()) + } + + /// Wrap a declared or client-supplied key as the row's identity. + /// + /// The value is taken verbatim and never interpreted as a storage key + /// or a surrogate. This is what makes it safe to call with a KV key, a + /// user-declared primary key, or any other engine-native identifier. + /// + /// Takes `impl Into` so an owned `String` moves in without a + /// copy. A `&str` caller still pays one copy, which is unavoidable. + pub fn from_user_key(key: impl Into) -> Self { + Self(key.into()) + } + + pub fn as_str(&self) -> &str { + &self.0 + } + + /// Consume this identity and return its inner `String` without copying. + pub fn into_string(self) -> String { + self.0 + } +} + +impl std::fmt::Display for RowIdentity { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(&self.0) + } +} + +/// Format a surrogate as the 8-character zero-padded lowercase hex string +/// used as the document's redb key. +/// +/// Thin wrapper over [`StorageKey::for_surrogate`], kept because 242 +/// call sites across the workspace hold the result as a plain `String` +/// (redb key params, msgpack field injection, WAL replay) rather than a +/// `StorageKey`. Converting all of them is a separate ripple from this one. +/// One allocation: the `Display` format. +pub fn surrogate_to_doc_id(surrogate: Surrogate) -> String { + StorageKey::for_surrogate(surrogate).to_string() +} + +/// Parse a hex-encoded document storage key back to a `Surrogate`. +/// +/// Returns `None` if the key is not exactly 8 lowercase hex characters — +/// this handles legacy non-surrogate document IDs gracefully. +/// +/// Thin wrapper over [`StorageKey::parse`], kept for the same reason as +/// [`surrogate_to_doc_id`]: 35 call sites hold a plain `&str` doc ID. +/// Allocation-free. +pub fn doc_id_to_surrogate(doc_id: &str) -> Option { + StorageKey::parse(doc_id).map(|key| key.surrogate()) +} + +/// The client-visible identity of a row stored under `doc_id`. +/// +/// A minted key renders its surrogate in decimal. Any other key is a user's +/// own value and passes through verbatim. +pub fn identity_of(doc_id: &str) -> RowIdentity { + StorageKey::parse(doc_id) + .map(|key| key.to_identity()) + .unwrap_or_else(|| RowIdentity::from_user_key(doc_id)) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn formats_zero_padded_lowercase() { + assert_eq!( + StorageKey::for_surrogate(Surrogate::new(0)).to_string(), + "00000000" + ); + assert_eq!( + StorageKey::for_surrogate(Surrogate::new(42)).to_string(), + "0000002a" + ); + assert_eq!( + StorageKey::for_surrogate(Surrogate::new(0xDEAD_BEEF)).to_string(), + "deadbeef" + ); + } + + #[test] + fn lex_order_matches_numeric() { + let a = StorageKey::for_surrogate(Surrogate::new(0x10)).to_string(); + let b = StorageKey::for_surrogate(Surrogate::new(0x100)).to_string(); + assert!(a < b); + } + + #[test] + fn parse_rejects_wrong_shape() { + assert!(StorageKey::parse("").is_none()); + assert!(StorageKey::parse("abc").is_none()); + assert!(StorageKey::parse("DEADBEEF").is_none()); + assert!(StorageKey::parse("zzzzzzzz").is_none()); + assert!(StorageKey::parse("123456789").is_none()); + } + + #[test] + fn parse_roundtrips_surrogate() { + let key = StorageKey::for_surrogate(Surrogate::new(0x2a)); + let parsed = StorageKey::parse(&key.to_string()).expect("valid storage key"); + assert_eq!(parsed.surrogate(), Surrogate::new(0x2a)); + } + + #[test] + fn to_identity_is_decimal() { + let key = StorageKey::for_surrogate(Surrogate::new(42)); + assert_eq!(key.to_identity().as_str(), "42"); + } + + #[test] + fn identity_of_minted_key_is_decimal() { + assert_eq!(identity_of("0000002a").as_str(), "42"); + } + + #[test] + fn identity_of_user_key_passes_through() { + assert_eq!(identity_of("user-declared-id").as_str(), "user-declared-id"); + } +} diff --git a/nodedb/src/bridge/envelope/error_code.rs b/nodedb/src/bridge/envelope/error_code.rs index 3c10d7422..b0a115121 100644 --- a/nodedb/src/bridge/envelope/error_code.rs +++ b/nodedb/src/bridge/envelope/error_code.rs @@ -57,6 +57,15 @@ pub enum ErrorCode { BalanceViolation { collection: String, detail: String }, /// Period is closed/locked: writes rejected. PeriodLocked { collection: String }, + /// A period-lock reference row exists but does not carry the + /// configured `status_column` — a misconfigured column name, refused + /// as a config error rather than admitted or treated as locked. + PeriodLockMisconfigured { + collection: String, + ref_table: String, + status_column: String, + row_identity: String, + }, /// Retention period not expired: DELETE rejected. RetentionViolation { collection: String }, /// Legal hold active: DELETE rejected. @@ -168,6 +177,17 @@ impl From for ErrorCode { ), }, crate::Error::PeriodLocked { collection, .. } => Self::PeriodLocked { collection }, + crate::Error::PeriodLockMisconfigured { + collection, + ref_table, + status_column, + row_identity, + } => Self::PeriodLockMisconfigured { + collection, + ref_table, + status_column, + row_identity, + }, crate::Error::RetentionViolation { collection, .. } => { Self::RetentionViolation { collection } } diff --git a/nodedb/src/control/cluster/data_plane_error_wire.rs b/nodedb/src/control/cluster/data_plane_error_wire.rs index 1d54be1a5..2ad97269b 100644 --- a/nodedb/src/control/cluster/data_plane_error_wire.rs +++ b/nodedb/src/control/cluster/data_plane_error_wire.rs @@ -89,6 +89,17 @@ impl From for DataPlaneErrorCode { Self::BalanceViolation { collection, detail } } ErrorCode::PeriodLocked { collection } => Self::PeriodLocked { collection }, + ErrorCode::PeriodLockMisconfigured { + collection, + ref_table, + status_column, + row_identity, + } => Self::PeriodLockMisconfigured { + collection, + ref_table, + status_column, + row_identity, + }, ErrorCode::RetentionViolation { collection } => Self::RetentionViolation { collection }, ErrorCode::LegalHoldActive { collection } => Self::LegalHoldActive { collection }, ErrorCode::StateTransitionViolation { collection, detail } => { @@ -171,6 +182,17 @@ impl From for ErrorCode { Self::BalanceViolation { collection, detail } } DataPlaneErrorCode::PeriodLocked { collection } => Self::PeriodLocked { collection }, + DataPlaneErrorCode::PeriodLockMisconfigured { + collection, + ref_table, + status_column, + row_identity, + } => Self::PeriodLockMisconfigured { + collection, + ref_table, + status_column, + row_identity, + }, DataPlaneErrorCode::RetentionViolation { collection } => { Self::RetentionViolation { collection } } diff --git a/nodedb/src/control/insert_select/expand_staged.rs b/nodedb/src/control/insert_select/expand_staged.rs index bd2e1f1c7..583c42011 100644 --- a/nodedb/src/control/insert_select/expand_staged.rs +++ b/nodedb/src/control/insert_select/expand_staged.rs @@ -77,7 +77,7 @@ pub(crate) async fn resolve_and_emit_insert_select_ops( // statement-level resolution, so without this a bound target collection // would fold against an empty resolution. let sum_bodies: Vec<&[u8]> = rows.iter().map(|(_, value, _)| value.as_slice()).collect(); - let resolved_sum_targets = + let mut resolved_sum_targets = crate::control::planner::materialized_sum::resolve_sum_targets_for_bodies( state, &sum_bodies, @@ -87,6 +87,20 @@ pub(crate) async fn resolve_and_emit_insert_select_ops( crate::types::TraceId::ZERO, ) .await?; + // Period-lock target for the same rows, in the SAME slot — a target + // collection under a period lock reads its reference row's surrogate off + // `resolved_sum_targets` exactly like a materialized-sum fold does. + resolved_sum_targets.extend( + crate::control::planner::period_lock::resolve_period_lock_targets_for_bodies( + state, + &sum_bodies, + target_collection.as_str(), + tenant_id, + task.database_id, + crate::types::TraceId::ZERO, + ) + .await?, + ); let mut out: Vec = Vec::with_capacity(rows.len()); for (document_id, value, surrogate) in rows { diff --git a/nodedb/src/control/insert_select/orchestrator.rs b/nodedb/src/control/insert_select/orchestrator.rs index 2377e1ce6..0c2d5c442 100644 --- a/nodedb/src/control/insert_select/orchestrator.rs +++ b/nodedb/src/control/insert_select/orchestrator.rs @@ -137,7 +137,7 @@ pub(crate) async fn run_insert_select( // atomic write. let page_bodies: Vec<&[u8]> = documents.iter().map(|(_, body)| body.as_slice()).collect(); - let resolved_sum_targets = + let mut resolved_sum_targets = crate::control::planner::materialized_sum::resolve_sum_targets_for_bodies( state, &page_bodies, @@ -147,6 +147,21 @@ pub(crate) async fn run_insert_select( crate::types::TraceId::ZERO, ) .await?; + // Period-lock target for the same page, in the SAME slot — a + // target collection under a period lock reads its reference row's + // surrogate off `resolved_sum_targets` exactly like a + // materialized-sum fold does. + resolved_sum_targets.extend( + crate::control::planner::period_lock::resolve_period_lock_targets_for_bodies( + state, + &page_bodies, + req.target_collection, + tenant_id, + database_id, + crate::types::TraceId::ZERO, + ) + .await?, + ); let plan = PhysicalPlan::Document(DocumentOp::BatchInsert { collection: nodedb_types::QualifiedCollection::from_stored( diff --git a/nodedb/src/control/merge_orchestrator/orchestrator.rs b/nodedb/src/control/merge_orchestrator/orchestrator.rs index d7867d41f..12f6b3d25 100644 --- a/nodedb/src/control/merge_orchestrator/orchestrator.rs +++ b/nodedb/src/control/merge_orchestrator/orchestrator.rs @@ -166,7 +166,7 @@ pub(crate) async fn run_merge(state: &SharedState, args: MergeArgs<'_>) -> crate .chain(arms.deletes.iter().map(|(_, _, body)| body.as_slice())) .chain(arms.inserts.iter().map(|(_, body)| body.as_slice())) .collect(); - let resolved_sum_targets = resolve_sum_targets_for_bodies( + let mut resolved_sum_targets = resolve_sum_targets_for_bodies( state, &sum_bodies, args.target_collection, @@ -175,6 +175,21 @@ pub(crate) async fn run_merge(state: &SharedState, args: MergeArgs<'_>) -> crate crate::types::TraceId::ZERO, ) .await?; + // Period-lock targets for the same arms, in the SAME slot: an UPDATE or + // DELETE arm on a period-locked target reads its reference row's + // surrogate off `resolved_sum_targets` exactly like a materialized-sum + // fold does, and an unresolved period refuses the whole apply. + resolved_sum_targets.extend( + crate::control::planner::period_lock::resolve_period_lock_targets_for_bodies( + state, + &sum_bodies, + args.target_collection, + args.tenant_id, + args.database_id, + crate::types::TraceId::ZERO, + ) + .await?, + ); let insert_rows = arms.inserts; diff --git a/nodedb/src/control/planner/materialized_sum/mod.rs b/nodedb/src/control/planner/materialized_sum/mod.rs index 53f431dc3..637cef9cb 100644 --- a/nodedb/src/control/planner/materialized_sum/mod.rs +++ b/nodedb/src/control/planner/materialized_sum/mod.rs @@ -21,6 +21,7 @@ pub mod index; pub mod predicate; pub mod recon; pub mod resolve; +pub mod resolve_target; pub mod settle; pub mod stored; @@ -30,3 +31,4 @@ pub use index::MaterializedSumIndex; pub use resolve::{ resolve_materialized_sum_targets, resolve_sum_targets_for_bodies, source_drives_bindings, }; +pub use resolve_target::resolve_one_target; diff --git a/nodedb/src/control/planner/materialized_sum/predicate.rs b/nodedb/src/control/planner/materialized_sum/predicate.rs index 9c30f5f7c..17071cc87 100644 --- a/nodedb/src/control/planner/materialized_sum/predicate.rs +++ b/nodedb/src/control/planner/materialized_sum/predicate.rs @@ -21,6 +21,7 @@ use nodedb_types::id::TxnId; use super::recon::recon_scan_rows; use super::resolve::{lookup_join_value, source_drives_bindings}; +use super::resolve_target::resolve_one_target; use super::settle::{ SettleInput, Settlement, co_resident_target_keys, omit_shipped, settle_cross_shard_images, }; @@ -156,26 +157,17 @@ async fn resolve_scanned_rows( let mut resolved: Vec = Vec::new(); for binding in bindings.iter() { for join_value in crate::query::binding_join_keys(binding, updates, rows)? { - if resolved - .iter() - .any(|entry| entry.addresses(&binding.target_collection, &join_value)) - { - continue; - } - let surrogate = lookup_join_value( - state, - binding, - &join_value, - tenant_id, - database_id, - trace_id, - ) - .await?; - resolved.push(ResolvedSumTarget::new( + resolve_one_target( + &mut resolved, &binding.target_collection, join_value, - surrogate, - )); + async |v| { + lookup_join_value(state, binding, v, tenant_id, database_id, trace_id) + .await + .map(Some) + }, + ) + .await?; } } Ok(resolved) diff --git a/nodedb/src/control/planner/materialized_sum/recon.rs b/nodedb/src/control/planner/materialized_sum/recon.rs index 961b3b024..8ef97adbc 100644 --- a/nodedb/src/control/planner/materialized_sum/recon.rs +++ b/nodedb/src/control/planner/materialized_sum/recon.rs @@ -42,7 +42,7 @@ use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; /// read-set entry so the Calvin OCC check aborts the statement if the source /// rows moved between this read and the apply. Rows without their version would /// be a silently stale total. -pub(super) struct ReconRead { +pub(crate) struct ReconRead { /// The decoded rows. pub rows: T, /// The source collection's write floor at read time — the comparand @@ -61,7 +61,7 @@ pub(super) struct ReconRead { /// /// Empty `filters` means "no WHERE clause" — every row, which is what `TRUNCATE` /// needs. -pub(super) async fn recon_scan_rows( +pub(in crate::control::planner) async fn recon_scan_rows( state: &SharedState, tenant_id: TenantId, database_id: DatabaseId, @@ -108,7 +108,7 @@ pub(super) async fn recon_scan_rows( /// /// Identity is the surrogate, exactly as on the write path — `document_id` is /// the user-facing primary key and carries no storage addressing. -pub(super) async fn recon_point_row( +pub(crate) async fn recon_point_row( state: &SharedState, tenant_id: TenantId, database_id: DatabaseId, diff --git a/nodedb/src/control/planner/materialized_sum/resolve.rs b/nodedb/src/control/planner/materialized_sum/resolve.rs index b901d67cd..38848361a 100644 --- a/nodedb/src/control/planner/materialized_sum/resolve.rs +++ b/nodedb/src/control/planner/materialized_sum/resolve.rs @@ -12,6 +12,7 @@ use nodedb_physical::physical_task::PhysicalTask; use nodedb_types::Surrogate; use super::extract::join_value_from_body; +use super::resolve_target::resolve_one_target; use super::settle::{ SettleInput, co_resident_target_keys, omit_shipped, settle_cross_shard_images, }; @@ -383,7 +384,6 @@ async fn resolve_bodies( ) -> crate::Result> { let mut resolved: Vec = Vec::new(); for binding in bindings.iter() { - let target = db_qualified(database_id, &binding.target_collection); for body in bodies { let Some(join_value) = join_value_from_body(body, &binding.join_column) else { // The row does not carry this binding's join column, so it does @@ -391,33 +391,17 @@ async fn resolve_bodies( // and nothing to add a delta to. continue; }; - if resolved - .iter() - .any(|entry| entry.addresses(&binding.target_collection, &join_value)) - { - continue; - } - let vshard = VShardId::from_key(join_value.as_bytes()); - let surrogate = lookup_surrogate_routed( - state, - vshard, - database_id, - tenant_id, - target.as_str(), - join_value.as_bytes(), - trace_id, - ) - .await? - .ok_or_else(|| crate::Error::MaterializedSumTargetNotFound { - target_collection: binding.target_collection.clone(), - join_column: binding.join_column.clone(), - join_value: join_value.clone(), - })?; - resolved.push(ResolvedSumTarget::new( + resolve_one_target( + &mut resolved, &binding.target_collection, join_value, - surrogate, - )); + async |v| { + lookup_join_value(state, binding, v, tenant_id, database_id, trace_id) + .await + .map(Some) + }, + ) + .await?; } } Ok(resolved) diff --git a/nodedb/src/control/planner/materialized_sum/resolve_target.rs b/nodedb/src/control/planner/materialized_sum/resolve_target.rs new file mode 100644 index 000000000..77a8fae9a --- /dev/null +++ b/nodedb/src/control/planner/materialized_sum/resolve_target.rs @@ -0,0 +1,35 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The shared "dedupe, resolve, push" step every materialized-sum and +//! period-lock resolution loop performs once per candidate `(target, value)` +//! pair. + +use nodedb_physical::physical_plan::ResolvedSumTarget; +use nodedb_types::Surrogate; + +/// Resolve one `(target, value)` pair into `resolved`, deduping on the pair +/// and skipping the push when `lookup` names no row. +/// +/// One entry per DISTINCT `(target, value)` pair: a page or scan that touches +/// the same target row many times resolves it once, so every caller checks +/// `resolved` before spending a lookup on a value it already holds. +/// +/// `lookup` differs by caller: a period-lock lookup can name no reference row +/// (`Ok(None)` skips the push, leaving the value unresolved for +/// `check_period_lock` to refuse), while a materialized-sum lookup always +/// resolves or fails outright — its `Err` propagates through the `?` below +/// before `Ok(None)` could ever apply. +pub async fn resolve_one_target( + resolved: &mut Vec, + target: &str, + value: String, + lookup: impl AsyncFnOnce(&str) -> crate::Result>, +) -> crate::Result<()> { + if resolved.iter().any(|entry| entry.addresses(target, &value)) { + return Ok(()); + } + if let Some(surrogate) = lookup(&value).await? { + resolved.push(ResolvedSumTarget::new(target, value, surrogate)); + } + Ok(()) +} diff --git a/nodedb/src/control/planner/materialized_sum/stored.rs b/nodedb/src/control/planner/materialized_sum/stored.rs index 7f03866c0..3701a7f88 100644 --- a/nodedb/src/control/planner/materialized_sum/stored.rs +++ b/nodedb/src/control/planner/materialized_sum/stored.rs @@ -29,6 +29,7 @@ use nodedb_types::Surrogate; use super::recon::recon_point_row; use super::resolve::lookup_join_value; +use super::resolve_target::resolve_one_target; use crate::control::state::SharedState; use crate::types::{DatabaseId, Lsn, TenantId, TraceId}; @@ -228,34 +229,25 @@ pub(super) async fn extend_with_stored_row( rows.extend(old.iter().cloned()); rows.extend(new.iter().cloned()); } + // One entry per DISTINCT `(target collection, join value)` PAIR across + // every binding and every source of join values, mirroring the + // body-driven resolution: a write whose old and new join keys are the + // same resolves that target once, while two bindings that share a join + // column and name different targets each keep their own entry. + // `resolve_one_target` enforces the dedupe against `resolved`. for binding in bindings.iter() { for join_value in crate::query::binding_join_keys(binding, &[], &rows)? { - // One entry per DISTINCT `(target collection, join value)` PAIR - // across every binding and every source of join values, mirroring - // the body-driven resolution: a write whose old and new join keys - // are the same resolves that target once, while two bindings that - // share a join column and name different targets each keep their - // own entry. - if resolved - .iter() - .any(|entry| entry.addresses(&binding.target_collection, &join_value)) - { - continue; - } - let surrogate = lookup_join_value( - state, - binding, - &join_value, - tenant_id, - database_id, - trace_id, - ) - .await?; - resolved.push(ResolvedSumTarget::new( + resolve_one_target( + resolved, &binding.target_collection, join_value, - surrogate, - )); + async |v| { + lookup_join_value(state, binding, v, tenant_id, database_id, trace_id) + .await + .map(Some) + }, + ) + .await?; } } Ok(outcome) diff --git a/nodedb/src/control/planner/mod.rs b/nodedb/src/control/planner/mod.rs index ac07b12b7..fb1e19a81 100644 --- a/nodedb/src/control/planner/mod.rs +++ b/nodedb/src/control/planner/mod.rs @@ -7,6 +7,7 @@ pub mod context; pub mod descriptor_set; pub mod implicit_edges; pub mod materialized_sum; +pub mod period_lock; pub(crate) mod plan_error_map; pub mod procedural; pub mod redaction_refusal; diff --git a/nodedb/src/control/planner/period_lock/gate.rs b/nodedb/src/control/planner/period_lock/gate.rs new file mode 100644 index 000000000..d38af5749 --- /dev/null +++ b/nodedb/src/control/planner/period_lock/gate.rs @@ -0,0 +1,92 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Which `DocumentOp` variants a period lock gates, and how a resolved entry +//! lands back on one. + +use nodedb_physical::physical_plan::{DocumentOp, ResolvedSumTarget}; + +/// The source collection a period lock would gate this op's write against, +/// or `None` for an op `check_period_lock` never runs against. Exhaustive so +/// a new `DocumentOp` variant must state which side it is on. +pub(super) fn period_lock_gated_collection(op: &DocumentOp) -> Option<&str> { + match op { + DocumentOp::PointPut { collection, .. } + | DocumentOp::PointInsert { collection, .. } + | DocumentOp::BatchInsert { collection, .. } + | DocumentOp::Upsert { collection, .. } + | DocumentOp::PointDelete { collection, .. } + | DocumentOp::PointUpdate { collection, .. } + | DocumentOp::BulkUpdate { collection, .. } + | DocumentOp::BulkDelete { collection, .. } => Some(collection.as_str()), + // `check_period_lock` runs only from `apply_point_put` / + // `apply_point_delete` / `execute_point_update` / + // `execute_bulk_update` / `execute_bulk_delete`. `Merge`, + // `UpdateFromJoin`, and `InsertSelect` are Control-Plane orchestrated + // over several round trips and resolve through their own + // orchestrator's per-row expansion (`resolve_period_lock_targets_for_bodies`), + // not this pass. + DocumentOp::PointGet { .. } + | DocumentOp::Scan { .. } + | DocumentOp::RangeScan { .. } + | DocumentOp::Register { .. } + | DocumentOp::IndexLookup { .. } + | DocumentOp::IndexedFetch { .. } + | DocumentOp::DropIndex { .. } + | DocumentOp::BackfillIndex { .. } + | DocumentOp::Truncate { .. } + | DocumentOp::EstimateCount { .. } + | DocumentOp::InsertSelect { .. } + | DocumentOp::UpdateFromJoin { .. } + | DocumentOp::Merge { .. } + | DocumentOp::MaterializeScan { .. } + | DocumentOp::ResolveWrite(_) + | DocumentOp::ResolvedWrite { .. } + | DocumentOp::ApplyBalanceDelta { .. } => None, + } +} + +/// Append `entry` to the op's `resolved_sum_targets` slot. Exhaustive for the +/// same reason [`period_lock_gated_collection`] is. Never called for +/// `BatchInsert`, `PointUpdate`, `BulkUpdate`, or `BulkDelete`, which the +/// caller handles directly. +pub(super) fn push_resolved(op: &mut DocumentOp, entry: ResolvedSumTarget) { + match op { + DocumentOp::PointPut { + resolved_sum_targets, + .. + } + | DocumentOp::PointInsert { + resolved_sum_targets, + .. + } + | DocumentOp::Upsert { + resolved_sum_targets, + .. + } + | DocumentOp::PointDelete { + resolved_sum_targets, + .. + } => resolved_sum_targets.push(entry), + DocumentOp::BatchInsert { .. } + | DocumentOp::PointGet { .. } + | DocumentOp::PointUpdate { .. } + | DocumentOp::Scan { .. } + | DocumentOp::RangeScan { .. } + | DocumentOp::Register { .. } + | DocumentOp::IndexLookup { .. } + | DocumentOp::IndexedFetch { .. } + | DocumentOp::DropIndex { .. } + | DocumentOp::BackfillIndex { .. } + | DocumentOp::Truncate { .. } + | DocumentOp::EstimateCount { .. } + | DocumentOp::InsertSelect { .. } + | DocumentOp::UpdateFromJoin { .. } + | DocumentOp::BulkUpdate { .. } + | DocumentOp::BulkDelete { .. } + | DocumentOp::Merge { .. } + | DocumentOp::MaterializeScan { .. } + | DocumentOp::ResolveWrite(_) + | DocumentOp::ResolvedWrite { .. } + | DocumentOp::ApplyBalanceDelta { .. } => {} + } +} diff --git a/nodedb/src/control/planner/period_lock/lookup.rs b/nodedb/src/control/planner/period_lock/lookup.rs new file mode 100644 index 000000000..958e716a1 --- /dev/null +++ b/nodedb/src/control/planner/period_lock/lookup.rs @@ -0,0 +1,55 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Shared lookups every period-lock resolution path in this module uses. + +use crate::control::server::surrogate_exchange::lookup_surrogate_routed; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; + +/// Request scope shared by every per-row period-lock resolution call within +/// one [`resolve_period_lock_targets`](super::resolve::resolve_period_lock_targets) +/// pass — bundled so the resolver functions stay within a sane argument +/// count. +pub(super) struct PeriodLockScope<'a> { + pub state: &'a SharedState, + pub tenant_id: TenantId, + pub database_id: DatabaseId, + pub trace_id: TraceId, +} + +/// Resolve one period value to its reference row's surrogate, `None` when no +/// reference row names it. +/// +/// `lookup_surrogate_routed`, never `assign_surrogate_routed`: a period value +/// that names no existing reference row is an unknown period, not a row to +/// mint identity for. +pub(super) async fn lookup_period_surrogate( + state: &SharedState, + ref_table: &str, + period_key: &str, + tenant_id: TenantId, + database_id: DatabaseId, + trace_id: TraceId, +) -> crate::Result> { + let vshard = VShardId::from_key(period_key.as_bytes()); + lookup_surrogate_routed( + state, + vshard, + database_id, + tenant_id, + ref_table, + period_key.as_bytes(), + trace_id, + ) + .await +} + +/// Strip the `"/"` prefix a planned collection name carries, yielding +/// the catalog name a collection lookup is keyed on. +pub(super) fn strip_db_prefix(database_id: DatabaseId, qualified: &str) -> &str { + if database_id == DatabaseId::DEFAULT { + return qualified; + } + let prefix = format!("{}/", database_id.as_u64()); + qualified.strip_prefix(prefix.as_str()).unwrap_or(qualified) +} diff --git a/nodedb/src/control/planner/period_lock/mod.rs b/nodedb/src/control/planner/period_lock/mod.rs new file mode 100644 index 000000000..4ebf0864a --- /dev/null +++ b/nodedb/src/control/planner/period_lock/mod.rs @@ -0,0 +1,16 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Control-Plane resolution of period-lock reference rows. See +//! [`resolve::resolve_period_lock_targets`] for the full contract. + +mod gate; +mod lookup; +mod point_update; +mod predicate; +mod resolve; +mod singular; + +pub use resolve::{ + resolve_period_lock_targets, resolve_period_lock_targets_for_bodies, + target_declares_period_lock, +}; diff --git a/nodedb/src/control/planner/period_lock/point_update.rs b/nodedb/src/control/planner/period_lock/point_update.rs new file mode 100644 index 000000000..8f641a50c --- /dev/null +++ b/nodedb/src/control/planner/period_lock/point_update.rs @@ -0,0 +1,72 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Period-value extraction for `PointUpdate`, whose write touches both a +//! pre-image and a post-image. + +use nodedb_physical::physical_plan::{ResolvedSumTarget, UpdateValue}; + +use super::lookup::{PeriodLockScope, lookup_period_surrogate}; +use crate::control::planner::materialized_sum::recon::recon_point_row; +use crate::control::planner::materialized_sum::resolve_one_target; +use crate::control::security::catalog::PeriodLockDef; + +/// Resolve the period value(s) a `PointUpdate` addresses: the stored row's +/// current period, and the period its assignments would produce, deduped when +/// they are the same value. +/// +/// A key matching no stored row resolves nothing — the update rewrites no row +/// and `execute_point_update` reports zero rows affected without ever reaching +/// `check_period_lock`. +pub(super) async fn resolve_update_period_values( + scope: &PeriodLockScope<'_>, + collection: &str, + document_id: &str, + surrogate: nodedb_types::Surrogate, + updates: &[(String, UpdateValue)], + def: &PeriodLockDef, +) -> crate::Result> { + let read = recon_point_row( + scope.state, + scope.tenant_id, + scope.database_id, + collection, + document_id, + surrogate, + ) + .await?; + let Some(old_row) = read.rows else { + return Ok(Vec::new()); + }; + let new_row = crate::query::apply_update_assignments(&old_row, updates)?; + + // `resolve_one_target` dedupes on `(target, value)` itself, so the old + // and new images resolve through the same call whether or not they carry + // the same period value. + let mut resolved: Vec = Vec::new(); + for row in [&old_row, &new_row] { + let Some(value) = row.get(def.period_column.as_str()).and_then(|v| v.as_str()) else { + continue; + }; + // No reference row names this period — `resolve_one_target` leaves + // it unresolved so `check_period_lock` reports the unknown-period + // refusal for it. + resolve_one_target( + &mut resolved, + &def.ref_table, + value.to_string(), + async |key| { + lookup_period_surrogate( + scope.state, + &def.ref_table, + key, + scope.tenant_id, + scope.database_id, + scope.trace_id, + ) + .await + }, + ) + .await?; + } + Ok(resolved) +} diff --git a/nodedb/src/control/planner/period_lock/predicate.rs b/nodedb/src/control/planner/period_lock/predicate.rs new file mode 100644 index 000000000..b62bd8d7a --- /dev/null +++ b/nodedb/src/control/planner/period_lock/predicate.rs @@ -0,0 +1,87 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Period-value extraction for the predicate-driven shapes `BulkUpdate` / +//! `BulkDelete`, which name their rows by predicate rather than by body or +//! key. + +use nodedb_physical::physical_plan::{ResolvedSumTarget, UpdateValue}; + +use super::lookup::{PeriodLockScope, lookup_period_surrogate}; +use crate::control::planner::materialized_sum::recon::recon_scan_rows; +use crate::control::planner::materialized_sum::resolve_one_target; +use crate::control::security::catalog::PeriodLockDef; + +/// What a predicate-driven statement does to each row it matches — decides +/// whether the resolution also needs the row's POST-image. +pub(super) enum PeriodLockEffect { + /// The rows are rewritten by the statement's assignments. + Assign, + /// The rows are removed; there is no post-image. + Remove, +} + +/// Resolve the period value(s) a `BulkUpdate` / `BulkDelete` addresses, from a +/// reconnaissance scan of the same predicate — mirroring +/// `materialized_sum::predicate::resolve_predicate_sum_targets`. +/// +/// A `BulkUpdate` resolves both images of every matched row (the period it +/// currently holds, and the period its assignments would produce), exactly +/// like `resolve_update_period_values`. A `BulkDelete` resolves only the +/// pre-image, its one and only image. +pub(super) async fn resolve_predicate_period_values( + scope: &PeriodLockScope<'_>, + collection: &str, + filters: Vec, + updates: &[(String, UpdateValue)], + effect: PeriodLockEffect, + def: &PeriodLockDef, +) -> crate::Result> { + let read = recon_scan_rows( + scope.state, + scope.tenant_id, + scope.database_id, + collection, + filters, + ) + .await?; + + // `resolve_one_target` dedupes on `(target, value)` itself, so every + // row's pre- and post-image resolves through the same call regardless of + // how many rows or images share a period value. + let mut resolved: Vec = Vec::new(); + for row in &read.rows { + resolve_period_value(scope, def, row, &mut resolved).await?; + if matches!(effect, PeriodLockEffect::Assign) { + let new_row = crate::query::apply_update_assignments(row, updates)?; + resolve_period_value(scope, def, &new_row, &mut resolved).await?; + } + } + Ok(resolved) +} + +/// Resolve one row's period value into `resolved`, when it carries the +/// configured period column at all. No reference row naming the period +/// leaves it unresolved so `check_period_lock` reports the unknown-period +/// refusal for it. +async fn resolve_period_value( + scope: &PeriodLockScope<'_>, + def: &PeriodLockDef, + row: &serde_json::Value, + resolved: &mut Vec, +) -> crate::Result<()> { + let Some(value) = row.get(def.period_column.as_str()).and_then(|v| v.as_str()) else { + return Ok(()); + }; + resolve_one_target(resolved, &def.ref_table, value.to_string(), async |key| { + lookup_period_surrogate( + scope.state, + &def.ref_table, + key, + scope.tenant_id, + scope.database_id, + scope.trace_id, + ) + .await + }) + .await +} diff --git a/nodedb/src/control/planner/period_lock/resolve.rs b/nodedb/src/control/planner/period_lock/resolve.rs new file mode 100644 index 000000000..2a14c1722 --- /dev/null +++ b/nodedb/src/control/planner/period_lock/resolve.rs @@ -0,0 +1,258 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Entry points: the per-plan pass wired into `plan_admission`, and the +//! body-driven seam the `MERGE` / `INSERT ... SELECT` / `UPDATE ... FROM` +//! orchestrators call directly. +//! +//! A period lock names its reference row by a VALUE — the write's period +//! column, e.g. `fiscal_period = "2024-Q1"` — read against the reference +//! table's declared primary key (`config.ref_pk`). Turning that value into +//! the reference row's storage key needs the pk → surrogate map, which lives +//! in the catalog redb — Control-Plane state the Data Plane never opens. So +//! the resolution happens here, at plan time, and travels on the plan in the +//! same `resolved_sum_targets` slot the materialized-sum resolution uses: +//! one `(target collection, value) -> surrogate` slot, one resolver on the +//! Data Plane ([`resolved_sum_surrogate`](nodedb_physical::physical_plan::resolved_sum_surrogate)), +//! for every cross-collection identity a write needs. +//! +//! Only the DML shapes the Data Plane's `check_period_lock` actually gates — +//! `PointPut`, `PointInsert`, `BatchInsert`, `Upsert`, `PointDelete`, +//! `PointUpdate`, `BulkUpdate`, `BulkDelete` — are resolved by +//! [`resolve_period_lock_targets`]. +//! +//! `PointUpdate` checks BOTH images: the stored row it rewrites and the +//! post-image its assignments produce. A closed period must reject an edit +//! to a row it already holds, and must reject an edit that assigns the +//! period column into it. Both period values are resolved here, so +//! `check_period_lock` finds an entry for whichever image it reads. +//! `BulkUpdate` / `BulkDelete` name their rows by PREDICATE rather than by +//! body or key, so their resolution reads a Control-Plane reconnaissance +//! scan of that same predicate — mirroring the materialized-sum predicate +//! resolution in `materialized_sum::predicate`. A row whose period value +//! drifts between this scan and the Data-Plane apply resolves to no entry +//! and the write is refused as an unknown period — fail-closed, never a +//! silent admit. +//! +//! `MERGE`, `INSERT ... SELECT`, and `UPDATE ... FROM` are Control-Plane +//! orchestrated over several round trips and never reach the per-plan pass — +//! each resolves through [`resolve_period_lock_targets_for_bodies`] in its +//! own orchestrator instead. +//! +//! # Plane discipline +//! +//! Runs on the coordinator's Control Plane (Tokio). The routed lookup and +//! the reconnaissance reads cross the SPSC bridge exactly as `SELECT` does — +//! no storage I/O and no io_uring here. + +use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan, ResolvedSumTarget}; +use nodedb_physical::physical_task::PhysicalTask; + +use super::gate::{period_lock_gated_collection, push_resolved}; +use super::lookup::{PeriodLockScope, lookup_period_surrogate, strip_db_prefix}; +use super::point_update::resolve_update_period_values; +use super::predicate::{PeriodLockEffect, resolve_predicate_period_values}; +use super::singular::{resolve_batch_period_values, singular_period_value}; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId, TraceId}; + +/// Resolve the period-lock reference row for every write in `tasks` that a +/// period lock gates, appending the entry to the op's `resolved_sum_targets` +/// slot. +/// +/// A period value the resolution cannot bind (no reference row named by that +/// value) adds no entry: `check_period_lock` treats an absent entry as an +/// unknown period and refuses the write, exactly as it treats a genuinely +/// closed one. +pub async fn resolve_period_lock_targets( + state: &SharedState, + tasks: &mut [PhysicalTask], + tenant_id: TenantId, + database_id: DatabaseId, + trace_id: TraceId, +) -> crate::Result<()> { + let catalog = state.credentials.catalog(); + let scope = PeriodLockScope { + state, + tenant_id, + database_id, + trace_id, + }; + for task in tasks.iter_mut() { + let PhysicalPlan::Document(op) = &mut task.plan else { + continue; + }; + let Some(collection) = period_lock_gated_collection(op) else { + continue; + }; + let collection = collection.to_string(); + let source = strip_db_prefix(database_id, &collection).to_string(); + let Some(def) = catalog + .get_collection(database_id, tenant_id.as_u64(), &source)? + .and_then(|coll| coll.period_lock) + else { + continue; + }; + + // `BatchInsert` carries a whole page of documents, each with its own + // period value, rather than the single value every other gated shape + // carries — resolved and pushed here, distinctly from the singular + // path below. + if let DocumentOp::BatchInsert { + documents, + resolved_sum_targets, + .. + } = op + { + let bodies: Vec<&[u8]> = documents.iter().map(|(_, v)| v.as_slice()).collect(); + let resolved = + resolve_batch_period_values(state, &bodies, &def, tenant_id, database_id, trace_id) + .await?; + resolved_sum_targets.extend(resolved); + continue; + } + + // `PointUpdate` carries field assignments rather than a whole row, and + // it may rewrite the period column itself — both images need their own + // resolution, distinctly from the single-value path below. + if let DocumentOp::PointUpdate { + document_id, + surrogate, + updates, + resolved_sum_targets, + .. + } = op + { + let resolved = resolve_update_period_values( + &scope, + &collection, + document_id, + *surrogate, + updates, + &def, + ) + .await?; + resolved_sum_targets.extend(resolved); + continue; + } + + // `BulkUpdate` / `BulkDelete` name their rows by PREDICATE: the + // Control Plane holds no body and no single key, so the resolution + // reads a reconnaissance scan of the same predicate instead. + if let DocumentOp::BulkUpdate { + filters, + updates, + resolved_sum_targets, + .. + } = op + { + let resolved = resolve_predicate_period_values( + &scope, + &collection, + filters.clone(), + updates, + PeriodLockEffect::Assign, + &def, + ) + .await?; + resolved_sum_targets.extend(resolved); + continue; + } + if let DocumentOp::BulkDelete { + filters, + resolved_sum_targets, + .. + } = op + { + let resolved = resolve_predicate_period_values( + &scope, + &collection, + filters.clone(), + &[], + PeriodLockEffect::Remove, + &def, + ) + .await?; + resolved_sum_targets.extend(resolved); + continue; + } + + let Some(period_key) = + singular_period_value(state, op, &collection, &def, tenant_id, database_id).await? + else { + continue; + }; + + let Some(surrogate) = lookup_period_surrogate( + state, + &def.ref_table, + &period_key, + tenant_id, + database_id, + trace_id, + ) + .await? + else { + // No reference row names this period — leave the slot unresolved + // so `check_period_lock` reports the unknown-period refusal. + continue; + }; + + push_resolved( + op, + ResolvedSumTarget::new(&def.ref_table, period_key, surrogate), + ); + } + Ok(()) +} + +/// Resolve the period-lock targets a page of row BODIES addresses, for a +/// caller that holds the bodies itself rather than a plan. +/// +/// This is the seam `MERGE`, `INSERT ... SELECT`, and `UPDATE ... FROM` +/// orchestrators use: each resolves its own rows on the Control Plane and +/// re-issues concrete work through `dispatch_local`, which never passes +/// through [`resolve_period_lock_targets`]. Mirrors +/// [`resolve_sum_targets_for_bodies`](crate::control::planner::materialized_sum::resolve_sum_targets_for_bodies) +/// — call both and merge the results into one `resolved_sum_targets` vec, +/// since a page can owe both a materialized-sum target and a period lock. +/// +/// Returns an empty vec — and issues no lookup at all — when the collection +/// declares no period lock. +pub async fn resolve_period_lock_targets_for_bodies( + state: &SharedState, + bodies: &[&[u8]], + target_collection: &str, + tenant_id: TenantId, + database_id: DatabaseId, + trace_id: TraceId, +) -> crate::Result> { + let catalog = state.credentials.catalog(); + let source = strip_db_prefix(database_id, target_collection); + let Some(def) = catalog + .get_collection(database_id, tenant_id.as_u64(), source)? + .and_then(|coll| coll.period_lock) + else { + return Ok(Vec::new()); + }; + resolve_batch_period_values(state, bodies, &def, tenant_id, database_id, trace_id).await +} + +/// Whether `target_collection` declares a period lock. +/// +/// The gate an orchestrator checks FIRST, alongside +/// [`source_drives_bindings`](crate::control::planner::materialized_sum::source_drives_bindings): +/// a target with neither a materialized-sum binding nor a period lock skips +/// the RESOLVE round trip its statement would otherwise pay for nothing. +pub fn target_declares_period_lock( + state: &SharedState, + target_collection: &str, + tenant_id: TenantId, + database_id: DatabaseId, +) -> crate::Result { + let catalog = state.credentials.catalog(); + let source = strip_db_prefix(database_id, target_collection); + Ok(catalog + .get_collection(database_id, tenant_id.as_u64(), source)? + .and_then(|coll| coll.period_lock) + .is_some()) +} diff --git a/nodedb/src/control/planner/period_lock/singular.rs b/nodedb/src/control/planner/period_lock/singular.rs new file mode 100644 index 000000000..a56207f4e --- /dev/null +++ b/nodedb/src/control/planner/period_lock/singular.rs @@ -0,0 +1,108 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Period-value extraction for the single-row and batch-body gated shapes: +//! `PointPut` / `PointInsert` / `Upsert` / `PointDelete` / `BatchInsert`. + +use nodedb_physical::physical_plan::{DocumentOp, ResolvedSumTarget}; + +use super::lookup::lookup_period_surrogate; +use crate::control::planner::materialized_sum::recon::recon_point_row; +use crate::control::planner::materialized_sum::{join_value_from_body, resolve_one_target}; +use crate::control::security::catalog::PeriodLockDef; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId, TraceId}; + +/// The period value a single-row gated write carries, or `None` when the +/// period column is absent — the same "not gated" outcome `check_period_lock` +/// reaches for a row with no period column. Never called for `BatchInsert`, +/// which resolves through [`resolve_batch_period_values`] instead. +/// +/// A body-carrying write (`PointPut`/`PointInsert`/`Upsert`) reads the value +/// off its submitted body. `PointDelete` carries no body, so its value comes +/// off the row it is about to remove. +pub(super) async fn singular_period_value( + state: &SharedState, + op: &DocumentOp, + collection: &str, + def: &PeriodLockDef, + tenant_id: TenantId, + database_id: DatabaseId, +) -> crate::Result> { + match op { + DocumentOp::PointPut { value, .. } + | DocumentOp::PointInsert { value, .. } + | DocumentOp::Upsert { value, .. } => Ok(join_value_from_body(value, &def.period_column)), + DocumentOp::PointDelete { + document_id, + surrogate, + .. + } => { + let read = recon_point_row( + state, + tenant_id, + database_id, + collection, + document_id, + *surrogate, + ) + .await?; + Ok(read + .rows + .as_ref() + .and_then(|row| row.get(def.period_column.as_str())) + .and_then(|v| v.as_str()) + .map(str::to_string)) + } + // `period_lock_gated_collection` only routes these eight variants + // here, and the caller handles `BatchInsert`, `PointUpdate`, + // `BulkUpdate`, and `BulkDelete` before reaching this function — + // every other variant is unreachable. + DocumentOp::BatchInsert { .. } + | DocumentOp::PointGet { .. } + | DocumentOp::PointUpdate { .. } + | DocumentOp::Scan { .. } + | DocumentOp::RangeScan { .. } + | DocumentOp::Register { .. } + | DocumentOp::IndexLookup { .. } + | DocumentOp::IndexedFetch { .. } + | DocumentOp::DropIndex { .. } + | DocumentOp::BackfillIndex { .. } + | DocumentOp::Truncate { .. } + | DocumentOp::EstimateCount { .. } + | DocumentOp::InsertSelect { .. } + | DocumentOp::UpdateFromJoin { .. } + | DocumentOp::BulkUpdate { .. } + | DocumentOp::BulkDelete { .. } + | DocumentOp::Merge { .. } + | DocumentOp::MaterializeScan { .. } + | DocumentOp::ResolveWrite(_) + | DocumentOp::ResolvedWrite { .. } + | DocumentOp::ApplyBalanceDelta { .. } => Ok(None), + } +} + +/// Resolve every DISTINCT period value across a page of row bodies to its +/// reference row's surrogate, one entry per value — mirroring how the +/// materialized-sum resolution dedupes a page onto one entry per distinct +/// join value. +pub(super) async fn resolve_batch_period_values( + state: &SharedState, + bodies: &[&[u8]], + def: &PeriodLockDef, + tenant_id: TenantId, + database_id: DatabaseId, + trace_id: TraceId, +) -> crate::Result> { + let mut resolved: Vec = Vec::new(); + for body in bodies { + let Some(period_key) = join_value_from_body(body, &def.period_column) else { + continue; + }; + resolve_one_target(&mut resolved, &def.ref_table, period_key, async |key| { + lookup_period_surrogate(state, &def.ref_table, key, tenant_id, database_id, trace_id) + .await + }) + .await?; + } + Ok(resolved) +} diff --git a/nodedb/src/control/server/dispatch_utils/write_abort.rs b/nodedb/src/control/server/dispatch_utils/write_abort.rs index 484953ed1..373523bef 100644 --- a/nodedb/src/control/server/dispatch_utils/write_abort.rs +++ b/nodedb/src/control/server/dispatch_utils/write_abort.rs @@ -119,6 +119,7 @@ pub(crate) fn write_definitely_not_applied(code: &ErrorCode) -> bool { | ErrorCode::AppendOnlyViolation { .. } | ErrorCode::BalanceViolation { .. } | ErrorCode::PeriodLocked { .. } + | ErrorCode::PeriodLockMisconfigured { .. } | ErrorCode::RetentionViolation { .. } | ErrorCode::LegalHoldActive { .. } | ErrorCode::StateTransitionViolation { .. } diff --git a/nodedb/src/control/server/native/dispatch/direct_ops.rs b/nodedb/src/control/server/native/dispatch/direct_ops.rs index 4c6b0e105..b80bc4a15 100644 --- a/nodedb/src/control/server/native/dispatch/direct_ops.rs +++ b/nodedb/src/control/server/native/dispatch/direct_ops.rs @@ -301,6 +301,20 @@ pub(crate) async fn handle_direct_op( return error_to_native(seq, &e); } + // Period-lock reference rows, resolved into the same plan slot as the + // materialized-sum targets just above. + if let Err(e) = crate::control::planner::period_lock::resolve_period_lock_targets( + ctx.state, + &mut tasks, + tenant_id, + ctx.database_id(), + TraceId::ZERO, + ) + .await + { + return error_to_native(seq, &e); + } + // The expanded set is the dispatch authorization boundary. let authorized_tasks = match crate::control::server::shared::authorization::authorize_task_set( diff --git a/nodedb/src/control/server/pgwire/handler/routing/execute.rs b/nodedb/src/control/server/pgwire/handler/routing/execute.rs index 26538d414..c6139687f 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/execute.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/execute.rs @@ -103,6 +103,19 @@ impl NodeDbPgHandler { ) .map_err(StatementSetupError::from)?; + // Period-lock reference rows are resolved here for the same + // reason as the materialized-sum targets just above, into the + // same plan slot. + crate::control::planner::period_lock::resolve_period_lock_targets( + &self.state, + &mut tasks, + tenant_id, + edge_database_id, + crate::types::TraceId::ZERO, + ) + .await + .map_err(StatementSetupError::from)?; + // The final task set must be authorized before any clone // interception, orchestration, staging, or dispatch path can // observe it. Descriptor admission follows this check so an diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs index 8370b4aee..5859703b5 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs @@ -224,6 +224,19 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a ddl_err(sqlstate, message) })?; + crate::control::planner::period_lock::resolve_period_lock_targets( + state, + &mut tasks, + tenant_id, + database_id, + TraceId::ZERO, + ) + .await + .map_err(|error| { + let (_, sqlstate, message) = error_to_sqlstate(&error); + ddl_err(sqlstate, message) + })?; + let authorized_tasks = authorize_final_task_set(state, identity, &tasks)?; // Admission follows final authorization so an implicit-edge target denied // by policy does not consume a descriptor lease. The scope remains live diff --git a/nodedb/src/control/server/shared/ddl/sqlstate.rs b/nodedb/src/control/server/shared/ddl/sqlstate.rs index 9230a4d00..4a3a4c0da 100644 --- a/nodedb/src/control/server/shared/ddl/sqlstate.rs +++ b/nodedb/src/control/server/shared/ddl/sqlstate.rs @@ -96,6 +96,19 @@ pub fn error_code_to_sqlstate(code: &ErrorCode) -> (&'static str, &'static str, sqlstate::PERIOD_LOCKED, format!("period locked: writes rejected on {collection}"), ), + ErrorCode::PeriodLockMisconfigured { + collection, + ref_table, + status_column, + row_identity, + } => ( + "ERROR", + sqlstate::PERIOD_LOCK_MISCONFIGURED, + format!( + "period lock on {collection} misconfigured: reference table \ + '{ref_table}' row '{row_identity}' has no column '{status_column}'" + ), + ), ErrorCode::RetentionViolation { collection } => ( "ERROR", sqlstate::RETENTION_VIOLATION, diff --git a/nodedb/src/control/server/shared/plan_admission.rs b/nodedb/src/control/server/shared/plan_admission.rs index 98f3a2c22..a64306050 100644 --- a/nodedb/src/control/server/shared/plan_admission.rs +++ b/nodedb/src/control/server/shared/plan_admission.rs @@ -146,6 +146,18 @@ async fn plan_authorize_and_admit_once( database_id, )?; + // Resolves each write's period-lock reference row into the same slot the + // materialized-sum resolution above just populated — see + // `period_lock::resolve_period_lock_targets`. + crate::control::planner::period_lock::resolve_period_lock_targets( + state, + &mut tasks, + tenant_id, + database_id, + trace_id, + ) + .await?; + // Deliberate gate: proves the final task set is authorizable before a // descriptor lease is acquired. The caller re-derives the capability per // task through the clone-check gate, immediately before each dispatch. diff --git a/nodedb/src/control/update_from_join_orchestrator/orchestrator.rs b/nodedb/src/control/update_from_join_orchestrator/orchestrator.rs index 038554cdf..ab50fba16 100644 --- a/nodedb/src/control/update_from_join_orchestrator/orchestrator.rs +++ b/nodedb/src/control/update_from_join_orchestrator/orchestrator.rs @@ -98,8 +98,9 @@ pub(crate) async fn run_update_from_join( state: &SharedState, args: UpdateFromJoinArgs<'_>, ) -> crate::Result { - // Checked once: a target driving no materialized-sum binding skips the - // RESOLVE round trip and retry loop entirely. + // Checked once: a target driving no materialized-sum binding AND + // declaring no period lock skips the RESOLVE round trip and retry loop + // entirely. let drives_bindings = source_drives_bindings( state, args.target_collection, @@ -107,6 +108,13 @@ pub(crate) async fn run_update_from_join( args.database_id, )? .is_some(); + let has_period_lock = crate::control::planner::period_lock::target_declares_period_lock( + state, + args.target_collection, + args.tenant_id, + args.database_id, + )?; + let needs_resolve = drives_bindings || has_period_lock; let mut attempt: u32 = 0; loop { @@ -121,13 +129,13 @@ pub(crate) async fn run_update_from_join( ) .await?; - let resolved_sum_targets = if drives_bindings { + let resolved_sum_targets = if needs_resolve { match resolve_matched_sum_targets(state, &args, source_rows.clone()).await? { Some(resolved) => resolved, // RESOLVE pass failed on the Data Plane; its response is the answer. None => { return Err(crate::Error::Dispatch { - detail: "UPDATE ... FROM materialized-sum resolve pass failed".into(), + detail: "UPDATE ... FROM target-row resolve pass failed".into(), }); } } @@ -194,9 +202,10 @@ pub(crate) async fn run_update_from_join( } } -/// Resolve the materialized-sum targets this statement's matched rows need. -/// Resolves both images of every matched row (a join-column rewrite debits -/// one target and credits another). `None` means RESOLVE failed. +/// Resolve the materialized-sum AND period-lock targets this statement's +/// matched rows need. Resolves both images of every matched row (a +/// join-column rewrite debits one target and credits another; a period-lock +/// check may read either image). `None` means RESOLVE failed. async fn resolve_matched_sum_targets( state: &SharedState, args: &UpdateFromJoinArgs<'_>, @@ -244,7 +253,7 @@ async fn resolve_matched_sum_targets( .iter() .flat_map(|(_, _, body, old_body)| [body.as_slice(), old_body.as_slice()]) .collect(); - resolve_sum_targets_for_bodies( + let mut resolved = resolve_sum_targets_for_bodies( state, &bodies, args.target_collection, @@ -252,6 +261,17 @@ async fn resolve_matched_sum_targets( args.database_id, crate::types::TraceId::ZERO, ) - .await - .map(Some) + .await?; + resolved.extend( + crate::control::planner::period_lock::resolve_period_lock_targets_for_bodies( + state, + &bodies, + args.target_collection, + args.tenant_id, + args.database_id, + crate::types::TraceId::ZERO, + ) + .await?, + ); + Ok(Some(resolved)) } diff --git a/nodedb/src/data/executor/core_loop/doc_config_seed.rs b/nodedb/src/data/executor/core_loop/doc_config_seed.rs index 9d1f1b5ec..b35565b2d 100644 --- a/nodedb/src/data/executor/core_loop/doc_config_seed.rs +++ b/nodedb/src/data/executor/core_loop/doc_config_seed.rs @@ -112,7 +112,7 @@ mod tests { DB, TID, COLL, - &surrogate_to_doc_id(Surrogate::new(surrogate)), + &nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)), ) .unwrap() .expect("document should have been replayed"); @@ -153,7 +153,7 @@ mod tests { DB, TID, COLL, - &surrogate_to_doc_id(Surrogate::new(surrogate)), + &nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)), ) .unwrap() .expect("document should have been replayed"); diff --git a/nodedb/src/data/executor/core_loop/tick.rs b/nodedb/src/data/executor/core_loop/tick.rs index 5ad0b6ca4..7bf187d88 100644 --- a/nodedb/src/data/executor/core_loop/tick.rs +++ b/nodedb/src/data/executor/core_loop/tick.rs @@ -256,7 +256,15 @@ mod tests { fn watermark_in_response() { let (mut core, mut req_tx, mut resp_rx, _dir) = make_core(); core.advance_watermark(Lsn::new(99)); - core.sparse.put(0, 1, "x", "y", b"data").unwrap(); + core.sparse + .put( + 0, + 1, + "x", + &nodedb_types::StorageKey::for_surrogate(nodedb_types::Surrogate::new(999)), + b"data", + ) + .unwrap(); req_tx .try_push(BridgeRequest { inner: make_request(PhysicalPlan::Document(DocumentOp::PointGet { @@ -355,7 +363,12 @@ mod tests { // "00000000". let stored = core .sparse - .get(0, 1, "orders", "00000000") + .get( + 0, + 1, + "orders", + &nodedb_types::StorageKey::for_surrogate(nodedb_types::Surrogate::ZERO), + ) .unwrap() .unwrap(); assert!(nodedb_query::msgpack_scan::map_header(&stored, 0).is_some()); diff --git a/nodedb/src/data/executor/core_loop/vector_index_rebuild.rs b/nodedb/src/data/executor/core_loop/vector_index_rebuild.rs index 3940870cf..0ba258ea5 100644 --- a/nodedb/src/data/executor/core_loop/vector_index_rebuild.rs +++ b/nodedb/src/data/executor/core_loop/vector_index_rebuild.rs @@ -79,16 +79,11 @@ impl CoreLoop { &collection, usize::MAX, |doc_id, value| { - if let Some(surrogate) = - crate::engine::document::store::doc_id_to_surrogate(doc_id) - { - let normalized = - crate::data::executor::scan_normalize::sparse_body_to_msgpack( - value, - body_format.as_format_ref(), - ); - docs.push((surrogate, normalized.into_owned())); - } + let normalized = crate::data::executor::scan_normalize::sparse_body_to_msgpack( + value, + body_format.as_format_ref(), + ); + docs.push((doc_id.surrogate(), normalized.into_owned())); Ok(()) }, ); diff --git a/nodedb/src/data/executor/enforcement/chain_guard.rs b/nodedb/src/data/executor/enforcement/chain_guard.rs index 13d62b502..6c13494bf 100644 --- a/nodedb/src/data/executor/enforcement/chain_guard.rs +++ b/nodedb/src/data/executor/enforcement/chain_guard.rs @@ -166,7 +166,7 @@ pub(in crate::data::executor) fn abort_after_apply( database_id: u64, tid: u64, collection: &str, - row_key: &str, + row_key: &crate::engine::document::store::StorageKey, ) { guard.restore(core); core.doc_cache diff --git a/nodedb/src/data/executor/enforcement/hash_chain.rs b/nodedb/src/data/executor/enforcement/hash_chain.rs index cb3780ef4..c6889a40a 100644 --- a/nodedb/src/data/executor/enforcement/hash_chain.rs +++ b/nodedb/src/data/executor/enforcement/hash_chain.rs @@ -241,7 +241,7 @@ mod tests { fn the_chain_head_survives_a_restart() { use crate::bridge::envelope::Status; use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; - use crate::engine::document::store::{CollectionConfig, surrogate_to_doc_id}; + use crate::engine::document::store::CollectionConfig; use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; use nodedb_types::{DatabaseId, Surrogate, TenantId}; @@ -333,7 +333,7 @@ mod tests { let stored_hashes: Vec = rows .iter() .map(|(surrogate, _, _)| { - let row_key = surrogate_to_doc_id(Surrogate::new(*surrogate)); + let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(*surrogate)); let stored = core .sparse .get(db.as_u64(), TID, COLL, &row_key) diff --git a/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs b/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs index 681c7ae92..b2ca030b4 100644 --- a/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs +++ b/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs @@ -187,11 +187,14 @@ impl CoreLoop { // cache entries those writes populated. Left behind, they // serve balances that no longer exist in storage. for write in &writes { + let key = crate::engine::document::store::StorageKey::for_surrogate( + write.surrogate, + ); self.doc_cache.invalidate( ctx.database_id, ctx.tid, &write.collection, - &write.document_id, + &key, ); } return Err(e); @@ -447,7 +450,7 @@ mod tests { row.insert("balance".to_string(), Value::String("100".into())); let tuple = strict_format::value_to_binary_tuple(&Value::Object(row), &schema, TARGET) .expect("encode seed tuple"); - let target_key = surrogate_to_doc_id(TARGET_SURROGATE); + let target_key = nodedb_types::StorageKey::for_surrogate(TARGET_SURROGATE); core.sparse .put(DB, TID, TARGET, &target_key, &tuple) .expect("seed target row"); @@ -502,7 +505,7 @@ mod tests { let seed = serde_json::json!({"id": ACCOUNT, "owner": "alice", "balance": "100"}); let body = doc_format::encode_to_msgpack(&seed); - let target_key = surrogate_to_doc_id(TARGET_SURROGATE); + let target_key = nodedb_types::StorageKey::for_surrogate(TARGET_SURROGATE); core.sparse .put(DB, TID, TARGET, &target_key, &body) .expect("seed target row"); @@ -534,7 +537,7 @@ mod tests { ); register_source(&mut core); - let target_key = surrogate_to_doc_id(TARGET_SURROGATE); + let target_key = nodedb_types::StorageKey::for_surrogate(TARGET_SURROGATE); let seed = serde_json::json!({"id": ACCOUNT, "balance": "100"}); core.sparse .put( @@ -668,7 +671,7 @@ mod tests { // Seeded so that an accidental apply would be VISIBLE as a moved total // rather than failing on an absent row and looking like a refusal. - let target_key = surrogate_to_doc_id(TARGET_SURROGATE); + let target_key = nodedb_types::StorageKey::for_surrogate(TARGET_SURROGATE); let seed = serde_json::json!({"id": ACCOUNT, "balance": "100"}); core.sparse .put( @@ -798,7 +801,7 @@ mod tests { DB, TID, collection, - &surrogate_to_doc_id(surrogate), + &nodedb_types::StorageKey::for_surrogate(surrogate), &doc_format::encode_to_msgpack(&row), ) .expect("seed target row"); @@ -811,7 +814,7 @@ mod tests { DB, TID, SOURCE, - &surrogate_to_doc_id(surrogate), + &nodedb_types::StorageKey::for_surrogate(surrogate), &doc_format::encode_to_msgpack(&row), ) .expect("seed source row"); @@ -824,7 +827,12 @@ mod tests { fn balance_in(core: &CoreLoop, collection: &str, surrogate: Surrogate) -> String { let stored = core .sparse - .get(DB, TID, collection, &surrogate_to_doc_id(surrogate)) + .get( + DB, + TID, + collection, + &nodedb_types::StorageKey::for_surrogate(surrogate), + ) .expect("read target row") .expect("target row must still exist"); doc_format::decode_document(&stored) @@ -957,7 +965,7 @@ mod tests { DB, TID, SOURCE, - &surrogate_to_doc_id(Surrogate(1)), + &nodedb_types::StorageKey::for_surrogate(Surrogate(1)), &doc_format::encode_to_msgpack(&entry), ) .expect("seed written row"); @@ -1103,7 +1111,12 @@ mod tests { assert_eq!(balance_of(&core, SURROGATE_B), "50"); assert!( core.sparse - .get(DB, TID, SOURCE, &surrogate_to_doc_id(Surrogate(1))) + .get( + DB, + TID, + SOURCE, + &nodedb_types::StorageKey::for_surrogate(Surrogate(1)) + ) .expect("read source row") .is_some(), "no source row may be removed on a refused statement" @@ -1182,7 +1195,12 @@ mod tests { assert_eq!(balance_in(&core, REMOTE_TARGET, SURROGATE_B), "50"); assert!( core.sparse - .get(DB, TID, SOURCE, &surrogate_to_doc_id(Surrogate(1))) + .get( + DB, + TID, + SOURCE, + &nodedb_types::StorageKey::for_surrogate(Surrogate(1)) + ) .expect("read source row") .is_none(), "the statement itself must have run: the matched source rows are gone" @@ -1558,7 +1576,12 @@ mod tests { ); assert!( core.sparse - .get(DB, TID, SOURCE, &surrogate_to_doc_id(Surrogate(1))) + .get( + DB, + TID, + SOURCE, + &nodedb_types::StorageKey::for_surrogate(Surrogate(1)) + ) .expect("read source row") .is_some(), "the source row must still have been written" diff --git a/nodedb/src/data/executor/enforcement/materialized_sum/divergence.rs b/nodedb/src/data/executor/enforcement/materialized_sum/divergence.rs index 96e589f57..d60890647 100644 --- a/nodedb/src/data/executor/enforcement/materialized_sum/divergence.rs +++ b/nodedb/src/data/executor/enforcement/materialized_sum/divergence.rs @@ -152,9 +152,15 @@ impl CoreLoop { } let mut rows: Vec = Vec::with_capacity(doc_ids.len()); for doc_id in doc_ids { + // `doc_id` is a bare string from a raw-table scan; a shape that + // fails to parse as a storage key contributes no row, same as a + // `get` miss right below. + let Some(key) = nodedb_types::StorageKey::parse(doc_id) else { + continue; + }; let Ok(Some(bytes)) = self.sparse - .get(check.database_id, check.tid, check.collection, doc_id) + .get(check.database_id, check.tid, check.collection, &key) else { continue; }; diff --git a/nodedb/src/data/executor/enforcement/materialized_sum/rmw.rs b/nodedb/src/data/executor/enforcement/materialized_sum/rmw.rs index 41a333903..cdb8f03ea 100644 --- a/nodedb/src/data/executor/enforcement/materialized_sum/rmw.rs +++ b/nodedb/src/data/executor/enforcement/materialized_sum/rmw.rs @@ -116,7 +116,8 @@ impl CoreLoop { }); } - let Some(old_bytes) = self.read_balance_row(txn, params, &document_id)? else { + let Some(old_bytes) = self.read_balance_row(txn, params, &document_id, params.surrogate)? + else { return Err(params.target_not_found()); }; @@ -169,6 +170,7 @@ impl CoreLoop { user_roles: &[], enforce: false, wal_lsn: params.wal_lsn, + resolved_targets: &[], }, ); let outcome = match put { @@ -181,7 +183,7 @@ impl CoreLoop { params.database_id, params.tid, params.target_collection, - &document_id, + &nodedb_types::StorageKey::for_surrogate(params.surrogate), ); return Err(e); } @@ -207,6 +209,7 @@ impl CoreLoop { txn: &WriteTransaction, params: &BalanceRmw<'_>, document_id: &str, + surrogate: Surrogate, ) -> crate::Result>> { if self.is_bitemporal(params.database_id, params.tid, params.target_collection) { self.sparse.versioned_get_current( @@ -216,12 +219,13 @@ impl CoreLoop { document_id, ) } else { + let key = nodedb_types::StorageKey::for_surrogate(surrogate); self.sparse.get_in_txn( txn, params.database_id, params.tid, params.target_collection, - document_id, + &key, ) } } diff --git a/nodedb/src/data/executor/enforcement/period_lock.rs b/nodedb/src/data/executor/enforcement/period_lock.rs index 6acd36876..e3fac160f 100644 --- a/nodedb/src/data/executor/enforcement/period_lock.rs +++ b/nodedb/src/data/executor/enforcement/period_lock.rs @@ -10,16 +10,24 @@ use sonic_rs; use crate::bridge::envelope::ErrorCode; use crate::engine::sparse::btree::SparseEngine; -use nodedb_physical::physical_plan::PeriodLockConfig; +use nodedb_physical::physical_plan::{PeriodLockConfig, ResolvedSumTarget, resolved_sum_surrogate}; +use nodedb_types::StorageKey; /// Check whether a write is allowed given the period lock configuration. /// /// `doc_bytes` is the document being written (INSERT/UPDATE) or the existing -/// document (DELETE). The period column value is extracted and looked up in -/// the reference collection. +/// document (DELETE). The period column value is extracted from `doc_bytes` +/// and resolved to the reference collection's row through `resolved_targets`. +/// +/// A declared primary-key VALUE resolves to a row's surrogate through the +/// Control Plane's pk → surrogate catalog, which is off-limits to the Data +/// Plane. The Control Plane resolves `config.ref_pk`'s value at plan time and +/// carries the reference row's surrogate in `resolved_targets`, keyed by +/// `(config.ref_table, period value)` — the same slot and lookup every other +/// cross-collection resolution in this codebase uses. /// /// Returns `Ok(())` if the write is allowed, or `Err(PeriodLocked)` if the -/// period is closed/locked. +/// period is closed, locked, or names no row the Control Plane could resolve. pub fn check_period_lock( sparse: &SparseEngine, database_id: u64, @@ -27,6 +35,7 @@ pub fn check_period_lock( collection: &str, doc_bytes: &[u8], config: &PeriodLockConfig, + resolved_targets: &[ResolvedSumTarget], ) -> Result<(), ErrorCode> { // Extract the period column value from the document. let period_value = extract_period_value(doc_bytes, &config.period_column); @@ -36,26 +45,48 @@ pub fn check_period_lock( return Ok(()); }; - // Look up the period status in the reference collection. - let ref_doc = sparse - .get(database_id, tid, &config.ref_table, &period_key) - .map_err(|e| ErrorCode::Internal { - detail: format!("period lock: failed to read {}: {e}", config.ref_table), - })?; - - let Some(ref_bytes) = ref_doc else { - // Period key not found in reference table — reject (unknown period). + // Resolve the period value to the reference row's surrogate, exactly as + // the Control Plane resolved it at plan time. A period value with no + // binding in `resolved_targets` names no reference row and is treated as + // an unknown period below. + let Some(surrogate) = resolved_sum_surrogate(resolved_targets, &config.ref_table, &period_key) + else { return Err(ErrorCode::PeriodLocked { collection: collection.to_string(), }); }; + let ref_bytes = match sparse.get( + database_id, + tid, + &config.ref_table, + &StorageKey::for_surrogate(surrogate), + ) { + Ok(Some(bytes)) => bytes, + Ok(None) => { + // Period key not found in reference table — reject (unknown period). + return Err(ErrorCode::PeriodLocked { + collection: collection.to_string(), + }); + } + Err(e) => { + return Err(ErrorCode::Internal { + detail: format!("period lock: failed to read {}: {e}", config.ref_table), + }); + } + }; // Extract the status value from the reference document. - let status = extract_field_string(&ref_bytes, &config.status_column); - let Some(status) = status else { - // Status column not found — treat as locked (defensive). - return Err(ErrorCode::PeriodLocked { + let Some(status) = extract_field_string(&ref_bytes, &config.status_column) else { + // The reference row exists but does not carry the configured + // `status_column` — a misconfigured column name, refused as a + // config error rather than admitted or treated as locked. + return Err(ErrorCode::PeriodLockMisconfigured { collection: collection.to_string(), + ref_table: config.ref_table.clone(), + status_column: config.status_column.clone(), + row_identity: StorageKey::for_surrogate(surrogate) + .to_identity() + .to_string(), }); }; @@ -100,6 +131,7 @@ fn extract_field_string(bytes: &[u8], field_name: &str) -> Option { #[cfg(test)] mod tests { use super::*; + use nodedb_types::Surrogate; fn make_config(allowed: &[&str]) -> PeriodLockConfig { PeriodLockConfig { @@ -161,4 +193,156 @@ mod tests { .any(|s| s.eq_ignore_ascii_case("CLOSED")) ); } + + const DB: u64 = 0; + const TID: u64 = 1; + const REF_TABLE: &str = "fiscal_periods"; + const REF_ROW: Surrogate = Surrogate(9001); + + fn open_sparse(dir: &std::path::Path) -> SparseEngine { + SparseEngine::open(&dir.join("sparse.redb")).expect("open sparse engine") + } + + /// Seed the reference row through the surrogate a resolved target names — + /// never a bare string key. A bare-string seed is exactly the shape that + /// let the pre-fix lookup APPEAR to work while never matching a real row. + fn seed_ref_row(sparse: &SparseEngine, status: &str) { + let doc = serde_json::json!({"period_key": "2024-Q1", "status": status}); + sparse + .put( + DB, + TID, + REF_TABLE, + &StorageKey::for_surrogate(REF_ROW), + &nodedb_types::json_to_msgpack(&doc).unwrap(), + ) + .expect("seed reference row"); + } + + fn resolved(period_key: &str) -> Vec { + vec![ResolvedSumTarget::new(REF_TABLE, period_key, REF_ROW)] + } + + fn entry(period: &str) -> Vec { + nodedb_types::json_to_msgpack(&serde_json::json!({"fiscal_period": period})).unwrap() + } + + #[test] + fn a_resolved_open_period_admits_the_write() { + let dir = tempfile::tempdir().expect("tempdir"); + let sparse = open_sparse(dir.path()); + seed_ref_row(&sparse, "OPEN"); + let config = make_config(&["OPEN", "ADJUSTING"]); + + assert!( + check_period_lock( + &sparse, + DB, + TID, + "journal_entries", + &entry("2024-Q1"), + &config, + &resolved("2024-Q1"), + ) + .is_ok() + ); + } + + #[test] + fn a_resolved_closed_period_refuses_the_write() { + let dir = tempfile::tempdir().expect("tempdir"); + let sparse = open_sparse(dir.path()); + seed_ref_row(&sparse, "CLOSED"); + let config = make_config(&["OPEN", "ADJUSTING"]); + + let err = check_period_lock( + &sparse, + DB, + TID, + "journal_entries", + &entry("2024-Q1"), + &config, + &resolved("2024-Q1"), + ) + .expect_err("a closed period must refuse the write"); + assert!(matches!(err, ErrorCode::PeriodLocked { .. })); + } + + /// A period value with no entry in `resolved_targets` names no reference + /// row — an unknown period, refused exactly like a closed one. + #[test] + fn an_unresolved_period_refuses_the_write() { + let dir = tempfile::tempdir().expect("tempdir"); + let sparse = open_sparse(dir.path()); + seed_ref_row(&sparse, "OPEN"); + let config = make_config(&["OPEN", "ADJUSTING"]); + + let err = check_period_lock( + &sparse, + DB, + TID, + "journal_entries", + &entry("2024-Q1"), + &config, + &[], + ) + .expect_err("an unresolved period must refuse the write"); + assert!(matches!(err, ErrorCode::PeriodLocked { .. })); + } + + /// A reference row that exists but is missing the configured + /// `status_column` names a misconfigured column, not a locked period — + /// a typo in `status_column` must not silently refuse every write. + #[test] + fn a_reference_row_missing_the_status_column_is_a_config_error() { + let dir = tempfile::tempdir().expect("tempdir"); + let sparse = open_sparse(dir.path()); + let doc = serde_json::json!({"period_key": "2024-Q1"}); + sparse + .put( + DB, + TID, + REF_TABLE, + &StorageKey::for_surrogate(REF_ROW), + &nodedb_types::json_to_msgpack(&doc).unwrap(), + ) + .expect("seed reference row without a status column"); + let config = make_config(&["OPEN", "ADJUSTING"]); + + let err = check_period_lock( + &sparse, + DB, + TID, + "journal_entries", + &entry("2024-Q1"), + &config, + &resolved("2024-Q1"), + ) + .expect_err("a missing status column must refuse as a config error"); + match err { + ErrorCode::PeriodLockMisconfigured { + collection, + ref_table, + status_column, + .. + } => { + assert_eq!(collection, "journal_entries"); + assert_eq!(ref_table, REF_TABLE); + assert_eq!(status_column, "status"); + } + other => panic!("expected PeriodLockMisconfigured, got {other:?}"), + } + } + + /// A document with no period column at all is not subject to the lock — + /// a schemaless collection may carry rows the lock never gates. + #[test] + fn a_missing_period_column_admits_the_write() { + let dir = tempfile::tempdir().expect("tempdir"); + let sparse = open_sparse(dir.path()); + let config = make_config(&["OPEN"]); + let doc = nodedb_types::json_to_msgpack(&serde_json::json!({"amount": 100})).unwrap(); + + assert!(check_period_lock(&sparse, DB, TID, "journal_entries", &doc, &config, &[]).is_ok()); + } } diff --git a/nodedb/src/data/executor/enforcement/statement.rs b/nodedb/src/data/executor/enforcement/statement.rs index 717a4dee2..1b6736151 100644 --- a/nodedb/src/data/executor/enforcement/statement.rs +++ b/nodedb/src/data/executor/enforcement/statement.rs @@ -99,7 +99,14 @@ impl CoreLoop { }; let mut entries = Vec::new(); for document_id in document_ids { - let Some(stored) = self.sparse.get(database_id, tid, collection, document_id)? else { + // `document_id` arrives as a bare string several calls removed + // from the scan that produced it. A shape that fails to parse + // as a storage key is treated the same as a row that is not + // there: this check contributes nothing for a row it cannot read. + let Some(key) = nodedb_types::StorageKey::parse(document_id) else { + continue; + }; + let Some(stored) = self.sparse.get(database_id, tid, collection, &key)? else { continue; }; // A stored row of a collection that declares constraints over its diff --git a/nodedb/src/data/executor/handlers/bulk_dml/admission.rs b/nodedb/src/data/executor/handlers/bulk_dml/admission.rs index a21cf6a20..2cde59983 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/admission.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/admission.rs @@ -135,15 +135,10 @@ impl CoreLoop { // are inherent methods — no extra trait import needed. let mut edges: Vec = Vec::new(); for doc_id in matching_ids { - let surrogate = if doc_id.len() == 8 { - match u32::from_str_radix(doc_id, 16) { - Ok(s) => s, - Err(_) => continue, - } - } else { + let Some(key) = crate::engine::document::store::StorageKey::parse(doc_id) else { continue; }; - let Ok(Some(bytes)) = self.sparse.get(database_id, tid, collection, doc_id) else { + let Ok(Some(bytes)) = self.sparse.get(database_id, tid, collection, &key) else { continue; }; let Ok(doc) = doc_format::decode_document(&bytes) else { @@ -157,7 +152,7 @@ impl CoreLoop { .and_then(|v| v.as_str()) .map(str::to_string); edges.push(OllpPredictedEdge { - surrogate, + surrogate: key.surrogate().as_u32(), from: from.to_string(), to: to.to_string(), label, diff --git a/nodedb/src/data/executor/handlers/bulk_dml/delete.rs b/nodedb/src/data/executor/handlers/bulk_dml/delete.rs index 418bc5b2f..72aa20df3 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/delete.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/delete.rs @@ -1,6 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 -use tracing::{debug, warn}; +use tracing::debug; use crate::bridge::envelope::{ErrorCode, Response, WriteSetEntry}; use crate::bridge::scan_filter::ScanFilter; @@ -15,6 +15,8 @@ use nodedb_physical::physical_plan::{ OllpPredictedEdge, ResolvedSumTarget, ReturningSpec, StorageMode, }; +use super::delete_cascade::BulkDeleteRowCascade; + /// OLLP prediction inputs threaded to `execute_bulk_delete`: the predicted /// matched-doc surrogate set and the predicted implicit-edge set. Both are /// verified against the actual scan at admission time, returning @@ -174,7 +176,13 @@ impl CoreLoop { nodedb_types::WriteGateDecision::AdmitAll ) { for doc_id in &apply_ids { - let stored = match self.sparse.get(database_id, tid, collection, doc_id) { + // `doc_id` is a bare string from a raw-table scan; a shape + // that fails to parse as a storage key is treated the same + // as the row-already-gone case right below it. + let Some(key) = crate::engine::document::store::StorageKey::parse(doc_id) else { + continue; + }; + let stored = match self.sparse.get(database_id, tid, collection, &key) { Ok(Some(bytes)) => bytes, Ok(None) => continue, Err(e) => return self.response_error(task, e), @@ -225,6 +233,13 @@ impl CoreLoop { Vec::new() }; for doc_id in &apply_ids { + // `doc_id` is a bare string from a raw-table scan several calls + // removed from `SparseEngine`'s typed scan methods. A shape that + // fails to parse as a storage key can hold no row in DOCUMENTS + // either way, so every call below that needs the typed key treats + // it exactly like the "row already gone" case it already handles. + let storage_key = crate::engine::document::store::StorageKey::parse(doc_id); + // Capture pre-deletion snapshot if RETURNING was requested, or if // the collection is indexed (needed to recompute the removed // secondary-index tuples below — the delete cascade's prefix scan @@ -236,12 +251,12 @@ impl CoreLoop { let pre_delete_doc: Option = if returning.is_some() || !index_paths.is_empty() { - match self - .sparse - .get(task.request.database_id.as_u64(), tid, collection, doc_id) - .ok() - .flatten() - { + match storage_key.and_then(|key| { + self.sparse + .get(task.request.database_id.as_u64(), tid, collection, &key) + .ok() + .flatten() + }) { Some(bytes) => { // `doc_id` is the storage key from the scan. `RETURNING` // reports the row's client-visible identity, not the @@ -270,17 +285,36 @@ impl CoreLoop { Ok(txn) => txn, Err(e) => return self.response_error(task, e), }; - let deleted_bytes = self - .sparse - .delete_in_txn( - &row_txn, - task.request.database_id.as_u64(), + let deleted_bytes = storage_key.and_then(|key| { + self.sparse + .delete_in_txn( + &row_txn, + task.request.database_id.as_u64(), + tid, + collection, + &key, + ) + .ok() + .flatten() + }); + // Period lock, the pre-deletion image — a delete has no other. + // Checked before `write_hook::run` and before commit: dropping + // `row_txn` un-committed on a refusal reverses the removal. + if let Some(bytes) = deleted_bytes.as_deref() + && let Some(config) = self.doc_configs.get(&config_key) + && let Some(ref pl) = config.enforcement.period_lock + && let Err(e) = crate::data::executor::enforcement::period_lock::check_period_lock( + &self.sparse, + database_id, tid, collection, - doc_id, + bytes, + pl, + resolved_sum_targets, ) - .ok() - .flatten(); + { + return self.response_error(task, e); + } let mut target_writes = Vec::new(); if let Some(bytes) = deleted_bytes.as_deref() { match write_hook::run( @@ -320,123 +354,25 @@ impl CoreLoop { // rows only, so without these a WAL-only restart leaves every total // as it stood before the delete. write_set.extend(write_hook::target_write_set(&target_writes)); - if let Some(deleted_bytes) = deleted_bytes.as_deref() { - // Cascade: inverted index. doc_id is the hex-encoded surrogate - // (the redb storage key). Parse back once for FTS removal and - // reused below for the write version + write-set entry. - let row_surrogate = crate::engine::document::store::doc_id_to_surrogate(doc_id); - match row_surrogate { - Some(surrogate) => { - if let Err(e) = self.inverted.remove_document( - task.request.database_id.as_u64(), - crate::types::TenantId::new(tid), - collection, - surrogate, - ) { - warn!(core = self.core_id, %collection, %doc_id, error = %e, "bulk delete: inverted index removal failed"); - } - } - None => { - warn!(core = self.core_id, %collection, %doc_id, "bulk delete: doc_id is not a valid surrogate; FTS entry may be orphaned"); - } - } - // Cascade: secondary indexes. - if let Err(e) = self.sparse.delete_indexes_for_document( - task.request.database_id.as_u64(), - tid, - collection, - doc_id, - ) { - warn!(core = self.core_id, %collection, %doc_id, error = %e, "bulk delete: secondary index cascade failed"); - } - // Cascade: graph edges. - let edges_removed = self - .csr_partition_mut(database_id, tid) - .remove_node_edges(doc_id); - let cascade_ord = self.hlc.next_ordinal(); - if edges_removed > 0 - && let Err(e) = self.edge_store.delete_edges_for_node( + if let Some(bytes) = deleted_bytes.as_deref() { + self.bulk_delete_row_cascade( + BulkDeleteRowCascade { + task, database_id, - nodedb_types::TenantId::new(tid), - doc_id, - cascade_ord, - ) - { - warn!(core = self.core_id, %doc_id, error = %e, "bulk delete: edge cascade failed"); - } - self.mark_node_deleted(database_id, tid, doc_id); - // Cascade: secondary HNSW vector index. The put path indexed - // this row's vectors under its surrogate; the delete must - // soft-delete those nodes and drop the reverse-map entry, or the - // leaked vector keeps scoring in KNN search in the same process. - if has_vectors { - self.remove_document_vector_indexes(database_id, tid, collection, doc_id); - } - self.doc_cache.invalidate( - task.request.database_id.as_u64(), - tid, - collection, - doc_id, - ); - // Record the committed delete's write version against its - // surrogate + collection. - if let Some(surrogate) = row_surrogate { - self.note_surrogate_write_lsn(task, tid, collection, surrogate.as_u32()); - // Record the removed secondary-index tuples into the - // per-index write-value substrate, recomputed from the - // pre-delete document (see `index_paths` comment above). - if let (Some(lsn), Some(doc)) = (task.wal_lsn(), pre_delete_doc.as_ref()) { - let tuples = self.index_tuples_for_doc(doc, &index_paths); - self.note_index_write_values( - task.request.database_id, - crate::types::TenantId::new(tid), - collection, - &tuples, - lsn, - ); - } - // Carry the surrogate back for a post-apply `Delete` redo so - // the removed vector node does not resurrect on a WAL-only - // restart. Gated on `has_vectors` — a non-vector collection - // pays nothing. A delete carries no post-image body. - if has_vectors { - write_set.push(WriteSetEntry { - surrogate: surrogate.as_u32(), - is_delete: true, - value: Vec::new(), - collection: None, - }); - } - } - // Emit a delete event per affected row to the Event Plane, so - // AFTER-DELETE triggers and CDC/change-stream consumers see - // each row a bulk DELETE removed — mirroring - // `execute_point_delete`'s single-row emit. `deleted_bytes` is - // the prior stored bytes `sparse.delete` returned above (no - // second read needed); `resolve_event_payload` handles the - // strict->msgpack conversion for triggers. Emitted per row - // (not a `WriteOp::BulkDelete` summary) — the Event Plane's - // WAL-replay bulk variant is aggregate metadata reconstructed - // only when the live per-row events were lost. - let old_converted = self.resolve_event_payload( - task.request.database_id.as_u64(), - tid, - collection, - deleted_bytes, - ); - let event_identity = crate::engine::document::store::identity_of(doc_id); - self.emit_document_delete_event( - task, - collection, - event_identity, - Some(old_converted.as_deref().unwrap_or(deleted_bytes)), + tid, + collection, + doc_id: doc_id.as_str(), + storage_key, + deleted_bytes: bytes, + has_vectors, + index_paths: &index_paths, + pre_delete_doc, + returning: returning.is_some(), + }, + &mut write_set, + &mut returned_docs, ); affected += 1; - if returning.is_some() - && let Some(doc) = pre_delete_doc - { - returned_docs.push(doc); - } } } diff --git a/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs b/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs new file mode 100644 index 000000000..1e12010d2 --- /dev/null +++ b/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs @@ -0,0 +1,183 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Post-commit cascade for one bulk-deleted row: inverted index, secondary +//! indexes, graph edges, vector index, doc cache, write-version tracking, +//! and the Event Plane emit. +//! +//! Runs AFTER the row's own transaction committed, so none of this reverses +//! on failure — each step logs and continues rather than aborting a +//! statement it cannot undo. + +use tracing::warn; + +use crate::bridge::envelope::WriteSetEntry; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::task::ExecutionTask; +use crate::engine::document::store::{IndexPath, RowIdentity, StorageKey}; + +/// Borrowed + owned inputs for [`CoreLoop::bulk_delete_row_cascade`], grouped +/// so the call stays within the argument-count budget. +pub(in crate::data::executor) struct BulkDeleteRowCascade<'a> { + pub task: &'a ExecutionTask, + pub database_id: u64, + pub tid: u64, + pub collection: &'a str, + pub doc_id: &'a str, + pub storage_key: Option, + /// The row's pre-deletion bytes, as `sparse.delete` returned them — + /// never re-read. + pub deleted_bytes: &'a [u8], + pub has_vectors: bool, + pub index_paths: &'a [IndexPath], + /// The pre-deletion document, when the caller captured one (`RETURNING` + /// or an indexed collection). `None` costs this cascade nothing beyond + /// skipping the write-value note and the `RETURNING` row. + pub pre_delete_doc: Option, + pub returning: bool, +} + +impl CoreLoop { + /// Cascade one committed row removal into every secondary structure a + /// bulk delete must also clean up, and account it into `write_set` / + /// `returned_docs`. + pub(in crate::data::executor) fn bulk_delete_row_cascade( + &mut self, + cascade: BulkDeleteRowCascade<'_>, + write_set: &mut Vec, + returned_docs: &mut Vec, + ) { + let BulkDeleteRowCascade { + task, + database_id, + tid, + collection, + doc_id, + storage_key, + deleted_bytes, + has_vectors, + index_paths, + pre_delete_doc, + returning, + } = cascade; + + // Cascade: inverted index. doc_id is the hex-encoded surrogate + // (the redb storage key). `storage_key` already holds the parsed + // form; reuse it for the write version + write-set entry below. + let row_surrogate = storage_key.map(|key| key.surrogate()); + match row_surrogate { + Some(surrogate) => { + if let Err(e) = self.inverted.remove_document( + task.request.database_id.as_u64(), + crate::types::TenantId::new(tid), + collection, + surrogate, + ) { + // Recorded here, at the detection site: the row's own + // transaction has already committed, so this cleanup + // failure cannot roll it back. + crate::diag::orphaned_index_entry_after_delete(&e, collection, "inverted"); + warn!(core = self.core_id, %collection, %doc_id, error = %e, "bulk delete: inverted index removal failed"); + } + } + None => { + warn!(core = self.core_id, %collection, %doc_id, "bulk delete: doc_id is not a valid surrogate; FTS entry may be orphaned"); + } + } + // Cascade: secondary indexes. + if let Err(e) = self.sparse.delete_indexes_for_document( + task.request.database_id.as_u64(), + tid, + collection, + doc_id, + ) { + crate::diag::orphaned_index_entry_after_delete(&e, collection, "secondary"); + warn!(core = self.core_id, %collection, %doc_id, error = %e, "bulk delete: secondary index cascade failed"); + } + // Cascade: graph edges. + let edges_removed = self + .csr_partition_mut(database_id, tid) + .remove_node_edges(doc_id); + let cascade_ord = self.hlc.next_ordinal(); + if edges_removed > 0 + && let Err(e) = self.edge_store.delete_edges_for_node( + database_id, + nodedb_types::TenantId::new(tid), + doc_id, + cascade_ord, + ) + { + crate::diag::orphaned_index_entry_after_delete(&e, collection, "graph_edge"); + warn!(core = self.core_id, %doc_id, error = %e, "bulk delete: edge cascade failed"); + } + self.mark_node_deleted(database_id, tid, doc_id); + // Cascade: secondary HNSW vector index. The put path indexed + // this row's vectors under its surrogate; the delete must + // soft-delete those nodes and drop the reverse-map entry, or the + // leaked vector keeps scoring in KNN search in the same process. + if has_vectors { + self.remove_document_vector_indexes(database_id, tid, collection, doc_id); + } + if let Some(key) = storage_key { + self.doc_cache + .invalidate(task.request.database_id.as_u64(), tid, collection, &key); + } + // Record the committed delete's write version against its + // surrogate + collection. + if let Some(surrogate) = row_surrogate { + self.note_surrogate_write_lsn(task, tid, collection, surrogate.as_u32()); + // Record the removed secondary-index tuples into the + // per-index write-value substrate, recomputed from the + // pre-delete document (see `index_paths` comment above). + if let (Some(lsn), Some(doc)) = (task.wal_lsn(), pre_delete_doc.as_ref()) { + let tuples = self.index_tuples_for_doc(doc, index_paths); + self.note_index_write_values( + task.request.database_id, + crate::types::TenantId::new(tid), + collection, + &tuples, + lsn, + ); + } + // Carry the surrogate back for a post-apply `Delete` redo so + // the removed vector node does not resurrect on a WAL-only + // restart. Gated on `has_vectors` — a non-vector collection + // pays nothing. A delete carries no post-image body. + if has_vectors { + write_set.push(WriteSetEntry { + surrogate: surrogate.as_u32(), + is_delete: true, + value: Vec::new(), + collection: None, + }); + } + } + // Emit a delete event per affected row to the Event Plane, so + // AFTER-DELETE triggers and CDC/change-stream consumers see + // each row a bulk DELETE removed — mirroring + // `execute_point_delete`'s single-row emit. `deleted_bytes` is + // the prior stored bytes `sparse.delete` returned above (no + // second read needed); `resolve_event_payload` handles the + // strict->msgpack conversion for triggers. Emitted per row + // (not a `WriteOp::BulkDelete` summary) — the Event Plane's + // WAL-replay bulk variant is aggregate metadata reconstructed + // only when the live per-row events were lost. + let old_converted = self.resolve_event_payload( + task.request.database_id.as_u64(), + tid, + collection, + deleted_bytes, + ); + let event_identity = storage_key + .map(|key| key.to_identity()) + .unwrap_or_else(|| RowIdentity::from_user_key(doc_id)); + self.emit_document_delete_event( + task, + collection, + event_identity, + Some(old_converted.as_deref().unwrap_or(deleted_bytes)), + ); + if returning && let Some(doc) = pre_delete_doc { + returned_docs.push(doc); + } + } +} diff --git a/nodedb/src/data/executor/handlers/bulk_dml/mod.rs b/nodedb/src/data/executor/handlers/bulk_dml/mod.rs index 40eef9a87..772811af6 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/mod.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/mod.rs @@ -7,6 +7,7 @@ pub mod admission; pub mod delete; +pub mod delete_cascade; pub mod scan; pub mod update; pub mod update_persist; diff --git a/nodedb/src/data/executor/handlers/bulk_dml/update.rs b/nodedb/src/data/executor/handlers/bulk_dml/update.rs index 78a926219..3a0f08a19 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/update.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/update.rs @@ -227,12 +227,42 @@ impl CoreLoop { for row in projected { let ProjectedUpdateRow { doc_id, + storage_key, current_bytes, old_doc: old_doc_json, mut doc, updated_bytes, } = row; let doc_id = doc_id.as_str(); + // Period lock, both images — matching `execute_point_update`: a + // closed period must reject an edit to a row it already holds, + // and must reject an edit that assigns the period column into it. + if let Some(config) = self.doc_configs.get(&config_key) + && let Some(ref pl) = config.enforcement.period_lock + { + if let Err(e) = crate::data::executor::enforcement::period_lock::check_period_lock( + &self.sparse, + database_id, + tid, + collection, + ¤t_bytes, + pl, + resolved_sum_targets, + ) { + return self.response_error(task, e); + } + if let Err(e) = crate::data::executor::enforcement::period_lock::check_period_lock( + &self.sparse, + database_id, + tid, + collection, + &updated_bytes, + pl, + resolved_sum_targets, + ) { + return self.response_error(task, e); + } + } // Gate the persist on the collection's write policy, decided // against this row's post-update image — `doc` already has // the assignments and any regenerated columns applied, so it @@ -255,6 +285,7 @@ impl CoreLoop { tid, collection, doc_id, + storage_key: &storage_key, new_body: &updated_bytes, index_paths: &index_paths, old_doc: &old_doc_json, @@ -302,7 +333,7 @@ impl CoreLoop { task.request.database_id.as_u64(), tid, collection, - doc_id, + &storage_key, &updated_bytes, ); // Record the committed row's write version against its diff --git a/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs b/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs index 770e31885..f9a3a65e4 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs @@ -19,6 +19,9 @@ use crate::types::{DatabaseId, TenantId}; pub(in crate::data::executor) struct ProjectedUpdateRow { /// Storage key (the surrogate hex). pub(in crate::data::executor) doc_id: String, + /// The same storage key, typed — parsed once here so consumers never + /// re-interpret `doc_id`'s shape. + pub(in crate::data::executor) storage_key: crate::engine::document::store::StorageKey, /// The row as stored before the update — the `old_value` of the emitted /// event and the old side of the secondary-index diff. pub(in crate::data::executor) current_bytes: Vec, @@ -70,7 +73,14 @@ impl CoreLoop { let mut projected = Vec::with_capacity(doc_ids.len()); for doc_id in doc_ids { - let Some(current_bytes) = self.sparse.get(database_id, tid, collection, doc_id)? else { + // `doc_id` is a bare string from the settled apply set, several + // calls removed from any typed scan. A shape that fails to parse + // as a storage key is skipped the same as a row deleted between + // the match and this pass. + let Some(key) = crate::engine::document::store::StorageKey::parse(doc_id) else { + continue; + }; + let Some(current_bytes) = self.sparse.get(database_id, tid, collection, &key)? else { continue; }; @@ -176,6 +186,7 @@ impl CoreLoop { projected.push(ProjectedUpdateRow { doc_id: doc_id.clone(), + storage_key: key, current_bytes, old_doc, doc, diff --git a/nodedb/src/data/executor/handlers/control/calvin.rs b/nodedb/src/data/executor/handlers/control/calvin.rs index c591d1cb7..e152e84fd 100644 --- a/nodedb/src/data/executor/handlers/control/calvin.rs +++ b/nodedb/src/data/executor/handlers/control/calvin.rs @@ -509,7 +509,6 @@ mod tests { use crate::data::executor::core_loop::tests::make_core_with_dir; use crate::data::executor::doc_format; use crate::data::executor::handlers::transaction::overlay::Staged; - use crate::engine::document::store::surrogate_to_doc_id; use crate::types::{DatabaseId, RequestId, TraceId, VShardId}; /// A minimal `ExecutionTask` homing to vShard 0, tenant 1, database @@ -598,7 +597,7 @@ mod tests { /// Seed a row directly into base storage (bypassing Calvin staging), the /// pre-existing state the active-path OLLP verifier scans against. fn seed_row(core: &mut CoreLoop, collection: &str, surrogate: u32) { - let doc_id = surrogate_to_doc_id(Surrogate::new(surrogate)); + let doc_id = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); let body = doc_format::canonicalize_document_for_storage(&doc_value("a", "1")); core.sparse .put(DatabaseId::DEFAULT.as_u64(), 1, collection, &doc_id, &body) @@ -1015,7 +1014,7 @@ mod tests { ); // No base mutation at stage time — the row appears only after flush. - let doc_id = surrogate_to_doc_id(Surrogate::new(7)); + let doc_id = nodedb_types::StorageKey::for_surrogate(Surrogate::new(7)); assert!( core.sparse .get(DatabaseId::DEFAULT.as_u64(), 1, "orders", &doc_id) diff --git a/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs b/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs index 4818de70e..8e707996f 100644 --- a/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs +++ b/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs @@ -111,10 +111,13 @@ impl CoreLoop { rls_write_check.decision(), nodedb_types::WriteGateDecision::AdmitAll ) { - for doc_id in &doc_ids { + for (&surrogate, doc_id) in predicted_sorted.iter().zip(&doc_ids) { + let key = nodedb_types::StorageKey::for_surrogate(nodedb_types::Surrogate::new( + surrogate, + )); if let Some(body) = self.sparse - .get(task.request.database_id.as_u64(), tid, collection, doc_id)? + .get(task.request.database_id.as_u64(), tid, collection, &key)? { let identity = crate::engine::document::store::identity_of(doc_id); self.stage_admit_write( @@ -177,6 +180,7 @@ impl CoreLoop { for surrogate in predicted_sorted { let doc_id = surrogate_to_doc_id(Surrogate::new(surrogate)); + let storage_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); // Current body: overlay wins over base (read-your-own-writes), // mirroring `stage_point_update`'s exact overlay-then-base read. @@ -198,7 +202,7 @@ impl CoreLoop { ) } else { self.sparse - .get(database_id.as_u64(), tid, collection, &doc_id) + .get(database_id.as_u64(), tid, collection, &storage_key) }; match read { Ok(Some(bytes)) => bytes, diff --git a/nodedb/src/data/executor/handlers/control/calvin_resolve.rs b/nodedb/src/data/executor/handlers/control/calvin_resolve.rs index 1e79269df..9145acebc 100644 --- a/nodedb/src/data/executor/handlers/control/calvin_resolve.rs +++ b/nodedb/src/data/executor/handlers/control/calvin_resolve.rs @@ -98,7 +98,6 @@ mod tests { use crate::bridge::envelope::{Admission, ExemptReason, Priority, Request, Status}; use crate::data::executor::core_loop::tests::make_core_with_dir; use crate::data::executor::handlers::control::calvin::CalvinExecCtx; - use crate::engine::document::store::surrogate_to_doc_id; use crate::types::{DatabaseId, RequestId, TenantId, TraceId, VShardId}; use crate::wal::RedoRecord; @@ -198,7 +197,7 @@ mod tests { /// pre-existing state the predicate-write staging tests below apply /// their predicted surrogate set against. fn seed_row(core: &mut CoreLoop, collection: &str, surrogate: u32, field: &str, val: &str) { - let doc_id = surrogate_to_doc_id(Surrogate::new(surrogate)); + let doc_id = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); let body = crate::data::executor::doc_format::canonicalize_document_for_storage( &doc_value(field, val), ); diff --git a/nodedb/src/data/executor/handlers/control/crdt_doc.rs b/nodedb/src/data/executor/handlers/control/crdt_doc.rs index 0f7ac9c3d..582f29c23 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_doc.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_doc.rs @@ -231,6 +231,7 @@ impl CoreLoop { surrogate, user_roles: &task.request.user_roles, enforce: false, + resolved_targets: &[], }, ) { Ok(outcome) => outcome, diff --git a/nodedb/src/data/executor/handlers/control/crdt_materialize.rs b/nodedb/src/data/executor/handlers/control/crdt_materialize.rs index 51160d8dc..b24d665ff 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_materialize.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_materialize.rs @@ -130,6 +130,7 @@ impl CoreLoop { user_roles: &task.request.user_roles, enforce: false, wal_lsn: task.wal_lsn(), + resolved_targets: &[], }, ) { Ok(p) => p, diff --git a/nodedb/src/data/executor/handlers/control/crdt_preview.rs b/nodedb/src/data/executor/handlers/control/crdt_preview.rs index 74e047685..12568e53b 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_preview.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_preview.rs @@ -119,7 +119,6 @@ mod tests { use crate::bridge::envelope::{ErrorCode, Status}; use crate::data::executor::core_loop::tests::make_core_with_dir; use crate::data::executor::task::ExecutionTask; - use crate::engine::document::store::surrogate_to_doc_id; fn task() -> ExecutionTask { crate::data::executor::core_loop::tests::make_default_task() @@ -267,7 +266,12 @@ mod tests { ); assert_eq!( core.sparse - .get(DatabaseId::DEFAULT.as_u64(), 1, "docs", "00000001") + .get( + DatabaseId::DEFAULT.as_u64(), + 1, + "docs", + &nodedb_types::StorageKey::for_surrogate(nodedb_types::Surrogate::new(1)), + ) .expect("sparse read"), None, "preview must not materialize sparse state" @@ -395,7 +399,12 @@ mod tests { assert_eq!(core.checkpoint_coordinator.total_dirty_pages(), 0); assert_eq!( core.sparse - .get(DatabaseId::DEFAULT.as_u64(), 1, "docs", "0000004d") + .get( + DatabaseId::DEFAULT.as_u64(), + 1, + "docs", + &nodedb_types::StorageKey::for_surrogate(nodedb_types::Surrogate::new(0x4d)), + ) .expect("sparse read"), None ); @@ -454,7 +463,7 @@ mod tests { .and_then(|engine| engine.read_row("docs", "one")) .is_some() ); - let doc_id = surrogate_to_doc_id(surrogate); + let doc_id = nodedb_types::StorageKey::for_surrogate(surrogate); assert!( core.sparse .get(DatabaseId::DEFAULT.as_u64(), 1, "docs", &doc_id) diff --git a/nodedb/src/data/executor/handlers/convert.rs b/nodedb/src/data/executor/handlers/convert.rs index ce3678fbb..c9abc2171 100644 --- a/nodedb/src/data/executor/handlers/convert.rs +++ b/nodedb/src/data/executor/handlers/convert.rs @@ -25,7 +25,6 @@ use crate::data::executor::response_codec; use crate::data::executor::scan_normalize::sparse_body_to_msgpack; use crate::data::executor::sparse_body_format::SparseBodyFormat; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::identity_of; /// Map the plan's declared source storage mode to the row-decode format. /// @@ -148,7 +147,7 @@ impl CoreLoop { for (doc_id, doc_bytes) in &docs { let normalized = sparse_body_to_msgpack(doc_bytes, source_format.as_format_ref()); - let identity = identity_of(doc_id); + let identity = doc_id.to_identity(); let with_id = msgpack_scan::inject_str_field(&normalized, "id", identity.as_str()); let tuple_bytes = match super::super::strict_format::bytes_to_binary_tuple( @@ -240,7 +239,7 @@ impl CoreLoop { SparseBodyFormat::Strict(schema) => { let mut converted = 0u64; for (doc_id, doc_bytes) in &docs { - let identity = identity_of(doc_id); + let identity = doc_id.to_identity(); let Some(mp) = super::super::strict_format::binary_tuple_to_msgpack(doc_bytes, &schema) else { diff --git a/nodedb/src/data/executor/handlers/document/apply_balance_delta.rs b/nodedb/src/data/executor/handlers/document/apply_balance_delta.rs index b39072d22..ab6845469 100644 --- a/nodedb/src/data/executor/handlers/document/apply_balance_delta.rs +++ b/nodedb/src/data/executor/handlers/document/apply_balance_delta.rs @@ -35,7 +35,6 @@ use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::enforcement::materialized_sum::rmw::BalanceRmw; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; use nodedb_types::Surrogate; /// Dispatch-side arguments for [`CoreLoop::execute_apply_balance_delta`]. @@ -97,7 +96,7 @@ impl CoreLoop { }; let database_id = task.request.database_id.as_u64(); - let row_key = surrogate_to_doc_id(surrogate); + let row_key = nodedb_types::StorageKey::for_surrogate(surrogate); let txn = match self.sparse.begin_write() { Ok(txn) => txn, diff --git a/nodedb/src/data/executor/handlers/document/index_fetch.rs b/nodedb/src/data/executor/handlers/document/index_fetch.rs index 4de029a1b..072119355 100644 --- a/nodedb/src/data/executor/handlers/document/index_fetch.rs +++ b/nodedb/src/data/executor/handlers/document/index_fetch.rs @@ -274,7 +274,14 @@ impl CoreLoop { self.sparse .versioned_get_current(database_id, tid, collection, doc_id) } else { - self.sparse.get(database_id, tid, collection, doc_id) + // `doc_id` is an index-lookup result, not a scan of + // DOCUMENTS itself; a shape that fails to parse as a + // storage key names no row in that table, matching what + // a lookup on the unparsed key would already have found. + match nodedb_types::StorageKey::parse(doc_id) { + Some(key) => self.sparse.get(database_id, tid, collection, &key), + None => Ok(None), + } } }); match fetched { diff --git a/nodedb/src/data/executor/handlers/document/index_maintenance.rs b/nodedb/src/data/executor/handlers/document/index_maintenance.rs index 435925040..ee08e0c7e 100644 --- a/nodedb/src/data/executor/handlers/document/index_maintenance.rs +++ b/nodedb/src/data/executor/handlers/document/index_maintenance.rs @@ -132,6 +132,11 @@ impl CoreLoop { let mut pending_keys: Vec = Vec::with_capacity(docs.len()); for (doc_id, bytes) in &docs { + // `IndexEntryTxn` and the dedup map below are INDEXES-table + // concerns, out of this unit's typed scope, so the storage key + // is rendered once here at the boundary. + let doc_id = doc_id.to_string(); + let doc_id = doc_id.as_str(); // A row skipped here is a row the finished index permanently omits, // and the index is then reported as built — every later lookup on // that row's value silently misses it. @@ -170,7 +175,7 @@ impl CoreLoop { ); } if unique { - seen.insert(stored.clone(), doc_id.clone()); + seen.insert(stored.clone(), doc_id.to_string()); } pending_keys.push(crate::engine::sparse::btree_index::index_key_for( crate::engine::sparse::btree_index::IndexEntryTxn { diff --git a/nodedb/src/data/executor/handlers/document/read/fetch.rs b/nodedb/src/data/executor/handlers/document/read/fetch.rs index 73fbcae7c..5d9b18a93 100644 --- a/nodedb/src/data/executor/handlers/document/read/fetch.rs +++ b/nodedb/src/data/executor/handlers/document/read/fetch.rs @@ -24,68 +24,19 @@ use std::cell::Cell; use tracing::warn; -use nodedb_types::columnar::schema::StrictSchema; +use nodedb_types::StorageKey; use super::audit_body::{inject_temporal_columns, strict_audit_body}; -use crate::bridge::scan_filter::ScanFilter; +use super::fetch_types::Fetched; +pub(in crate::data::executor) use super::fetch_types::{ + DocFetchParams, DocScanMode, FetchedRows, RowOrigin, +}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::core_loop::filter_match::matches_with_resolved_schema; use crate::data::executor::scan_normalize::{sparse_body_to_msgpack, sparse_row_to_doc}; use crate::data::executor::sparse_body_format::{SparseBodyFormat, SparseBodyFormatRef}; use crate::data::executor::task::ExecutionTask; -/// Which temporal slice of a document collection a scan fetches. -pub(in crate::data::executor) enum DocScanMode { - /// Newest live version per document. Bitemporal collections read current - /// state from the versioned store; plain collections from the live table. - Current, - /// Newest version per document visible at a system-time cutoff and/or a - /// valid-time instant (`AS OF SYSTEM TIME` / `AS OF VALID TIME`). - AsOf { - system_as_of_ms: Option, - valid_at_ms: Option, - }, - /// Every system-time version of every document (`AS OF SYSTEM TIME NULL` - /// audit log), each row carrying the synthetic `_ts_*` temporal columns. - AllVersions { valid_at_ms: Option }, -} - -impl DocScanMode { - /// The current-time read is the only mode that folds this transaction's - /// staging overlay onto the base result — temporal reads never see staged - /// (current-version-only) writes. - pub(in crate::data::executor) fn is_current(&self) -> bool { - matches!(self, DocScanMode::Current) - } -} - -/// Borrowed inputs for [`CoreLoop::document_scan_fetch`]. -pub(in crate::data::executor) struct DocFetchParams<'a> { - pub collection: &'a str, - pub mode: &'a DocScanMode, - pub limit: usize, - pub offset: usize, - pub filter_predicates: &'a [ScanFilter], - pub strict_schema: Option<&'a StrictSchema>, - /// The fetch may not stop at `limit`: a downstream ORDER BY or DISTINCT - /// decides which rows survive, so the first `limit` rows the store happens - /// to return are not the first `limit` rows of the answer. The fetch is - /// bounded by the memory budget instead, and the caller surfaces - /// `ResourcesExhausted` rather than truncating silently. - pub full_fetch: bool, -} - -/// Raw rows plus the schema the downstream should decode them with. -pub(in crate::data::executor) struct FetchedRows { - pub rows: Vec<(String, Vec)>, - pub effective_schema: Option, - /// The statement's deadline passed while the storage scan was running, so - /// `rows` holds an arbitrary prefix of the answer. The caller MUST fail the - /// statement rather than emit these rows — a truncated result set is - /// indistinguishable from a complete one at the client. - pub deadline_expired: bool, -} - impl CoreLoop { /// Fetch the raw rows for a document scan according to `mode`, feeding the /// shared downstream shaping pipeline in [`super::scan`]. @@ -172,6 +123,7 @@ impl CoreLoop { .collect(); Ok(FetchedRows { rows, + origin: RowOrigin::Sparse, effective_schema: None, deadline_expired: deadline.tripped(), }) @@ -226,6 +178,7 @@ impl CoreLoop { } Ok(FetchedRows { rows, + origin: RowOrigin::Sparse, effective_schema: None, deadline_expired: deadline.tripped(), }) @@ -308,21 +261,46 @@ impl CoreLoop { } } }; + // `scan_documents_filtered` hands the predicate a typed `StorageKey`; + // `matches` (and `versioned_scan_as_of`'s predicate, and the + // `scan_collection` fallback filter) still take the row's storage + // key as text, so this renders it once per candidate row. + let matches_by_key = |key: &StorageKey, value: &[u8]| matches(&key.to_string(), value); - let rows = if filter_predicates.is_empty() { + // `versioned_scan_as_of` hands back a rendered storage key, parsed + // back ONCE here rather than carried as text and re-parsed later. A + // shape that fails to parse names a fetch-pipeline bug, not a row to + // skip. + let parse_row_key = |id: String, body: Vec| -> crate::Result<(StorageKey, Vec)> { + let key = StorageKey::parse(&id).ok_or_else(|| crate::Error::Storage { + engine: "sparse".into(), + detail: format!( + "collection '{collection}' fetched a row whose id is not a valid storage key: '{id}'" + ), + })?; + Ok((key, body)) + }; + + let rows: Fetched = if filter_predicates.is_empty() { if bitemporal { - self.sparse.versioned_scan_as_of( - crate::engine::sparse::btree_versioned::VersionedScanParams { - database_id, - tenant: tid, - coll: collection, - sys_cutoff_ms: None, - valid_at_ms: None, - limit: fetch_limit, - }, - &|_, _| true, - &stop, - )? + Fetched::Sparse( + self.sparse + .versioned_scan_as_of( + crate::engine::sparse::btree_versioned::VersionedScanParams { + database_id, + tenant: tid, + coll: collection, + sys_cutoff_ms: None, + valid_at_ms: None, + limit: fetch_limit, + }, + &|_, _| true, + &stop, + )? + .into_iter() + .map(|(id, body)| parse_row_key(id, body)) + .collect::>>()?, + ) } else { // Routed through the filtered scan with an always-true // predicate rather than `scan_documents`: the unfiltered full @@ -334,7 +312,7 @@ impl CoreLoop { tid, collection, fetch_limit, - &|_: &str, _: &[u8]| true, + &|_: &StorageKey, _: &[u8]| true, &stop, ); match sparse_result { @@ -349,64 +327,77 @@ impl CoreLoop { "document scan fallback to scan_collection" ); } - fallback + Fetched::Foreign(fallback) } - other => other?, + other => Fetched::Sparse(other?), } } } else if strict_schema.is_some() { if bitemporal { - self.sparse.versioned_scan_as_of( - crate::engine::sparse::btree_versioned::VersionedScanParams { - database_id, - tenant: tid, - coll: collection, - sys_cutoff_ms: None, - valid_at_ms: None, - limit: fetch_limit, - }, - &matches, - &stop, - )? + Fetched::Sparse( + self.sparse + .versioned_scan_as_of( + crate::engine::sparse::btree_versioned::VersionedScanParams { + database_id, + tenant: tid, + coll: collection, + sys_cutoff_ms: None, + valid_at_ms: None, + limit: fetch_limit, + }, + &matches, + &stop, + )? + .into_iter() + .map(|(id, body)| parse_row_key(id, body)) + .collect::>>()?, + ) } else { - self.sparse.scan_documents_filtered( + Fetched::Sparse(self.sparse.scan_documents_filtered( database_id, tid, collection, fetch_limit, - &matches, + &matches_by_key, &stop, - )? + )?) } } else if bitemporal { - self.sparse.versioned_scan_as_of( - crate::engine::sparse::btree_versioned::VersionedScanParams { - database_id, - tenant: tid, - coll: collection, - sys_cutoff_ms: None, - valid_at_ms: None, - limit: fetch_limit, - }, - &matches, - &stop, - )? + Fetched::Sparse( + self.sparse + .versioned_scan_as_of( + crate::engine::sparse::btree_versioned::VersionedScanParams { + database_id, + tenant: tid, + coll: collection, + sys_cutoff_ms: None, + valid_at_ms: None, + limit: fetch_limit, + }, + &matches, + &stop, + )? + .into_iter() + .map(|(id, body)| parse_row_key(id, body)) + .collect::>>()?, + ) } else { let sparse_result = self.sparse.scan_documents_filtered( database_id, tid, collection, fetch_limit, - &matches, + &matches_by_key, &stop, ); match sparse_result { - Ok(docs) if docs.is_empty() => self - .scan_collection(database_id, tid, collection, fetch_limit)? - .into_iter() - .filter(|(id, data)| matches(id, data)) - .collect(), - other => other?, + Ok(docs) if docs.is_empty() => Fetched::Foreign( + self.scan_collection(database_id, tid, collection, fetch_limit)? + .into_iter() + .filter(|(id, data)| matches(id, data)) + .collect(), + ), + other => Fetched::Sparse(other?), } }; @@ -420,17 +411,33 @@ impl CoreLoop { // every downstream transform — sort, window functions, computed // columns, projection, DISTINCT — sees the same standard-msgpack shape // it sees for every other collection. Without it the tagged values pass - // through untouched and reach the client as `[4,"alice"]`. - let rows = if is_vector_sidecar { - rows.into_iter() - .map(|(id, body)| sparse_row_to_doc(&id, &body, SparseBodyFormatRef::VectorSidecar)) - .collect() - } else { - rows + // through untouched and reach the client as `[4,"alice"]`. The key + // stays typed from the scan above, so this needs no re-parse. + let (rows, origin): (Vec<(String, Vec)>, RowOrigin) = match rows { + Fetched::Sparse(rows) if is_vector_sidecar => ( + rows.into_iter() + .map(|(key, body)| { + sparse_row_to_doc(&key, &body, SparseBodyFormatRef::VectorSidecar) + }) + .collect(), + RowOrigin::Sparse, + ), + Fetched::Sparse(rows) => ( + rows.into_iter() + .map(|(key, body)| (key.to_string(), body)) + .collect(), + RowOrigin::Sparse, + ), + // An empty fallback proves nothing about which engine owns the + // collection, and the rows a transaction overlay merges in later + // are sparse-shaped, so an empty fetch reports the sparse origin. + Fetched::Foreign(rows) if rows.is_empty() => (rows, RowOrigin::Sparse), + Fetched::Foreign(rows) => (rows, RowOrigin::Foreign), }; Ok(FetchedRows { rows, + origin, effective_schema: strict_schema.cloned(), deadline_expired: deadline.tripped(), }) diff --git a/nodedb/src/data/executor/handlers/document/read/fetch_types.rs b/nodedb/src/data/executor/handlers/document/read/fetch_types.rs new file mode 100644 index 000000000..d7e297e7a --- /dev/null +++ b/nodedb/src/data/executor/handlers/document/read/fetch_types.rs @@ -0,0 +1,81 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Inputs and outputs of the document scan fetch stage. + +use nodedb_types::StorageKey; +use nodedb_types::columnar::schema::StrictSchema; + +use crate::bridge::scan_filter::ScanFilter; + +/// Which temporal slice of a document collection a scan fetches. +pub(in crate::data::executor) enum DocScanMode { + /// Newest live version per document. Bitemporal collections read current + /// state from the versioned store; plain collections from the live table. + Current, + /// Newest version per document visible at a system-time cutoff and/or a + /// valid-time instant (`AS OF SYSTEM TIME` / `AS OF VALID TIME`). + AsOf { + system_as_of_ms: Option, + valid_at_ms: Option, + }, + /// Every system-time version of every document (`AS OF SYSTEM TIME NULL` + /// audit log), each row carrying the synthetic `_ts_*` temporal columns. + AllVersions { valid_at_ms: Option }, +} + +impl DocScanMode { + /// The current-time read is the only mode that folds this transaction's + /// staging overlay onto the base result — temporal reads never see staged + /// (current-version-only) writes. + pub(in crate::data::executor) fn is_current(&self) -> bool { + matches!(self, DocScanMode::Current) + } +} + +/// Rows a current-mode fetch produced, typed by [`RowOrigin`]. +pub(super) enum Fetched { + Sparse(Vec<(StorageKey, Vec)>), + Foreign(Vec<(String, Vec)>), +} + +/// Borrowed inputs for [`crate::data::executor::core_loop::CoreLoop::document_scan_fetch`]. +pub(in crate::data::executor) struct DocFetchParams<'a> { + pub collection: &'a str, + pub mode: &'a DocScanMode, + pub limit: usize, + pub offset: usize, + pub filter_predicates: &'a [ScanFilter], + pub strict_schema: Option<&'a StrictSchema>, + /// The fetch may not stop at `limit`: a downstream ORDER BY or DISTINCT + /// decides which rows survive, so the first `limit` rows the store happens + /// to return are not the first `limit` rows of the answer. The fetch is + /// bounded by the memory budget instead, and the caller surfaces + /// `ResourcesExhausted` rather than truncating silently. + pub full_fetch: bool, +} + +/// Which engine keyed the rows a fetch produced. A fetch never mixes the +/// two: the foreign fallback runs only when the sparse store holds nothing +/// for the collection, and an empty fetch always reports `Sparse`. +#[derive(Clone, Copy)] +pub(in crate::data::executor) enum RowOrigin { + /// Sparse-store rows. Every id is a rendered storage key. + Sparse, + /// `scan_collection` fallback rows. A KV row is keyed by its user key and + /// a columnar row by its `id` column, so the id is the engine's own + /// identity text, never a storage key, and the body is already a + /// standard msgpack map. + Foreign, +} + +/// Raw rows plus the schema the downstream should decode them with. +pub(in crate::data::executor) struct FetchedRows { + pub rows: Vec<(String, Vec)>, + pub origin: RowOrigin, + pub effective_schema: Option, + /// The statement's deadline passed while the storage scan was running, so + /// `rows` holds an arbitrary prefix of the answer. The caller MUST fail the + /// statement rather than emit these rows — a truncated result set is + /// indistinguishable from a complete one at the client. + pub deadline_expired: bool, +} diff --git a/nodedb/src/data/executor/handlers/document/read/materialize_scan.rs b/nodedb/src/data/executor/handlers/document/read/materialize_scan.rs index 7181494f5..a904f0caa 100644 --- a/nodedb/src/data/executor/handlers/document/read/materialize_scan.rs +++ b/nodedb/src/data/executor/handlers/document/read/materialize_scan.rs @@ -168,7 +168,12 @@ impl CoreLoop { self.sparse_body_format(task.request.database_id, TenantId::new(tid), collection); let format_ref = body_format.as_format_ref(); for entry in &mut entries { - let (_, normalized) = sparse_row_to_doc(&entry.0, &entry.2, format_ref); + // `entry.1` is this row's surrogate, already parsed out of + // `entry.0` when the row was collected above — minted directly + // from it rather than re-parsed. + let key = + nodedb_types::StorageKey::for_surrogate(nodedb_types::Surrogate::new(entry.1)); + let (_, normalized) = sparse_row_to_doc(&key, &entry.2, format_ref); entry.2 = normalized; } diff --git a/nodedb/src/data/executor/handlers/document/read/mod.rs b/nodedb/src/data/executor/handlers/document/read/mod.rs index 6d261a3d4..d628bb04b 100644 --- a/nodedb/src/data/executor/handlers/document/read/mod.rs +++ b/nodedb/src/data/executor/handlers/document/read/mod.rs @@ -6,6 +6,7 @@ mod audit_body; pub mod decode; pub mod emit; pub mod fetch; +pub mod fetch_types; pub mod materialize_scan; pub mod projection; pub mod scan; diff --git a/nodedb/src/data/executor/handlers/document/read/scan.rs b/nodedb/src/data/executor/handlers/document/read/scan.rs index e8f4b329b..8cb4aae9a 100644 --- a/nodedb/src/data/executor/handlers/document/read/scan.rs +++ b/nodedb/src/data/executor/handlers/document/read/scan.rs @@ -4,7 +4,7 @@ use tracing::{debug, warn}; -use super::fetch::{DocFetchParams, DocScanMode}; +use super::fetch::{DocFetchParams, DocScanMode, RowOrigin}; use super::projection::{apply_projection, apply_projection_msgpack}; use crate::bridge::envelope::{ErrorCode, Response}; use crate::bridge::scan_filter::ScanFilter; @@ -15,6 +15,35 @@ use crate::data::executor::scan_normalize::sparse_row_to_doc; use crate::data::executor::sparse_body_format::SparseBodyFormatRef; use crate::data::executor::task::ExecutionTask; +/// Shape one fetched row into `(identity, standard msgpack body)`. +/// +/// A sparse row's id reaches this handler as rendered storage-key text +/// several calls removed from the scan that produced it (`document_scan_fetch` +/// unifies every source — current, `AS OF`, bitemporal — to +/// `(String, Vec)`), so a shape that fails to parse names a +/// fetch-pipeline bug rather than a row to skip. A foreign row already +/// carries its engine's identity and a msgpack body, so it passes through. +fn fetched_row_to_doc( + origin: RowOrigin, + collection: &str, + id: String, + body: Vec, + body_format: SparseBodyFormatRef<'_>, +) -> crate::Result<(String, Vec)> { + match origin { + RowOrigin::Sparse => { + let key = nodedb_types::StorageKey::parse(&id).ok_or_else(|| crate::Error::Storage { + engine: "sparse".into(), + detail: format!( + "collection '{collection}' scan fetched a row whose id is not a valid storage key: '{id}'" + ), + })?; + Ok(sparse_row_to_doc(&key, &body, body_format)) + } + RowOrigin::Foreign => Ok((id, body)), + } +} + /// Parameters for [`CoreLoop::execute_document_scan`]. pub(in crate::data::executor) struct DocumentScanParams<'a> { pub tid: u64, @@ -147,6 +176,7 @@ impl CoreLoop { return self.response_error(task, ErrorCode::DeadlineExceeded); } let mut filtered = fetched.rows; + let origin = fetched.origin; let effective_schema = fetched.effective_schema; // The encoding the rows arrive in from the fetch stage. It is // NOT the collection's stored encoding: the fetch stage has @@ -236,10 +266,16 @@ impl CoreLoop { // the shared converter, which leaves an already-msgpack body // borrowed and so costs nothing on the schemaless path. let filtered = if !sort_keys.is_empty() || !projection.is_empty() { - filtered + match filtered .into_iter() - .map(|(id, bytes)| sparse_row_to_doc(&id, &bytes, body_format)) - .collect() + .map(|(id, bytes)| { + fetched_row_to_doc(origin, collection, id, bytes, body_format) + }) + .collect::>>() + { + Ok(rows) => rows, + Err(e) => return self.response_error(task, e), + } } else { filtered }; @@ -286,7 +322,8 @@ impl CoreLoop { let projected_rows: Vec<_> = match sorted .into_iter() .map(|(doc_id, val)| { - let (doc_id, mp) = sparse_row_to_doc(&doc_id, &val, body_format); + let (doc_id, mp) = + fetched_row_to_doc(origin, collection, doc_id, val, body_format)?; let projected = apply_projection_msgpack(&mp, &computed_cols, projection)?; Ok((doc_id, projected)) @@ -316,7 +353,8 @@ impl CoreLoop { let mut decoded_rows: Vec<(String, serde_json::Value)> = match sorted .into_iter() .map(|(id, val)| { - let (doc_id, mp) = sparse_row_to_doc(&id, &val, body_format); + let (doc_id, mp) = + fetched_row_to_doc(origin, collection, id, val, body_format)?; crate::data::executor::doc_format::decode_document(&mp) .map(|doc| (doc_id, doc)) }) @@ -370,7 +408,13 @@ impl CoreLoop { let projected_rows: Vec<_> = match sorted .into_iter() .map(|(doc_id, value)| { - let (doc_id, mp) = sparse_row_to_doc(&doc_id, &value, body_format); + let (doc_id, mp) = fetched_row_to_doc( + origin, + collection, + doc_id, + value, + body_format, + )?; let projected = apply_projection_msgpack(&mp, &computed_cols, projection)?; Ok((doc_id, projected)) diff --git a/nodedb/src/data/executor/handlers/document/resolve/apply.rs b/nodedb/src/data/executor/handlers/document/resolve/apply.rs index 3e1a07086..87344aa40 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/apply.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/apply.rs @@ -16,7 +16,7 @@ use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::rls_write_gate; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::StorageKey; impl CoreLoop { /// Handle `DocumentOp::ResolvedWrite`: check every precondition, apply every @@ -124,13 +124,9 @@ impl CoreLoop { ) -> Result<(), ErrorCode> { let database_id = task.request.database_id.as_u64(); for mutation in mutations { - let row_key = surrogate_to_doc_id(mutation.surrogate()); - let current = self.doc_current_bytes( - database_id, - tid, - mutation.collection().as_str(), - row_key.as_str(), - )?; + let row_key = StorageKey::for_surrogate(mutation.surrogate()); + let current = + self.doc_current_bytes(database_id, tid, mutation.collection().as_str(), &row_key)?; if current.as_deref() != mutation.precondition() { return Err(ErrorCode::OllpRetryRequired); } diff --git a/nodedb/src/data/executor/handlers/document/resolve/apply_row.rs b/nodedb/src/data/executor/handlers/document/resolve/apply_row.rs index d297345ec..ee8ebf5a0 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/apply_row.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/apply_row.rs @@ -81,13 +81,14 @@ impl CoreLoop { user_roles: &task.request.user_roles, enforce: true, wal_lsn: task.wal_lsn(), + resolved_targets: resolved_sum_targets, }, ) { Ok(outcome) => outcome, Err(e) => { // Dropping `txn` reverses the write but not the cache entry. self.doc_cache - .invalidate(database_id, tid, collection, row_key); + .invalidate(database_id, tid, collection, &storage_key); return Err(ErrorCode::from(e)); } }; @@ -113,7 +114,7 @@ impl CoreLoop { Ok(enforcement) => enforcement, Err(e) => { self.doc_cache - .invalidate(database_id, tid, collection, row_key); + .invalidate(database_id, tid, collection, &storage_key); return Err(ErrorCode::from(e)); } }; @@ -123,7 +124,7 @@ impl CoreLoop { self.settle_balanced_entries(database_id, tid, collection, enforcement.balanced_entries) { self.doc_cache - .invalidate(database_id, tid, collection, row_key); + .invalidate(database_id, tid, collection, &storage_key); return Err(ErrorCode::from(e)); } @@ -201,6 +202,7 @@ impl CoreLoop { surrogate, user_roles: &task.request.user_roles, enforce: true, + resolved_targets: resolved_sum_targets, }, ) .map_err(ErrorCode::from)?; diff --git a/nodedb/src/data/executor/handlers/document/resolve/bulk.rs b/nodedb/src/data/executor/handlers/document/resolve/bulk.rs index 8a7e9df6b..57d901745 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/bulk.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/bulk.rs @@ -24,7 +24,7 @@ use crate::data::executor::handlers::bulk_dml::update_project::{ }; use crate::data::executor::handlers::{returning_rows, rls_write_gate}; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::{RowIdentity, doc_id_to_surrogate}; +use crate::engine::document::store::RowIdentity; /// Borrowed arguments for [`CoreLoop::resolve_bulk_update`]. pub(super) struct ResolveBulkUpdate<'a> { @@ -107,6 +107,7 @@ impl CoreLoop { for row in projected { let ProjectedUpdateRow { doc_id, + storage_key, current_bytes, old_doc: _, doc, @@ -116,10 +117,9 @@ impl CoreLoop { // `execute_bulk_update` decides it — on the same JSON document. rls_write_gate::admit_row(rls_write_check, &doc, tid, collection) .map_err(ErrorCode::from)?; - let Some(surrogate) = doc_id_to_surrogate(&doc_id) else { - // No surrogate identity to address on a replica; live handler skips too. - continue; - }; + // The projection stage already parsed `doc_id` as a storage key, + // so its surrogate identity is always available here. + let surrogate = storage_key.surrogate(); mutations.push(put_mutation(ResolvedPut { collection, document_id: &doc_id, @@ -170,16 +170,19 @@ impl CoreLoop { let mut mutations = Vec::with_capacity(doc_ids.len()); let mut rows: Vec<(RowIdentity, Vec)> = Vec::new(); for doc_id in doc_ids { - // A row that vanished between the scan and this read removes - // nothing, so it carries no image for the policy to restrict. - let Some(stored) = self.doc_resolve_read(&ctx, collection, &doc_id)? else { + // `doc_id` is a bare string from the raw-table scan. A shape + // that fails to parse as a storage key can hold no row in + // DOCUMENTS either way, so it is skipped the same as a row that + // vanished between the scan and this read. + let Some(key) = crate::engine::document::store::StorageKey::parse(&doc_id) else { + continue; + }; + let Some(stored) = self.doc_resolve_read(&ctx, collection, &key)? else { continue; }; - // `doc_id` is the storage key from the scan. A value that fails - // to parse as a minted key is a legacy or user key, taken - // verbatim; `RETURNING` reports the client-visible identity - // either way, never the storage key. - let identity = crate::engine::document::store::identity_of(&doc_id); + // `RETURNING` reports the row's client-visible identity, never + // the storage key. + let identity = key.to_identity(); rls_write_gate::admit_stored_row( rls_write_check, &stored, @@ -189,9 +192,7 @@ impl CoreLoop { collection, ) .map_err(ErrorCode::from)?; - let Some(surrogate) = doc_id_to_surrogate(&doc_id) else { - continue; - }; + let surrogate = key.surrogate(); mutations.push(delete_mutation( collection, &doc_id, diff --git a/nodedb/src/data/executor/handlers/document/resolve/context.rs b/nodedb/src/data/executor/handlers/document/resolve/context.rs index abe188a94..255c677b7 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/context.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/context.rs @@ -15,7 +15,7 @@ use crate::bridge::envelope::ErrorCode; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::returning_rows; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::{RowIdentity, surrogate_to_doc_id}; +use crate::engine::document::store::{RowIdentity, StorageKey}; /// What a resolver returns: the decided mutations and the decided reply, or the /// error the live handler would have returned for the same input. @@ -67,7 +67,7 @@ impl CoreLoop { &self, ctx: &DocResolveCtx, collection: &str, - row_key: &str, + row_key: &StorageKey, ) -> Result>, ErrorCode> { self.doc_current_bytes(ctx.database_id, ctx.tid, collection, row_key) } @@ -79,11 +79,11 @@ impl CoreLoop { database_id: u64, tid: u64, collection: &str, - row_key: &str, + row_key: &StorageKey, ) -> Result>, ErrorCode> { let read = if self.is_bitemporal(database_id, tid, collection) { self.sparse - .versioned_get_current(database_id, tid, collection, row_key) + .versioned_get_current(database_id, tid, collection, &row_key.to_string()) } else { self.sparse.get(database_id, tid, collection, row_key) }; @@ -136,8 +136,8 @@ pub(super) fn delete_mutation( } /// The storage key for a row identity — the form every document reader uses. -pub(super) fn row_key_of(surrogate: Surrogate) -> String { - surrogate_to_doc_id(surrogate) +pub(super) fn row_key_of(surrogate: Surrogate) -> StorageKey { + StorageKey::for_surrogate(surrogate) } /// The `{"affected": N}` reply a write with no `RETURNING` clause returns. diff --git a/nodedb/src/data/executor/handlers/document/resolve/point.rs b/nodedb/src/data/executor/handlers/document/resolve/point.rs index 2e8ef7310..a5670c0d5 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/point.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/point.rs @@ -75,7 +75,6 @@ impl CoreLoop { } = args; let ctx = self.doc_resolve_ctx(task, tid, collection); let row_key = row_key_of(surrogate); - let row_key = row_key.as_str(); let row_identity = StorageKey::for_surrogate(surrogate).to_identity(); let document_identity = RowIdentity::from_user_key(document_id); @@ -98,7 +97,7 @@ impl CoreLoop { } // A gone row reports `{"affected": 0}`, same as `execute_point_update`. - let Some(current_bytes) = self.doc_resolve_read(&ctx, collection, row_key)? else { + let Some(current_bytes) = self.doc_resolve_read(&ctx, collection, &row_key)? else { return Ok(DocumentResolveOutcome { mutations: Vec::new(), response_payload: affected_payload(0), @@ -188,13 +187,12 @@ impl CoreLoop { } = args; let ctx = self.doc_resolve_ctx(task, tid, collection); let row_key = row_key_of(surrogate); - let row_key = row_key.as_str(); let row_identity = StorageKey::for_surrogate(surrogate).to_identity(); let document_identity = RowIdentity::from_user_key(document_id); // A row that is already absent removes nothing, so there is no image for // the policy to restrict — the same admission `gate_point_delete` makes. - let Some(prior) = self.doc_resolve_read(&ctx, collection, row_key)? else { + let Some(prior) = self.doc_resolve_read(&ctx, collection, &row_key)? else { return Ok(DocumentResolveOutcome { mutations: Vec::new(), response_payload: resolved_response_payload( diff --git a/nodedb/src/data/executor/handlers/document/resolve/upsert.rs b/nodedb/src/data/executor/handlers/document/resolve/upsert.rs index ccb9745c5..93cf6d295 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/upsert.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/upsert.rs @@ -57,12 +57,11 @@ impl CoreLoop { } = args; let ctx = self.doc_resolve_ctx(task, tid, collection); let row_key = row_key_of(surrogate); - let row_key = row_key.as_str(); let row_identity = crate::engine::document::store::StorageKey::for_surrogate(surrogate).to_identity(); let document_identity = RowIdentity::from_user_key(document_id); - let existing = self.doc_resolve_read(&ctx, collection, row_key)?; + let existing = self.doc_resolve_read(&ctx, collection, &row_key)?; let (body, precondition) = match existing { Some(current_bytes) => { let merged = self.merge_upsert_body( diff --git a/nodedb/src/data/executor/handlers/document/write/batch_insert.rs b/nodedb/src/data/executor/handlers/document/write/batch_insert.rs index a04774f17..02443ff5b 100644 --- a/nodedb/src/data/executor/handlers/document/write/batch_insert.rs +++ b/nodedb/src/data/executor/handlers/document/write/batch_insert.rs @@ -10,7 +10,6 @@ use crate::data::executor::enforcement::chain_guard::ChainGuard; use crate::data::executor::enforcement::write_hook::{self, HookCtx, ImageBody, WriteImages}; use crate::data::executor::handlers::point::apply_put::PointPutParams; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; use nodedb_physical::physical_plan::{ResolvedSumTarget, ReturningSpec}; /// Parameters for [`CoreLoop::execute_document_batch_insert`]. @@ -146,11 +145,13 @@ impl CoreLoop { let has_vectors = self.collection_has_vectors(database_id, tid, collection) || self.collection_has_sparse(database_id, tid, collection); - // Row key for post-commit event emission, captured as each row applies - // successfully; the value bytes are re-borrowed from `documents` after - // commit rather than cloned here. On any error we return early - // (dropping `txn`, which rolls back every row applied so far). - let mut applied: Vec = Vec::with_capacity(documents.len()); + // Row key + storage key for post-commit event emission and cache + // invalidation, captured as each row applies successfully; the value + // bytes are re-borrowed from `documents` after commit rather than + // cloned here. On any error we return early (dropping `txn`, which + // rolls back every row applied so far). + let mut applied: Vec<(String, nodedb_types::StorageKey)> = + Vec::with_capacity(documents.len()); let mut write_set: Vec = Vec::new(); // Per-row secondary-index tuples (added ∪ removed ∪ bitemporal), // parallel to `applied`. Recorded into the per-index write-value @@ -173,10 +174,11 @@ impl CoreLoop { // document-cache entries `apply_point_put` populated — are reversed in // one place. Dropping `txn` reverses the durable writes; it does not // reverse either of those. - let mut failure: Option<(String, crate::Error)> = None; + let mut failure: Option<(nodedb_types::StorageKey, crate::Error)> = None; for (i, (document_id, value)) in documents.iter().enumerate() { let surrogate = surrogates[i]; - let row_key = surrogate_to_doc_id(surrogate); + let key = nodedb_types::StorageKey::for_surrogate(surrogate); + let row_key = key.to_string(); // Every row of a batch insert is INSERT-shaped, so every row is a // chain link. The chain rewrites the BODY, so it runs before the // body is encoded and stored. The link covers the user-visible @@ -185,9 +187,10 @@ impl CoreLoop { let chained = match chain.chain_insert(self, database_id, tid, document_id, value) { Ok(chained) => chained, Err(e) => { - // Cloned, not moved: `row_key` is still borrowed by the - // parameters of the call this arm is handling. - failure = Some((row_key.clone(), e)); + // `key` is `Copy`, so this costs nothing and `row_key` + // stays borrowed by the parameters of the call this arm + // is handling. + failure = Some((key, e)); break; } }; @@ -205,13 +208,15 @@ impl CoreLoop { user_roles: &task.request.user_roles, enforce: true, wal_lsn: task.wal_lsn(), + resolved_targets: resolved_sum_targets, }, ) { Ok(o) => o, Err(e) => { - // Cloned, not moved: `row_key` is still borrowed by the - // parameters of the call this arm is handling. - failure = Some((row_key.clone(), e)); + // `key` is `Copy`, so this costs nothing and `row_key` + // stays borrowed by the parameters of the call this arm + // is handling. + failure = Some((key, e)); break; } }; @@ -229,9 +234,10 @@ impl CoreLoop { ) { Ok(enforcement) => enforcement, Err(e) => { - // Cloned, not moved: `row_key` is still borrowed by the - // parameters of the call this arm is handling. - failure = Some((row_key.clone(), e)); + // `key` is `Copy`, so this costs nothing and `row_key` + // stays borrowed by the parameters of the call this arm + // is handling. + failure = Some((key, e)); break; } }; @@ -254,18 +260,21 @@ impl CoreLoop { tuples.extend(outcome.bitemporal_index_tuples); row_index_tuples.push(tuples); } - applied.push(row_key); + applied.push((row_key, key)); } - if let Some((failed_row_key, error)) = failure { + if let Some((failed_key, error)) = failure { // The whole page rolls back, so put the chain head back where it // started and drop every cache entry the abandoned rows populated — // a cached body for a row that never committed is served to readers // as though it had. chain.restore(self); - for row_key in applied.iter().chain(std::iter::once(&failed_row_key)) { - self.doc_cache - .invalidate(database_id, tid, collection, row_key); + for key in applied + .iter() + .map(|(_, key)| key) + .chain(std::iter::once(&failed_key)) + { + self.doc_cache.invalidate(database_id, tid, collection, key); } return self.response_error(task, error); } @@ -276,9 +285,8 @@ impl CoreLoop { if let Err(e) = self.settle_balanced_entries(database_id, tid, collection, balanced_entries) { chain.restore(self); - for row_key in &applied { - self.doc_cache - .invalidate(database_id, tid, collection, row_key); + for (_, key) in &applied { + self.doc_cache.invalidate(database_id, tid, collection, key); } return self.response_error(task, e); } @@ -287,9 +295,8 @@ impl CoreLoop { // hashes it covers. if let Err(e) = chain.persist_head(self, &txn) { chain.restore(self); - for row_key in &applied { - self.doc_cache - .invalidate(database_id, tid, collection, row_key); + for (_, key) in &applied { + self.doc_cache.invalidate(database_id, tid, collection, key); } return self.response_error(task, e); } @@ -324,8 +331,8 @@ impl CoreLoop { m.record_document_insert(); } - for (i, row_key) in applied.iter().enumerate() { - let identity = crate::engine::document::store::identity_of(row_key); + for (i, (_, key)) in applied.iter().enumerate() { + let identity = key.to_identity(); self.emit_put_event(task, tid, collection, identity, &documents[i].1, None); } @@ -453,7 +460,7 @@ mod tests { DatabaseId::DEFAULT.as_u64(), TID, COLL, - &surrogate_to_doc_id(surrogate), + &nodedb_types::StorageKey::for_surrogate(surrogate), ) .unwrap() } @@ -621,7 +628,7 @@ mod tests { DatabaseId::DEFAULT.as_u64(), TID, SUM_TARGET, - &surrogate_to_doc_id(SUM_T1), + &nodedb_types::StorageKey::for_surrogate(SUM_T1), &doc_format::encode_to_msgpack(&seed), ) .expect("seed target row"); @@ -644,7 +651,7 @@ mod tests { DatabaseId::DEFAULT.as_u64(), TID, SUM_TARGET, - &surrogate_to_doc_id(surrogate), + &nodedb_types::StorageKey::for_surrogate(surrogate), ) .expect("read target row") .expect("target row must still exist"); diff --git a/nodedb/src/data/executor/handlers/facet.rs b/nodedb/src/data/executor/handlers/facet.rs index 42e43e93b..37eeb8c43 100644 --- a/nodedb/src/data/executor/handlers/facet.rs +++ b/nodedb/src/data/executor/handlers/facet.rs @@ -162,7 +162,14 @@ impl CoreLoop { ); let mut counts: HashMap = HashMap::new(); for doc_id in matching_ids { - if let Ok(Some(bytes)) = self.sparse.get(database_id, tid, collection, doc_id) { + // `doc_id` is a bare string from a raw-table scan several calls + // removed from `SparseEngine`'s typed scan methods; a shape that + // fails to parse as a storage key contributes no row, the same as + // a `get` miss below. + let Some(key) = nodedb_types::StorageKey::parse(doc_id) else { + continue; + }; + if let Ok(Some(bytes)) = self.sparse.get(database_id, tid, collection, &key) { let mp = crate::data::executor::scan_normalize::sparse_body_to_msgpack( &bytes, body_format.as_format_ref(), diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/abort.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/abort.rs index dd1c166bb..e12fad985 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/abort.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/abort.rs @@ -43,7 +43,14 @@ impl CoreLoop { keys: &[String], ) { for key in keys { - self.doc_cache.invalidate(database_id, tid, collection, key); + // `key` is a bare string accumulated several calls removed from + // any typed scan. Eviction is always safe to skip: a shape that + // fails to parse as a storage key can hold no live cache entry + // either way. + if let Some(key) = crate::engine::document::store::StorageKey::parse(key) { + self.doc_cache + .invalidate(database_id, tid, collection, &key); + } } } diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/insert_rows.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/insert_rows.rs index 50cf7cb81..fef8936d6 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/insert_rows.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/insert_rows.rs @@ -114,6 +114,7 @@ impl CoreLoop { user_roles: &task.request.user_roles, enforce: true, wal_lsn: task.wal_lsn(), + resolved_targets: resolved_sum_targets, }, ) { Ok(mut outcome) => { diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs index e402160a9..065997d27 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs @@ -115,6 +115,7 @@ impl CoreLoop { user_roles: &task.request.user_roles, enforce: true, wal_lsn: task.wal_lsn(), + resolved_targets: resolved_sum_targets, }, ) { Ok(mut outcome) => { @@ -199,45 +200,29 @@ impl CoreLoop { } } None => { - // Legacy non-surrogate target row: raw in-txn body rewrite - // (no cross-engine index — these rows predate surrogate - // keying and were never indexed). - applied_keys.push(upd.doc_id.clone()); - if let Err(e) = self.sparse.put_in_txn( - txn, + // A target row whose `doc_id` does not parse as a storage + // key: `put_in_txn` addresses DOCUMENTS rows by + // `StorageKey` only, and the workspace carries no + // on-disk-format compatibility burden for a row shape + // that predates surrogate keying, so this arm is refused + // rather than written through a raw string key. + return Err(self.abort_merge_apply(MergeAbort { + task, database_id, tid, collection, - &upd.doc_id, - &upd.body, - ) { - return Err(self.abort_merge_apply(MergeAbort { - task, - database_id, - tid, - collection, - applied_keys: applied_keys.as_slice(), - undo_log: std::mem::take(undo_log), - err: e.into(), - })); - } - if returning { - match returning_doc(&upd.body, &upd.doc_id) { - Ok(doc) => returned_docs.push(doc), - Err(e) => { - return Err(self.abort_merge_apply(MergeAbort { - task, - database_id, - tid, - collection, - applied_keys: applied_keys.as_slice(), - undo_log: std::mem::take(undo_log), - err: e.into(), - })); - } + applied_keys: applied_keys.as_slice(), + undo_log: std::mem::take(undo_log), + err: crate::Error::Storage { + engine: "document".into(), + detail: format!( + "MERGE UPDATE target row '{}' in '{collection}' has no \ + surrogate storage key", + upd.doc_id + ), } - } - *affected += 1; + .into(), + })); } } } diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs index 79c33a69f..34369c884 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs @@ -91,6 +91,7 @@ impl CoreLoop { surrogate, user_roles: &task.request.user_roles, enforce: true, + resolved_targets, }, ) { Ok(outcome) => { @@ -171,23 +172,23 @@ impl CoreLoop { } } None => { - if let Err(e) = self - .sparse - .delete(database_id, tid, collection, &del.doc_id) - { - return Err(self.response_error(task, e)); - } - // Legacy non-surrogate row: the raw delete reports no prior - // value, so the plan's captured pre-image is the only image - // of the removed row — without it a RETURNING delete of such - // a row would silently drop it from the result set. - if returning { - match returning_doc(&del.body, &del.doc_id) { - Ok(doc) => returned_docs.push(doc), - Err(e) => return Err(self.response_error(task, e)), - } - } - *affected += 1; + // A target row whose `doc_id` does not parse as a storage + // key: `delete` addresses DOCUMENTS rows by `StorageKey` + // only, and the workspace carries no on-disk-format + // compatibility burden for a row shape that predates + // surrogate keying, so this arm is refused rather than + // removed through a raw string key. + return Err(self.response_error( + task, + crate::Error::Storage { + engine: "document".into(), + detail: format!( + "MERGE DELETE target row '{}' in '{collection}' has no \ + surrogate storage key", + del.doc_id + ), + }, + )); } } } diff --git a/nodedb/src/data/executor/handlers/point/apply_delete.rs b/nodedb/src/data/executor/handlers/point/apply_delete.rs index 571721fd2..a62728e95 100644 --- a/nodedb/src/data/executor/handlers/point/apply_delete.rs +++ b/nodedb/src/data/executor/handlers/point/apply_delete.rs @@ -13,6 +13,7 @@ use tracing::warn; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::enforcement::{append_only, period_lock, retention}; +use nodedb_physical::physical_plan::ResolvedSumTarget; use nodedb_types::Surrogate; use crate::data::executor::handlers::point::apply_put::VectorIndexDelta; @@ -38,6 +39,11 @@ pub(in crate::data::executor) struct PointDeleteParams<'a> { /// deletes (e.g. CRDT-sync materialization) whose admission already /// happened on their origin replica. pub enforce: bool, + /// `(target collection, join-key value)` → target row surrogate, resolved + /// on the Control Plane at plan time — read by period-lock enforcement to + /// find its reference row. Empty for `enforce: false` callers, which + /// never read it. + pub resolved_targets: &'a [ResolvedSumTarget], } /// Capture of the mutations an [`CoreLoop::apply_point_delete`] performed, so @@ -134,11 +140,13 @@ impl CoreLoop { surrogate, user_roles, enforce, + resolved_targets, } = params; let _ = user_roles; let row_key = crate::engine::document::store::surrogate_to_doc_id(surrogate); let row_key = row_key.as_str(); + let storage_key = crate::engine::document::store::StorageKey::for_surrogate(surrogate); let bitemporal = self.is_bitemporal(database_id, tid, collection); let config_key = ( crate::types::DatabaseId::new(database_id), @@ -167,6 +175,7 @@ impl CoreLoop { collection, config, Some(body), + resolved_targets, )?; } let sys_from = self.bitemporal_now_ms(); @@ -223,7 +232,9 @@ impl CoreLoop { prior } else { if enforce && let Some(config) = self.doc_configs.get(&config_key) { - let old_value = self.sparse.get(database_id, tid, collection, row_key)?; + let old_value = self + .sparse + .get(database_id, tid, collection, &storage_key)?; run_delete_enforcement( &self.sparse, database_id, @@ -231,10 +242,11 @@ impl CoreLoop { collection, config, old_value.as_deref(), + resolved_targets, )?; } self.sparse - .delete_in_txn(txn, database_id, tid, collection, row_key)? + .delete_in_txn(txn, database_id, tid, collection, &storage_key)? }; // Capture the plain secondary-index `(field, value)` tuples this @@ -399,7 +411,7 @@ impl CoreLoop { // Invalidate document cache. self.doc_cache - .invalidate(database_id, tid, collection, row_key); + .invalidate(database_id, tid, collection, &storage_key); // Invalidate aggregate cache — a delete changes count(*) for this // collection. Only needed when a row was actually removed. @@ -431,14 +443,23 @@ fn run_delete_enforcement( collection: &str, config: &crate::engine::document::store::CollectionConfig, old_value: Option<&[u8]>, + resolved_targets: &[ResolvedSumTarget], ) -> crate::Result<()> { append_only::check_point_delete(collection, &config.enforcement) .map_err(map_enforcement_error)?; if let Some(ref pl) = config.enforcement.period_lock && let Some(old_bytes) = old_value { - period_lock::check_period_lock(sparse, database_id, tid, collection, old_bytes, pl) - .map_err(map_enforcement_error)?; + period_lock::check_period_lock( + sparse, + database_id, + tid, + collection, + old_bytes, + pl, + resolved_targets, + ) + .map_err(map_enforcement_error)?; } let created_at = old_value.and_then(retention::extract_created_at_secs); retention::check_delete_allowed(collection, &config.enforcement, created_at) diff --git a/nodedb/src/data/executor/handlers/point/apply_put/core.rs b/nodedb/src/data/executor/handlers/point/apply_put/core.rs index f6412d9be..9249e1e8c 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/core.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/core.rs @@ -43,7 +43,10 @@ impl CoreLoop { user_roles, enforce, wal_lsn, + resolved_targets, } = params; + // `surrogate` IS the storage key: no parse, no failure arm. + let storage_key = crate::engine::document::store::StorageKey::for_surrogate(surrogate); let config_key = ( crate::types::DatabaseId::new(database_id), crate::types::TenantId::new(tid), @@ -100,7 +103,8 @@ impl CoreLoop { self.sparse .versioned_get_current(database_id, tid, collection, document_id)? } else if need_old { - self.sparse.get(database_id, tid, collection, document_id)? + self.sparse + .get(database_id, tid, collection, &storage_key)? } else { None }; @@ -131,6 +135,7 @@ impl CoreLoop { value, old_value: &old_value, user_roles, + resolved_targets, }, )?; @@ -152,7 +157,7 @@ impl CoreLoop { old_value } else { self.sparse - .put_in_txn(txn, database_id, tid, collection, document_id, &stored)? + .put_in_txn(txn, database_id, tid, collection, &storage_key, &stored)? }; // Pre-image capture for the column-stats read-modify-write, so a @@ -205,7 +210,7 @@ impl CoreLoop { } self.doc_cache - .put(database_id, tid, collection, document_id, &stored); + .put(database_id, tid, collection, &storage_key, &stored); // Secondary index extraction into the caller's write txn — the // non-_in_txn variant would deadlock since `execute_point_put` @@ -395,8 +400,10 @@ mod tests { } fn stored_row(core: &CoreLoop, row_key: &str) -> Option> { + let key = crate::engine::document::store::StorageKey::parse(row_key) + .expect("test row_key is always a rendered storage key"); core.sparse - .get(DatabaseId::DEFAULT.as_u64(), TID, COLL, row_key) + .get(DatabaseId::DEFAULT.as_u64(), TID, COLL, &key) .unwrap() } @@ -466,9 +473,11 @@ mod tests { "the rejected write must leave no committed row — a stored row whose \ index update failed is invisible to full-text search forever" ); + let key = crate::engine::document::store::StorageKey::parse(&row_key) + .expect("test row_key is always a rendered storage key"); assert!( core.doc_cache - .get(DatabaseId::DEFAULT.as_u64(), TID, COLL, &row_key) + .get(DatabaseId::DEFAULT.as_u64(), TID, COLL, &key) .is_none(), "the rejected write must not populate the document cache either, or \ reads would serve a row that is not in durable storage" @@ -498,6 +507,7 @@ mod tests { user_roles: &[], enforce: true, wal_lsn: None, + resolved_targets: &[], }, ); @@ -538,6 +548,7 @@ mod tests { user_roles: &[], enforce: true, wal_lsn: None, + resolved_targets: &[], }, ); diff --git a/nodedb/src/data/executor/handlers/point/apply_put/enforce.rs b/nodedb/src/data/executor/handlers/point/apply_put/enforce.rs index 5d74f0f97..661755625 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/enforce.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/enforce.rs @@ -17,6 +17,7 @@ use crate::data::executor::enforcement::{ append_only, period_lock, state_transition, transition_check, }; use crate::types::{DatabaseId, TenantId}; +use nodedb_physical::physical_plan::ResolvedSumTarget; use super::types::map_enforcement_error; @@ -32,6 +33,11 @@ pub(in crate::data::executor::handlers::point) struct PutEnforcement<'a> { /// The row as currently stored, when one exists. pub(in crate::data::executor::handlers::point) old_value: &'a Option>, pub(in crate::data::executor::handlers::point) user_roles: &'a [String], + /// `(target collection, join-key value)` → target row surrogate, resolved + /// on the Control Plane at plan time. A period-lock check reads its + /// reference row's surrogate off this slice, keyed by + /// `(config.ref_table, period value)`. + pub(in crate::data::executor::handlers::point) resolved_targets: &'a [ResolvedSumTarget], } impl CoreLoop { @@ -58,6 +64,7 @@ impl CoreLoop { value, old_value, user_roles, + resolved_targets, } = p; if !enforce { return Ok(()); @@ -69,8 +76,16 @@ impl CoreLoop { append_only::check_point_put(collection, &config.enforcement, old_value) .map_err(map_enforcement_error)?; if let Some(ref pl) = config.enforcement.period_lock { - period_lock::check_period_lock(&self.sparse, database_id, tid, collection, value, pl) - .map_err(map_enforcement_error)?; + period_lock::check_period_lock( + &self.sparse, + database_id, + tid, + collection, + value, + pl, + resolved_targets, + ) + .map_err(map_enforcement_error)?; } // Both images must be readable whenever a transition rule is // configured: skipping a configured check because an image would diff --git a/nodedb/src/data/executor/handlers/point/apply_put/types.rs b/nodedb/src/data/executor/handlers/point/apply_put/types.rs index 26f80d5a0..9a8aaa214 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/types.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/types.rs @@ -7,6 +7,7 @@ use nodedb_types::Surrogate; use crate::bridge::envelope::ErrorCode; use crate::data::executor::spatial_key::SpatialIndexKey; +use nodedb_physical::physical_plan::ResolvedSumTarget; /// Parameters for [`CoreLoop::apply_point_put`](crate::data::executor::core_loop::CoreLoop::apply_point_put). pub(in crate::data::executor) struct PointPutParams<'a> { @@ -43,6 +44,11 @@ pub(in crate::data::executor) struct PointPutParams<'a> { /// record the vector checkpoint already absorbed. On the replay paths this /// carries the record's own LSN. pub wal_lsn: Option, + /// `(target collection, join-key value)` → target row surrogate, resolved + /// on the Control Plane at plan time — read by period-lock enforcement to + /// find its reference row. Empty for a caller whose statement type + /// resolves nothing, or for `enforce: false` callers, which never read it. + pub resolved_targets: &'a [ResolvedSumTarget], } /// Capture of the mutations an [`CoreLoop::apply_point_put`](crate::data::executor::core_loop::CoreLoop::apply_point_put) @@ -108,6 +114,17 @@ pub(in crate::data::executor) fn map_enforcement_error(e: ErrorCode) -> crate::E collection, detail: "period is closed or locked".to_string(), }, + ErrorCode::PeriodLockMisconfigured { + collection, + ref_table, + status_column, + row_identity, + } => crate::Error::PeriodLockMisconfigured { + collection, + ref_table, + status_column, + row_identity, + }, ErrorCode::StateTransitionViolation { collection, detail } => { crate::Error::StateTransitionViolation { collection, detail } } diff --git a/nodedb/src/data/executor/handlers/point/delete.rs b/nodedb/src/data/executor/handlers/point/delete.rs index 96e9d681f..b9f8fd104 100644 --- a/nodedb/src/data/executor/handlers/point/delete.rs +++ b/nodedb/src/data/executor/handlers/point/delete.rs @@ -90,6 +90,7 @@ impl CoreLoop { surrogate, user_roles: &task.request.user_roles, enforce: true, + resolved_targets: resolved_sum_targets, }, ) { Ok(outcome) => outcome, @@ -284,7 +285,8 @@ impl CoreLoop { self.sparse .versioned_get_current(database_id, tid, collection, row_key)? } else { - self.sparse.get(database_id, tid, collection, row_key)? + self.sparse + .get(database_id, tid, collection, &storage_key)? }; let Some(body) = stored else { return Ok(()); @@ -318,7 +320,7 @@ mod tests { use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; use crate::data::executor::doc_format; use crate::data::executor::handlers::point::insert::PointInsertParams; - use crate::engine::document::store::{CollectionConfig, surrogate_to_doc_id}; + use crate::engine::document::store::CollectionConfig; use crate::types::{DatabaseId, TenantId}; const DB: u64 = 0; @@ -375,7 +377,7 @@ mod tests { DB, TID, TARGET, - &surrogate_to_doc_id(T1), + &nodedb_types::StorageKey::for_surrogate(T1), &doc_format::encode_to_msgpack(&seed), ) .expect("seed target row"); @@ -394,7 +396,12 @@ mod tests { fn balance(core: &CoreLoop, surrogate: Surrogate) -> String { let stored = core .sparse - .get(DB, TID, TARGET, &surrogate_to_doc_id(surrogate)) + .get( + DB, + TID, + TARGET, + &nodedb_types::StorageKey::for_surrogate(surrogate), + ) .expect("read target") .expect("target row must exist"); doc_format::decode_document(&stored) @@ -516,7 +523,12 @@ mod tests { assert_eq!(resp.status, Status::Error); assert!( core.sparse - .get(DB, TID, SOURCE, &surrogate_to_doc_id(Surrogate(91))) + .get( + DB, + TID, + SOURCE, + &nodedb_types::StorageKey::for_surrogate(Surrogate(91)) + ) .expect("read back") .is_some(), "a refused delete must leave the chained row in place" diff --git a/nodedb/src/data/executor/handlers/point/get.rs b/nodedb/src/data/executor/handlers/point/get.rs index 75108c956..35e1f6594 100644 --- a/nodedb/src/data/executor/handlers/point/get.rs +++ b/nodedb/src/data/executor/handlers/point/get.rs @@ -41,6 +41,7 @@ impl CoreLoop { } = p; let row_key = surrogate_to_doc_id(surrogate); let row_key = row_key.as_str(); + let storage_key = nodedb_types::StorageKey::for_surrogate(surrogate); debug!( core = self.core_id, %collection, @@ -99,7 +100,7 @@ impl CoreLoop { } else { let cached = self .doc_cache - .get(database_id, tid, collection, row_key) + .get(database_id, tid, collection, &storage_key) .map(|v| v.to_vec()); if let Some(data) = cached { data @@ -108,12 +109,12 @@ impl CoreLoop { self.sparse .versioned_get_current(database_id, tid, collection, row_key) } else { - self.sparse.get(database_id, tid, collection, row_key) + self.sparse.get(database_id, tid, collection, &storage_key) }; match res { Ok(Some(data)) => { self.doc_cache - .put(database_id, tid, collection, row_key, &data); + .put(database_id, tid, collection, &storage_key, &data); data } Ok(None) => return self.response_with_payload(task, Vec::new()), @@ -149,7 +150,8 @@ impl CoreLoop { let transcoded = { let normalized = sparse_body_to_msgpack(&data, body_format.as_format_ref()); if !rls_filters.is_empty() { - let (_, gated) = sparse_row_to_doc(row_key, &data, body_format.as_format_ref()); + let (_, gated) = + sparse_row_to_doc(&storage_key, &data, body_format.as_format_ref()); if !super::super::rls_eval::rls_check_msgpack_bytes(rls_filters, &gated) { return self.response_with_payload(task, Vec::new()); } diff --git a/nodedb/src/data/executor/handlers/point/insert.rs b/nodedb/src/data/executor/handlers/point/insert.rs index 7e4b1a9f3..795017e14 100644 --- a/nodedb/src/data/executor/handlers/point/insert.rs +++ b/nodedb/src/data/executor/handlers/point/insert.rs @@ -115,7 +115,7 @@ impl CoreLoop { .versioned_exists_current_in_txn(&txn, database_id, tid, collection, row_key) } else { self.sparse - .exists_in_txn(&txn, database_id, tid, collection, row_key) + .exists_in_txn(&txn, database_id, tid, collection, &storage_key) }; match exists_result { Ok(true) => { @@ -175,11 +175,19 @@ impl CoreLoop { user_roles: &task.request.user_roles, enforce: true, wal_lsn: task.wal_lsn(), + resolved_targets: resolved_sum_targets, }, ) { Ok(o) => o, Err(e) => { - chain_guard::abort_after_apply(self, &chain, database_id, tid, collection, row_key); + chain_guard::abort_after_apply( + self, + &chain, + database_id, + tid, + collection, + &storage_key, + ); return self.response_error(task, e); } }; @@ -187,7 +195,14 @@ impl CoreLoop { // The advanced head lands in the SAME transaction as the row whose hash // it is, so head and row commit or roll back as one unit. if let Err(e) = chain.persist_head(self, &txn) { - chain_guard::abort_after_apply(self, &chain, database_id, tid, collection, row_key); + chain_guard::abort_after_apply( + self, + &chain, + database_id, + tid, + collection, + &storage_key, + ); return self.response_error(task, e); } @@ -206,7 +221,14 @@ impl CoreLoop { ) { Ok(outcome) => outcome, Err(e) => { - chain_guard::abort_after_apply(self, &chain, database_id, tid, collection, row_key); + chain_guard::abort_after_apply( + self, + &chain, + database_id, + tid, + collection, + &storage_key, + ); return self.response_error(task, e); } }; @@ -222,7 +244,14 @@ impl CoreLoop { if let Err(e) = self.settle_balanced_entries(database_id, tid, collection, enforcement.balanced_entries) { - chain_guard::abort_after_apply(self, &chain, database_id, tid, collection, row_key); + chain_guard::abort_after_apply( + self, + &chain, + database_id, + tid, + collection, + &storage_key, + ); return self.response_error(task, e); } @@ -308,7 +337,7 @@ mod tests { use crate::bridge::envelope::Status; use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; use crate::data::executor::doc_format; - use crate::engine::document::store::{CollectionConfig, surrogate_to_doc_id}; + use crate::engine::document::store::CollectionConfig; use crate::types::{DatabaseId, TenantId}; const DB: u64 = 0; @@ -370,7 +399,7 @@ mod tests { DB, TID, TARGET, - &surrogate_to_doc_id(T1), + &nodedb_types::StorageKey::for_surrogate(T1), &doc_format::encode_to_msgpack(&seed), ) .expect("seed target row"); @@ -389,7 +418,12 @@ mod tests { fn balance(core: &CoreLoop, surrogate: Surrogate) -> String { let stored = core .sparse - .get(DB, TID, TARGET, &surrogate_to_doc_id(surrogate)) + .get( + DB, + TID, + TARGET, + &nodedb_types::StorageKey::for_surrogate(surrogate), + ) .expect("read target") .expect("target row must exist"); doc_format::decode_document(&stored) @@ -483,7 +517,12 @@ mod tests { let stored = core .sparse - .get(DB, TID, SOURCE, &surrogate_to_doc_id(Surrogate(91))) + .get( + DB, + TID, + SOURCE, + &nodedb_types::StorageKey::for_surrogate(Surrogate(91)), + ) .expect("read back") .expect("row must exist"); let doc = doc_format::decode_document(&stored).expect("decode"); diff --git a/nodedb/src/data/executor/handlers/point/put.rs b/nodedb/src/data/executor/handlers/point/put.rs index 01684d98c..221c2d375 100644 --- a/nodedb/src/data/executor/handlers/point/put.rs +++ b/nodedb/src/data/executor/handlers/point/put.rs @@ -73,7 +73,7 @@ impl CoreLoop { let chained = if chain.enabled() && self .sparse - .get(database_id, tid, collection, row_key) + .get(database_id, tid, collection, &storage_key) .ok() .flatten() .is_none() @@ -109,17 +109,32 @@ impl CoreLoop { user_roles: &task.request.user_roles, enforce: true, wal_lsn: task.wal_lsn(), + resolved_targets: resolved_sum_targets, }, ) { Ok(p) => p, Err(e) => { - chain_guard::abort_after_apply(self, &chain, database_id, tid, collection, row_key); + chain_guard::abort_after_apply( + self, + &chain, + database_id, + tid, + collection, + &storage_key, + ); return self.response_error(task, e); } }; if let Err(e) = chain.persist_head(self, &txn) { - chain_guard::abort_after_apply(self, &chain, database_id, tid, collection, row_key); + chain_guard::abort_after_apply( + self, + &chain, + database_id, + tid, + collection, + &storage_key, + ); return self.response_error(task, e); } @@ -139,7 +154,14 @@ impl CoreLoop { let enforcement = match write_hook::run(self, &txn, &hook_ctx, images) { Ok(outcome) => outcome, Err(e) => { - chain_guard::abort_after_apply(self, &chain, database_id, tid, collection, row_key); + chain_guard::abort_after_apply( + self, + &chain, + database_id, + tid, + collection, + &storage_key, + ); return self.response_error(task, e); } }; @@ -151,7 +173,14 @@ impl CoreLoop { if let Err(e) = self.settle_balanced_entries(database_id, tid, collection, enforcement.balanced_entries) { - chain_guard::abort_after_apply(self, &chain, database_id, tid, collection, row_key); + chain_guard::abort_after_apply( + self, + &chain, + database_id, + tid, + collection, + &storage_key, + ); return self.response_error(task, e); } @@ -226,7 +255,7 @@ mod tests { use crate::bridge::envelope::Status; use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; use crate::data::executor::doc_format; - use crate::engine::document::store::{CollectionConfig, surrogate_to_doc_id}; + use crate::engine::document::store::CollectionConfig; use crate::types::{DatabaseId, TenantId}; const DB: u64 = 0; @@ -283,7 +312,7 @@ mod tests { DB, TID, TARGET, - &surrogate_to_doc_id(T1), + &nodedb_types::StorageKey::for_surrogate(T1), &doc_format::encode_to_msgpack(&seed), ) .expect("seed target row"); @@ -302,7 +331,12 @@ mod tests { fn balance(core: &CoreLoop, surrogate: Surrogate) -> String { let stored = core .sparse - .get(DB, TID, TARGET, &surrogate_to_doc_id(surrogate)) + .get( + DB, + TID, + TARGET, + &nodedb_types::StorageKey::for_surrogate(surrogate), + ) .expect("read target") .expect("target row must exist"); doc_format::decode_document(&stored) diff --git a/nodedb/src/data/executor/handlers/point/update/exec.rs b/nodedb/src/data/executor/handlers/point/update/exec.rs index 27ca912f9..8ec3dbc0a 100644 --- a/nodedb/src/data/executor/handlers/point/update/exec.rs +++ b/nodedb/src/data/executor/handlers/point/update/exec.rs @@ -140,10 +140,33 @@ impl CoreLoop { self.sparse .versioned_get_current(database_id, tid, collection, row_key) } else { - self.sparse.get(database_id, tid, collection, row_key) + self.sparse.get(database_id, tid, collection, &storage_key) }; match get_result { Ok(Some(current_bytes)) => { + // A period lock gates both images of an update: the PRE-image, + // because a closed period must reject any edit to a row it + // already holds, and the POST-image, because the update may + // itself assign the period column into a closed period. Put + // and delete each check the one image they have; update has + // both, and skipping either admits a write put and delete + // both refuse. + if let Some(config) = self.doc_configs.get(&config_key) + && let Some(ref pl) = config.enforcement.period_lock + && let Err(e) = + crate::data::executor::enforcement::period_lock::check_period_lock( + &self.sparse, + database_id, + tid, + collection, + ¤t_bytes, + pl, + resolved_sum_targets, + ) + { + return self.response_error(task, e); + } + let has_generated = self.doc_configs.get(&config_key).is_some_and(|c| { !c.enforcement.generated_columns.is_empty() && crate::data::executor::handlers::generated::needs_recomputation( @@ -167,6 +190,25 @@ impl CoreLoop { Err(e) => return self.response_error(task, e), }; + // The POST-image half of the period-lock check: refuses an + // update that assigns the period column into a closed period, + // even when the pre-image lived in an open one. + if let Some(config) = self.doc_configs.get(&config_key) + && let Some(ref pl) = config.enforcement.period_lock + && let Err(e) = + crate::data::executor::enforcement::period_lock::check_period_lock( + &self.sparse, + database_id, + tid, + collection, + &updated_bytes, + pl, + resolved_sum_targets, + ) + { + return self.response_error(task, e); + } + // Gate the persist on the collection's write policy, decided // against the post-update image the row will actually hold. // Placed after the generated columns are recomputed — a policy @@ -189,6 +231,7 @@ impl CoreLoop { tid, collection, row_key, + storage_key: &storage_key, current_bytes: ¤t_bytes, updated_bytes: &updated_bytes, bitemporal, @@ -202,7 +245,7 @@ impl CoreLoop { task.request.database_id.as_u64(), tid, collection, - row_key, + &storage_key, &updated_bytes, ); @@ -311,7 +354,7 @@ mod tests { use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; use crate::data::executor::doc_format; use crate::data::executor::handlers::point::insert::PointInsertParams; - use crate::engine::document::store::{CollectionConfig, surrogate_to_doc_id}; + use crate::engine::document::store::CollectionConfig; use crate::types::{DatabaseId, TenantId}; const DB: u64 = 0; @@ -375,7 +418,7 @@ mod tests { DB, TID, TARGET, - &surrogate_to_doc_id(surrogate), + &nodedb_types::StorageKey::for_surrogate(surrogate), &doc_format::encode_to_msgpack(&seed), ) .expect("seed target row"); @@ -395,7 +438,12 @@ mod tests { fn balance(core: &CoreLoop, surrogate: Surrogate) -> String { let stored = core .sparse - .get(DB, TID, TARGET, &surrogate_to_doc_id(surrogate)) + .get( + DB, + TID, + TARGET, + &nodedb_types::StorageKey::for_surrogate(surrogate), + ) .expect("read target") .expect("target row must exist"); doc_format::decode_document(&stored) @@ -568,7 +616,12 @@ mod tests { let stored = core .sparse - .get(DB, TID, SOURCE, &surrogate_to_doc_id(Surrogate(91))) + .get( + DB, + TID, + SOURCE, + &nodedb_types::StorageKey::for_surrogate(Surrogate(91)), + ) .expect("read back") .expect("row must exist"); let doc = doc_format::decode_document(&stored).expect("decode"); diff --git a/nodedb/src/data/executor/handlers/point/update/persist.rs b/nodedb/src/data/executor/handlers/point/update/persist.rs index 328731339..fbaec126d 100644 --- a/nodedb/src/data/executor/handlers/point/update/persist.rs +++ b/nodedb/src/data/executor/handlers/point/update/persist.rs @@ -36,6 +36,9 @@ pub(in crate::data::executor) struct PointUpdatePersist<'a> { pub(in crate::data::executor) collection: &'a str, /// Storage key (the surrogate hex). pub(in crate::data::executor) row_key: &'a str, + /// The same storage key, typed — passed alongside `row_key` because the + /// versioned-table methods below still take the rendered text. + pub(in crate::data::executor) storage_key: &'a crate::engine::document::store::StorageKey, /// The row as it was before this update — the old side of the index diff, /// and the pre-image every folded constraint subtracts. pub(in crate::data::executor) current_bytes: &'a [u8], @@ -69,6 +72,7 @@ impl CoreLoop { tid, collection, row_key, + storage_key, current_bytes, updated_bytes, bitemporal, @@ -163,7 +167,14 @@ impl CoreLoop { // No secondary index to maintain — nothing to diff, and no index // tuples to publish, so the body write is the whole write. self.sparse - .put_in_txn(&txn, database_id, tid, collection, row_key, updated_bytes) + .put_in_txn( + &txn, + database_id, + tid, + collection, + storage_key, + updated_bytes, + ) .map(|_prior| Vec::new()) } else { // Reconcile the plain secondary index atomically with the @@ -191,6 +202,7 @@ impl CoreLoop { tid, collection, doc_id: row_key, + storage_key, new_body: updated_bytes, index_paths: &index_paths, old_doc: &old_doc, diff --git a/nodedb/src/data/executor/handlers/point/update_reindex.rs b/nodedb/src/data/executor/handlers/point/update_reindex.rs index b4659aaea..353d79440 100644 --- a/nodedb/src/data/executor/handlers/point/update_reindex.rs +++ b/nodedb/src/data/executor/handlers/point/update_reindex.rs @@ -44,6 +44,8 @@ pub(in crate::data::executor) struct NonbitemporalUpdateReindex<'a> { pub tid: u64, pub collection: &'a str, pub doc_id: &'a str, + /// The same storage key, typed — `put_in_txn` below takes it directly. + pub storage_key: &'a crate::engine::document::store::StorageKey, /// New stored bytes for the primary document row. pub new_body: &'a [u8], pub index_paths: &'a [IndexPath], @@ -193,7 +195,7 @@ impl CoreLoop { p.database_id, p.tid, p.collection, - p.doc_id, + p.storage_key, p.new_body, )?; diff --git a/nodedb/src/data/executor/handlers/recursive.rs b/nodedb/src/data/executor/handlers/recursive.rs index 7c6dbcc44..2a554da56 100644 --- a/nodedb/src/data/executor/handlers/recursive.rs +++ b/nodedb/src/data/executor/handlers/recursive.rs @@ -124,7 +124,7 @@ impl CoreLoop { // identity only in the storage key, never in the body, so the row // image must inject it before any predicate runs — otherwise // `id IS NULL` and RETURNING rows both lose the identity. - let to_msgpack = |doc_id: &str, value: &[u8]| -> Vec { + let to_msgpack = |doc_id: &nodedb_types::StorageKey, value: &[u8]| -> Vec { crate::data::executor::scan_normalize::sparse_row_to_doc( doc_id, value, @@ -268,7 +268,7 @@ impl CoreLoop { let key = if distinct { nodedb_types::msgpack_to_json_string(&mp).unwrap_or_default() } else { - doc_id.clone() + doc_id.to_string() }; if !distinct || seen_keys.insert(key) { new_rows.push(mp); diff --git a/nodedb/src/data/executor/handlers/returning_rows.rs b/nodedb/src/data/executor/handlers/returning_rows.rs index 43505e0c2..83aa32145 100644 --- a/nodedb/src/data/executor/handlers/returning_rows.rs +++ b/nodedb/src/data/executor/handlers/returning_rows.rs @@ -158,7 +158,7 @@ impl CoreLoop { task: &ExecutionTask, spec: &ReturningSpec, rls_filters: &[u8], - row_key: &str, + row_key: &nodedb_types::StorageKey, sidecar: &[u8], ) -> Response { let (_id, mp) = sparse_row_to_doc(row_key, sidecar, SparseBodyFormatRef::VectorSidecar); diff --git a/nodedb/src/data/executor/handlers/spatial.rs b/nodedb/src/data/executor/handlers/spatial.rs index b78930f3a..95a789252 100644 --- a/nodedb/src/data/executor/handlers/spatial.rs +++ b/nodedb/src/data/executor/handlers/spatial.rs @@ -203,16 +203,22 @@ impl CoreLoop { None => continue, }; + // The doc-map id is the hex surrogate for a document-collection + // row, and a columnar-family user `id` otherwise — the latter + // never parses as a storage key, so it falls through exactly + // where a sparse miss on a parsed key would. + let storage_key = nodedb_types::StorageKey::parse(&doc_id); + // Prefilter: skip candidates not in the surrogate bitmap before // any geometry evaluation. The doc_id is a hex-encoded surrogate. if let Some(bitmap) = prefilter { - match u32::from_str_radix(&doc_id, 16) { - Ok(raw) => { - if !bitmap.contains(Surrogate(raw)) { + match storage_key { + Some(key) => { + if !bitmap.contains(key.surrogate()) { continue; } } - Err(_) => continue, + None => continue, } } @@ -223,10 +229,24 @@ impl CoreLoop { // is resolved from the columnar id → doc map. Both forms normalise // to standard msgpack maps that `decode_document_value` and // `extract_geometry` read identically. - let doc = match self.sparse.get(database_id, tid, collection, &doc_id) { + let sparse_result = match storage_key { + Some(key) => self.sparse.get(database_id, tid, collection, &key), + None => Ok(None), + }; + let doc = match sparse_result { Ok(Some(raw)) => { + // `sparse_result` is only `Ok(Some(_))` when `storage_key` + // resolved to a key above. + let Some(key) = storage_key else { + return self.response_error( + task, + ErrorCode::Internal { + detail: "spatial scan: sparse hit with no storage key".into(), + }, + ); + }; let (_, doc_mp) = crate::data::executor::scan_normalize::sparse_row_to_doc( - &doc_id, + &key, &raw, body_format.as_format_ref(), ); @@ -538,8 +558,9 @@ mod tests { }); let msgpack = nodedb_types::json_to_msgpack(&geojson).unwrap(); + let storage_key = nodedb_types::StorageKey::for_surrogate(surrogate); core.sparse - .put(0, tid, collection, &doc_id, &msgpack) + .put(0, tid, collection, &storage_key, &msgpack) .unwrap(); // Manually populate the R-tree and the doc-map. diff --git a/nodedb/src/data/executor/handlers/spatial_sync.rs b/nodedb/src/data/executor/handlers/spatial_sync.rs index 1eb923a61..dd8cd7bb0 100644 --- a/nodedb/src/data/executor/handlers/spatial_sync.rs +++ b/nodedb/src/data/executor/handlers/spatial_sync.rs @@ -140,11 +140,12 @@ impl CoreLoop { } }; + let storage_key = nodedb_types::StorageKey::for_surrogate(surrogate); if let Err(e) = self.sparse.put( task.request.database_id.as_u64(), tid, collection, - &doc_id, + &storage_key, &msgpack, ) { error!( @@ -247,10 +248,13 @@ impl CoreLoop { } } - if let Err(e) = - self.sparse - .delete(task.request.database_id.as_u64(), tid, collection, &doc_id) - { + let storage_key = nodedb_types::StorageKey::for_surrogate(surrogate); + if let Err(e) = self.sparse.delete( + task.request.database_id.as_u64(), + tid, + collection, + &storage_key, + ) { error!( core = self.core_id, %collection, diff --git a/nodedb/src/data/executor/handlers/text_search.rs b/nodedb/src/data/executor/handlers/text_search.rs index 8e061dc72..710afa17e 100644 --- a/nodedb/src/data/executor/handlers/text_search.rs +++ b/nodedb/src/data/executor/handlers/text_search.rs @@ -216,8 +216,9 @@ impl CoreLoop { break; } let hex_key = crate::engine::document::store::surrogate_to_doc_id(surrogate); + let storage_key = nodedb_types::StorageKey::for_surrogate(surrogate); let bytes_opt = match self.overlay_or_base_body(txn_id, &coll_key, &hex_key, || { - self.sparse.get(database_id, tid, collection, &hex_key) + self.sparse.get(database_id, tid, collection, &storage_key) }) { Ok(b) => b, Err(e) => { diff --git a/nodedb/src/data/executor/handlers/text_search_hybrid.rs b/nodedb/src/data/executor/handlers/text_search_hybrid.rs index 246f850b2..8702953cd 100644 --- a/nodedb/src/data/executor/handlers/text_search_hybrid.rs +++ b/nodedb/src/data/executor/handlers/text_search_hybrid.rs @@ -226,12 +226,16 @@ impl CoreLoop { if rls_filters.is_empty() { return true; } - match self.sparse.get( - task.request.database_id.as_u64(), - tid, - collection, - &f.document_id, - ) { + // `document_id` is a fused-result string several hops from any + // scan; a shape that fails to parse as a storage key is + // treated the same as a row the lookup below could not find. + let Some(key) = nodedb_types::StorageKey::parse(&f.document_id) else { + return false; + }; + match self + .sparse + .get(task.request.database_id.as_u64(), tid, collection, &key) + { Ok(Some(bytes)) => { let normalized = sparse_body_to_msgpack(&bytes, body_format.as_format_ref()); diff --git a/nodedb/src/data/executor/handlers/text_search_scan.rs b/nodedb/src/data/executor/handlers/text_search_scan.rs index fa143050d..d5bfbf7be 100644 --- a/nodedb/src/data/executor/handlers/text_search_scan.rs +++ b/nodedb/src/data/executor/handlers/text_search_scan.rs @@ -204,8 +204,11 @@ impl CoreLoop { collection, BM25_SCAN_MAX_HITS, ); - let mut docs = match scan_result { - Ok(d) => d, + // Rendered to text here: `merge_fts_rows_from_score_map` below and the + // per-row surrogate lookup both operate on the hex storage key as a + // string, out of this unit's typed scope. + let mut docs: Vec<(String, Vec)> = match scan_result { + Ok(d) => d.into_iter().map(|(k, v)| (k.to_string(), v)).collect(), Err(e) => { return self.response_error( task, diff --git a/nodedb/src/data/executor/handlers/text_search_triple.rs b/nodedb/src/data/executor/handlers/text_search_triple.rs index 5db1dceaf..b3b96365a 100644 --- a/nodedb/src/data/executor/handlers/text_search_triple.rs +++ b/nodedb/src/data/executor/handlers/text_search_triple.rs @@ -234,12 +234,16 @@ impl CoreLoop { if rls_filters.is_empty() { return true; } - match self.sparse.get( - task.request.database_id.as_u64(), - tid, - collection, - &f.document_id, - ) { + // `document_id` is a fused-result string several hops from any + // scan; a shape that fails to parse as a storage key is + // treated the same as a row the lookup below could not find. + let Some(key) = nodedb_types::StorageKey::parse(&f.document_id) else { + return false; + }; + match self + .sparse + .get(task.request.database_id.as_u64(), tid, collection, &key) + { Ok(Some(bytes)) => { let normalized = sparse_body_to_msgpack(&bytes, body_format.as_format_ref()); diff --git a/nodedb/src/data/executor/handlers/transaction/batch.rs b/nodedb/src/data/executor/handlers/transaction/batch.rs index 57566910a..34f58df62 100644 --- a/nodedb/src/data/executor/handlers/transaction/batch.rs +++ b/nodedb/src/data/executor/handlers/transaction/batch.rs @@ -503,7 +503,7 @@ mod tests { use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; use crate::data::executor::doc_format; use crate::data::executor::handlers::point::insert::PointInsertParams; - use crate::engine::document::store::{CollectionConfig, surrogate_to_doc_id}; + use crate::engine::document::store::CollectionConfig; use crate::types::{DatabaseId, TenantId}; use nodedb_physical::physical_plan::{DocumentOp, ResolvedSumTarget}; use nodedb_types::{QualifiedCollection, Surrogate}; @@ -562,7 +562,7 @@ mod tests { DB, TID, TARGET, - &surrogate_to_doc_id(T1), + &nodedb_types::StorageKey::for_surrogate(T1), &doc_format::encode_to_msgpack(&seed), ) .expect("seed target row"); @@ -581,7 +581,12 @@ mod tests { fn balance(core: &CoreLoop, surrogate: Surrogate) -> String { let stored = core .sparse - .get(DB, TID, TARGET, &surrogate_to_doc_id(surrogate)) + .get( + DB, + TID, + TARGET, + &nodedb_types::StorageKey::for_surrogate(surrogate), + ) .expect("read target") .expect("target row must exist"); doc_format::decode_document(&stored) diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs b/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs index 189592388..86e9971c6 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs @@ -742,7 +742,7 @@ mod tests { let txn = TxnId::new(41); let task = make_stage_task(txn); let surrogate = 5u32; - let row_key = surrogate_to_doc_id(Surrogate::new(surrogate)); + let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); // Seed a base row directly into the scan-visible sparse store. core.sparse @@ -750,14 +750,14 @@ mod tests { DatabaseId::DEFAULT.as_u64(), TID, "notes", - row_key.as_str(), + &row_key, &schemaless_body("alice"), ) .expect("seed base row"); let plan = PhysicalPlan::Document(DocumentOp::PointUpdate { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "notes"), - document_id: row_key.as_str().to_string(), + document_id: row_key.to_string(), surrogate: Surrogate::new(surrogate), pk_bytes: Vec::new(), updates: vec![("name".to_string(), literal_str("bob"))], @@ -798,13 +798,13 @@ mod tests { let task = make_stage_task(txn); for s in [1u32, 2u32] { - let row_key = surrogate_to_doc_id(Surrogate::new(s)); + let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(s)); core.sparse .put( DatabaseId::DEFAULT.as_u64(), TID, "notes", - row_key.as_str(), + &row_key, &schemaless_body("old"), ) .expect("seed base row"); @@ -857,13 +857,13 @@ mod tests { let task = make_stage_task(txn); for s in [1u32, 2u32] { - let row_key = surrogate_to_doc_id(Surrogate::new(s)); + let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(s)); core.sparse .put( DatabaseId::DEFAULT.as_u64(), TID, "notes", - row_key.as_str(), + &row_key, &schemaless_body("doomed"), ) .expect("seed base row"); @@ -951,10 +951,10 @@ mod tests { .expect("redo replay must succeed"); for s in [1u32, 2u32] { - let row_key = surrogate_to_doc_id(Surrogate::new(s)); + let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(s)); let stored = dst .sparse - .get(DatabaseId::DEFAULT.as_u64(), TID, "notes", row_key.as_str()) + .get(DatabaseId::DEFAULT.as_u64(), TID, "notes", &row_key) .expect("get") .expect("updated row must replay from resolve output"); assert_eq!( @@ -1384,7 +1384,7 @@ mod tests { let task = make_task(); let txn = TxnId::new(20); let surrogate = 7u32; - let row_key = surrogate_to_doc_id(Surrogate::new(surrogate)); + let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); src.txn_overlay_mut(txn).insert_put( coll_key("sdocs"), @@ -1411,7 +1411,7 @@ mod tests { let stored = dst .sparse - .get(DatabaseId::DEFAULT.as_u64(), TID, "sdocs", row_key.as_str()) + .get(DatabaseId::DEFAULT.as_u64(), TID, "sdocs", &row_key) .expect("get") .expect("strict document row must be restored from redo replay"); let decoded = strict_format::binary_tuple_to_value(&stored, &strict_schema()) @@ -1434,7 +1434,7 @@ mod tests { let task = make_task(); let txn = TxnId::new(21); let surrogate = 3u32; - let row_key = surrogate_to_doc_id(Surrogate::new(surrogate)); + let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); let body = schemaless_body("alice"); src.txn_overlay_mut(txn) @@ -1455,7 +1455,7 @@ mod tests { let stored = dst .sparse - .get(DatabaseId::DEFAULT.as_u64(), TID, "notes", row_key.as_str()) + .get(DatabaseId::DEFAULT.as_u64(), TID, "notes", &row_key) .expect("get") .expect("schemaless document row must replay"); assert_eq!(stored, body, "schemaless body round-trips verbatim"); @@ -1467,7 +1467,7 @@ mod tests { let task = make_task(); let txn = TxnId::new(22); let surrogate = 11u32; - let row_key = surrogate_to_doc_id(Surrogate::new(surrogate)); + let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); src.txn_overlay_mut(txn) .insert_tombstone(coll_key("notes"), surrogate, "gone"); @@ -1519,7 +1519,7 @@ mod tests { .expect("redo replay must succeed"); assert!( dst.sparse - .get(DatabaseId::DEFAULT.as_u64(), TID, "notes", row_key.as_str()) + .get(DatabaseId::DEFAULT.as_u64(), TID, "notes", &row_key) .expect("get") .is_some(), "row seeded" @@ -1534,7 +1534,7 @@ mod tests { .expect("redo replay must succeed"); assert!( dst.sparse - .get(DatabaseId::DEFAULT.as_u64(), TID, "notes", row_key.as_str()) + .get(DatabaseId::DEFAULT.as_u64(), TID, "notes", &row_key) .expect("get") .is_none(), "redo delete must remove the document row" @@ -1565,7 +1565,7 @@ mod tests { let task = make_task(); let txn = TxnId::new(23); let surrogate = 1u32; - let row_key = surrogate_to_doc_id(Surrogate::new(surrogate)); + let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); // Seed a base document row, then stage a DIFFERENT body for it. let seed = wrap_redo(&RedoRecord { @@ -1591,7 +1591,7 @@ mod tests { .expect("redo replay must succeed"); let before = core .sparse - .get(DatabaseId::DEFAULT.as_u64(), TID, "notes", row_key.as_str()) + .get(DatabaseId::DEFAULT.as_u64(), TID, "notes", &row_key) .expect("get"); assert_eq!(before.as_deref(), Some(schemaless_body("base").as_slice())); @@ -1608,7 +1608,7 @@ mod tests { // Base is untouched: resolve reads the overlay only, never writes base. let after = core .sparse - .get(DatabaseId::DEFAULT.as_u64(), TID, "notes", row_key.as_str()) + .get(DatabaseId::DEFAULT.as_u64(), TID, "notes", &row_key) .expect("get"); assert_eq!( after.as_deref(), @@ -1623,7 +1623,7 @@ mod tests { let task = make_task(); let txn = TxnId::new(24); let doc_surrogate = 5u32; - let doc_row_key = surrogate_to_doc_id(Surrogate::new(doc_surrogate)); + let doc_row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(doc_surrogate)); { let overlay = src.txn_overlay_mut(txn); @@ -1667,7 +1667,7 @@ mod tests { ); assert!( dst.sparse - .get(db, TID, "notes", doc_row_key.as_str()) + .get(db, TID, "notes", &doc_row_key) .expect("get") .is_some(), "document sub-record must replay" @@ -2045,7 +2045,7 @@ mod tests { let task = make_task(); let txn = TxnId::new(35); let doc_surrogate = 6u32; - let doc_row_key = surrogate_to_doc_id(Surrogate::new(doc_surrogate)); + let doc_row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(doc_surrogate)); { let overlay = src.txn_overlay_mut(txn); @@ -2091,7 +2091,7 @@ mod tests { assert!( dst_core .sparse - .get(db, TID, "notes", doc_row_key.as_str()) + .get(db, TID, "notes", &doc_row_key) .expect("get") .is_some(), "document sub-record must replay" @@ -2966,15 +2966,10 @@ mod tests { dst.spatial_doc_map.contains_key(&doc_map_key), "surrogate -> doc-id reverse map must be rebuilt" ); - let row_key = surrogate_to_doc_id(Surrogate::new(surrogate)); + let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); assert!( dst.sparse - .get( - DatabaseId::DEFAULT.as_u64(), - TID, - "places", - row_key.as_str() - ) + .get(DatabaseId::DEFAULT.as_u64(), TID, "places", &row_key) .expect("get") .is_some(), "sparse geometry document must be rebuilt by replay" @@ -3038,15 +3033,10 @@ mod tests { 0, "redo delete must remove the R-tree entry" ); - let row_key = surrogate_to_doc_id(Surrogate::new(surrogate)); + let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); assert!( dst.sparse - .get( - DatabaseId::DEFAULT.as_u64(), - TID, - "places", - row_key.as_str() - ) + .get(DatabaseId::DEFAULT.as_u64(), TID, "places", &row_key) .expect("get") .is_none(), "redo delete must remove the sparse geometry document" @@ -3107,15 +3097,10 @@ mod tests { !core.spatial_indexes.contains_key(&key), "resolve must not mutate the base spatial R-tree" ); - let row_key = surrogate_to_doc_id(Surrogate::new(surrogate)); + let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); assert!( core.sparse - .get( - DatabaseId::DEFAULT.as_u64(), - TID, - "places", - row_key.as_str() - ) + .get(DatabaseId::DEFAULT.as_u64(), TID, "places", &row_key) .expect("get") .is_none(), "resolve must not mutate the base sparse store" diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/constraint.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/constraint.rs index 482c35e55..c08307159 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/constraint.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/constraint.rs @@ -30,10 +30,9 @@ impl CoreLoop { /// True when the primary key is present under BASE ∪ OVERLAY semantics. pub(super) fn stage_pk_present( &self, - database_id: u64, - tid: u64, - collection: &str, + ctx: &StageCtx<'_>, row_key: &str, + storage_key: &crate::engine::document::store::StorageKey, bitemporal: bool, overlay: OverlayPk, ) -> crate::Result { @@ -49,14 +48,19 @@ impl CoreLoop { let exists = if bitemporal { self.sparse.versioned_exists_current_in_txn( &txn, - database_id, - tid, - collection, + ctx.database_id, + ctx.tid, + ctx.collection, row_key, )? } else { - self.sparse - .exists_in_txn(&txn, database_id, tid, collection, row_key)? + self.sparse.exists_in_txn( + &txn, + ctx.database_id, + ctx.tid, + ctx.collection, + storage_key, + )? }; drop(txn); Ok(exists) diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs index d0783719f..52bd68a80 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs @@ -197,7 +197,13 @@ impl CoreLoop { .map_err(|e| self.response_error(task, e))?; let mut rows: Vec<(String, Vec)> = Vec::with_capacity(matching_ids.len()); for doc_id in matching_ids { - if let Ok(Some(bytes)) = self.sparse.get(database_id, tid, collection, &doc_id) { + // `doc_id` is a bare string from a raw-table scan; a shape that + // fails to parse as a storage key contributes no row, same as a + // `get` miss right below. + let Some(key) = crate::engine::document::store::StorageKey::parse(&doc_id) else { + continue; + }; + if let Ok(Some(bytes)) = self.sparse.get(database_id, tid, collection, &key) { rows.push((doc_id, bytes)); } } diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_point_document.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_point_document.rs index 4ddfdf41f..0ba14e5ac 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_point_document.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_point_document.rs @@ -35,15 +35,15 @@ impl CoreLoop { value: &[u8], if_absent: bool, ) -> Response { - let row_key = StorageKey::for_surrogate(ctx.surrogate).to_string(); + let storage_key = StorageKey::for_surrogate(ctx.surrogate); + let row_key = storage_key.to_string(); let bitemporal = self.is_bitemporal(ctx.database_id, ctx.tid, ctx.collection); let overlay_pk = self.stage_overlay_pk(ctx); let present = match self.stage_pk_present( - ctx.database_id, - ctx.tid, - ctx.collection, + ctx, row_key.as_str(), + &storage_key, bitemporal, overlay_pk, ) { @@ -102,10 +102,9 @@ impl CoreLoop { let bitemporal = self.is_bitemporal(ctx.database_id, ctx.tid, ctx.collection); let overlay_pk = self.stage_overlay_pk(ctx); let present = match self.stage_pk_present( - ctx.database_id, - ctx.tid, - ctx.collection, + ctx, row_key.as_str(), + &storage_key, bitemporal, overlay_pk, ) { @@ -193,7 +192,7 @@ impl CoreLoop { ) } else { self.sparse - .get(ctx.database_id, ctx.tid, ctx.collection, row_key.as_str()) + .get(ctx.database_id, ctx.tid, ctx.collection, &storage_key) }; match read { Ok(Some(bytes)) => bytes, diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs index 022a5520e..e2ee76496 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs @@ -102,8 +102,8 @@ impl CoreLoop { Some(Staged::Tombstone) => Ok(None), None => { let bitemporal = self.is_bitemporal(ctx.database_id, ctx.tid, ctx.collection); - let row_key = surrogate_to_doc_id(ctx.surrogate); if bitemporal { + let row_key = surrogate_to_doc_id(ctx.surrogate); self.sparse.versioned_get_current( ctx.database_id, ctx.tid, @@ -111,8 +111,9 @@ impl CoreLoop { row_key.as_str(), ) } else { + let storage_key = nodedb_types::StorageKey::for_surrogate(ctx.surrogate); self.sparse - .get(ctx.database_id, ctx.tid, ctx.collection, row_key.as_str()) + .get(ctx.database_id, ctx.tid, ctx.collection, &storage_key) } } } diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs index e76c55c66..79d4601d0 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs @@ -83,6 +83,7 @@ impl CoreLoop { surrogate, user_roles, enforce: true, + resolved_targets: resolved_sum_targets, }, )?; diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs index f64291f75..5339951ea 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs @@ -70,6 +70,7 @@ impl CoreLoop { } = p; let row_key = crate::engine::document::store::surrogate_to_doc_id(surrogate); let row_key = row_key.as_str(); + let storage_key = crate::engine::document::store::StorageKey::for_surrogate(surrogate); let database_id = dummy_task.request.database_id.as_u64(); // Pre-read the plain-table value: it decides insert-vs-update for the @@ -94,7 +95,7 @@ impl CoreLoop { let folds_images = write_hook::folds_images(self, &hook_ctx); let prior_bytes = if chain.enabled() || folds_images { self.sparse - .get(database_id, tid, collection, row_key) + .get(database_id, tid, collection, &storage_key) .ok() .flatten() } else { @@ -138,7 +139,7 @@ impl CoreLoop { ) } else { self.sparse - .exists_in_txn(&txn, database_id, tid, collection, row_key) + .exists_in_txn(&txn, database_id, tid, collection, &storage_key) }; let exists = match exists_result { Ok(exists) => exists, @@ -188,6 +189,7 @@ impl CoreLoop { user_roles, enforce: true, wal_lsn: dummy_task.wal_lsn(), + resolved_targets: resolved_sum_targets, }, ) { Ok(o) => o, @@ -196,7 +198,14 @@ impl CoreLoop { // after we mutated the chain head and, on the later rejections, // after it had already cached the row. Reverse both so the // aborted op leaves no trace, then propagate the typed error. - chain_guard::abort_after_apply(self, &chain, database_id, tid, collection, row_key); + chain_guard::abort_after_apply( + self, + &chain, + database_id, + tid, + collection, + &storage_key, + ); return Err(e.into()); } }; @@ -206,7 +215,14 @@ impl CoreLoop { // and drops `txn` uncommitted, so a rejected insert never leaves a head // behind on disk either. if let Err(e) = chain.persist_head(self, &txn) { - chain_guard::abort_after_apply(self, &chain, database_id, tid, collection, row_key); + chain_guard::abort_after_apply( + self, + &chain, + database_id, + tid, + collection, + &storage_key, + ); return Err(ErrorCode::from(e)); } @@ -232,7 +248,14 @@ impl CoreLoop { let enforcement = match write_hook::run(self, &txn, &hook_ctx, images) { Ok(outcome) => outcome, Err(e) => { - chain_guard::abort_after_apply(self, &chain, database_id, tid, collection, row_key); + chain_guard::abort_after_apply( + self, + &chain, + database_id, + tid, + collection, + &storage_key, + ); return Err(ErrorCode::from(e)); } }; @@ -248,7 +271,14 @@ impl CoreLoop { // inside an explicit transaction. if let Err(e) = self.settle_balanced_entries(database_id, tid, collection, balanced_entries) { - chain_guard::abort_after_apply(self, &chain, database_id, tid, collection, row_key); + chain_guard::abort_after_apply( + self, + &chain, + database_id, + tid, + collection, + &storage_key, + ); return Err(ErrorCode::from(e)); } diff --git a/nodedb/src/data/executor/handlers/transaction/undo/document.rs b/nodedb/src/data/executor/handlers/transaction/undo/document.rs index 8a43c132e..de44923fe 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/document.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/document.rs @@ -51,6 +51,11 @@ impl CoreLoop { collection: &collection, document_id: &document_id, }; + // The entry carries the surrogate directly, so the storage + // key is minted from it rather than re-parsed out of + // `document_id`'s text. + let storage_key = + crate::engine::document::store::StorageKey::for_surrogate(surrogate); if let Some(sys_from_ms) = bitemporal_sys_from_ms { // Bitemporal op: never wrote the non-versioned table, so // physically remove the appended version row (+ its index @@ -61,12 +66,12 @@ impl CoreLoop { } else { let result = if let Some(old) = old_value { self.sparse - .put(database_id, tid, &collection, &document_id, &old) + .put(database_id, tid, &collection, &storage_key, &old) .map(|_| ()) .map_err(|e| e.to_string()) } else { self.sparse - .delete(database_id, tid, &collection, &document_id) + .delete(database_id, tid, &collection, &storage_key) .map(|_| ()) .map_err(|e| e.to_string()) }; @@ -119,7 +124,7 @@ impl CoreLoop { // a stale hit would otherwise resurrect a rolled-back put; the // worst case here is a cache miss. self.doc_cache - .invalidate(database_id, tid, &collection, &document_id); + .invalidate(database_id, tid, &collection, &storage_key); self.undo_chain_hash(database_id, tid, &collection, entry_index, chain_hash_prior)?; Ok(()) } @@ -140,11 +145,16 @@ impl CoreLoop { collection: &collection, document_id: &document_id, }; + // The entry carries the surrogate directly, so the storage + // key is minted from it rather than re-parsed out of + // `document_id`'s text. + let storage_key = + crate::engine::document::store::StorageKey::for_surrogate(surrogate); if let Some(sys_from_ms) = bitemporal_sys_from_ms { self.undo_bitemporal_write(ctx, sys_from_ms, &bitemporal_index_tuples)?; } else { self.sparse - .put(database_id, tid, &collection, &document_id, &old_value) + .put(database_id, tid, &collection, &storage_key, &old_value) .map(|_| ()) .map_err(|e| { error!( @@ -175,7 +185,7 @@ impl CoreLoop { // PutDocument branch): reversing a delete restores the row, so a // stale post-delete cache entry must not linger. self.doc_cache - .invalidate(database_id, tid, &collection, &document_id); + .invalidate(database_id, tid, &collection, &storage_key); self.undo_chain_hash(database_id, tid, &collection, entry_index, chain_hash_prior)?; Ok(()) } @@ -534,17 +544,26 @@ mod tests { assert!(!core.chain_hashes.contains_key(&key())); } + fn storage_key(surrogate: u32) -> crate::engine::document::store::StorageKey { + crate::engine::document::store::StorageKey::for_surrogate(nodedb_types::Surrogate::new( + surrogate, + )) + } + #[test] - fn plain_put_undo_backward_compatible() { + fn plain_put_undo_restores_or_removes_the_surrogate_row() { let dir = tempfile::tempdir().unwrap(); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - // Overwrite case: current holds "new", undo restores "old". - core.sparse.put(DB, TID, "c", "d1", b"new").unwrap(); + // Overwrite case: current holds "new", undo restores "old". `undo` + // reads the row through the entry's `surrogate`, not `document_id`'s + // text, so the seeded row and the entry share one surrogate. + let key1 = storage_key(1); + core.sparse.put(DB, TID, "c", &key1, b"new").unwrap(); let overwrite = UndoEntry::PutDocument { collection: "c".into(), - document_id: "d1".into(), - surrogate: nodedb_types::Surrogate::ZERO, + document_id: key1.to_string(), + surrogate: key1.surrogate(), old_value: Some(b"old".to_vec()), bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), @@ -554,16 +573,17 @@ mod tests { }; core.apply_undo_document(DB, TID, 0, overwrite).unwrap(); assert_eq!( - core.sparse.get(DB, TID, "c", "d1").unwrap(), + core.sparse.get(DB, TID, "c", &key1).unwrap(), Some(b"old".to_vec()) ); // Insert case: undo deletes the row. - core.sparse.put(DB, TID, "c", "d2", b"inserted").unwrap(); + let key2 = storage_key(2); + core.sparse.put(DB, TID, "c", &key2, b"inserted").unwrap(); let insert = UndoEntry::PutDocument { collection: "c".into(), - document_id: "d2".into(), - surrogate: nodedb_types::Surrogate::ZERO, + document_id: key2.to_string(), + surrogate: key2.surrogate(), old_value: None, bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), @@ -572,19 +592,20 @@ mod tests { chain_hash_prior: None, }; core.apply_undo_document(DB, TID, 0, insert).unwrap(); - assert!(core.sparse.get(DB, TID, "c", "d2").unwrap().is_none()); + assert!(core.sparse.get(DB, TID, "c", &key2).unwrap().is_none()); } #[test] - fn plain_delete_undo_backward_compatible() { + fn plain_delete_undo_reinserts_the_surrogate_row() { let dir = tempfile::tempdir().unwrap(); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); // Row was deleted by the forward op; undo re-inserts its prior value. + let key1 = storage_key(1); let entry = UndoEntry::DeleteDocument { collection: "c".into(), - document_id: "d1".into(), - surrogate: nodedb_types::Surrogate::ZERO, + document_id: key1.to_string(), + surrogate: key1.surrogate(), old_value: b"prior".to_vec(), bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), @@ -593,7 +614,7 @@ mod tests { }; core.apply_undo_document(DB, TID, 0, entry).unwrap(); assert_eq!( - core.sparse.get(DB, TID, "c", "d1").unwrap(), + core.sparse.get(DB, TID, "c", &key1).unwrap(), Some(b"prior".to_vec()) ); } diff --git a/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs b/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs index fe5165632..ec6d37cea 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs @@ -94,6 +94,7 @@ mod tests { core.apply_point_put( &txn, PointPutParams { + resolved_targets: &[], database_id: DB, tid: TID, collection: COLL, diff --git a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs index cfd6459cb..91ac69351 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs @@ -404,6 +404,7 @@ mod tests { core.apply_point_put( &txn, PointPutParams { + resolved_targets: &[], database_id: DB, tid: TID, collection: COLL, @@ -427,6 +428,7 @@ mod tests { core.apply_point_delete( &txn, PointDeleteParams { + resolved_targets: &[], database_id: DB, tid: TID, collection: COLL, diff --git a/nodedb/src/data/executor/handlers/truncate.rs b/nodedb/src/data/executor/handlers/truncate.rs index 53b98c37e..e285dc1a2 100644 --- a/nodedb/src/data/executor/handlers/truncate.rs +++ b/nodedb/src/data/executor/handlers/truncate.rs @@ -107,6 +107,13 @@ impl CoreLoop { // `execute_bulk_delete`'s `write_set` cascade. let mut write_set: Vec = Vec::new(); for doc_id in &all_ids { + // `doc_id` is a bare string from a raw-table scan several calls + // removed from `SparseEngine`'s typed scan methods. A shape that + // fails to parse as a storage key can hold no row in DOCUMENTS + // either way, so `delete_in_txn` below sees the same "nothing to + // remove" outcome a lookup miss would have produced. + let storage_key = crate::engine::document::store::StorageKey::parse(doc_id); + // One transaction per removed row, shared with the materialized-sum // delta that row owes — identical to `execute_bulk_delete`, so a // TRUNCATE and a `DELETE` with no predicate leave the same totals. @@ -114,11 +121,12 @@ impl CoreLoop { Ok(txn) => txn, Err(e) => return self.response_error(task, e), }; - let deleted_bytes = self - .sparse - .delete_in_txn(&row_txn, database_id, tid, collection, doc_id) - .ok() - .flatten(); + let deleted_bytes = storage_key.and_then(|key| { + self.sparse + .delete_in_txn(&row_txn, database_id, tid, collection, &key) + .ok() + .flatten() + }); let mut target_writes = Vec::new(); if let Some(bytes) = deleted_bytes.as_deref() { match write_hook::run( @@ -154,9 +162,9 @@ impl CoreLoop { write_set.extend(write_hook::target_write_set(&target_writes)); if let Some(deleted_bytes) = deleted_bytes.as_deref() { // doc_id is the hex-encoded surrogate (the redb storage key). - // Parse back to Surrogate for FTS removal. Non-hex keys - // (legacy non-surrogate docs) produce None and skip FTS. - if let Some(surrogate) = crate::engine::document::store::doc_id_to_surrogate(doc_id) + // `storage_key` already holds the parsed form. Non-hex keys + // (legacy non-surrogate docs) hold `None` and skip FTS. + if let Some(surrogate) = storage_key.map(|key| key.surrogate()) && let Err(e) = self.inverted.remove_document( database_id, crate::types::TenantId::new(tid), @@ -179,9 +187,7 @@ impl CoreLoop { // process (mirrors `execute_bulk_delete`'s vector cascade). if has_vectors { self.remove_document_vector_indexes(database_id, tid, collection, doc_id); - if let Some(surrogate) = - crate::engine::document::store::doc_id_to_surrogate(doc_id) - { + if let Some(surrogate) = storage_key.map(|key| key.surrogate()) { write_set.push(WriteSetEntry { surrogate: surrogate.as_u32(), is_delete: true, @@ -204,12 +210,14 @@ impl CoreLoop { { warn!(core = self.core_id, %doc_id, error = %e, "truncate: edge cascade failed"); } - self.doc_cache.invalidate( - task.request.database_id.as_u64(), - tid, - collection, - doc_id, - ); + if let Some(key) = storage_key { + self.doc_cache.invalidate( + task.request.database_id.as_u64(), + tid, + collection, + &key, + ); + } // Emit a delete event per removed row to the Event Plane, so // AFTER-DELETE triggers and CDC/change-stream consumers see // each row TRUNCATE removed — mirroring `execute_point_delete` @@ -227,7 +235,9 @@ impl CoreLoop { collection, deleted_bytes, ); - let identity = crate::engine::document::store::identity_of(doc_id); + let identity = storage_key.map(|key| key.to_identity()).unwrap_or_else(|| { + crate::engine::document::store::RowIdentity::from_user_key(doc_id.as_str()) + }); self.emit_document_delete_event( task, collection, diff --git a/nodedb/src/data/executor/handlers/update_from_join_write.rs b/nodedb/src/data/executor/handlers/update_from_join_write.rs index b10c2bf68..2dc76d8f3 100644 --- a/nodedb/src/data/executor/handlers/update_from_join_write.rs +++ b/nodedb/src/data/executor/handlers/update_from_join_write.rs @@ -59,6 +59,11 @@ impl CoreLoop { want_returning, } = ctx; let database_id = task.request.database_id.as_u64(); + let config_key = ( + crate::types::DatabaseId::new(database_id), + crate::types::TenantId::new(tid), + target_collection.to_string(), + ); let mut affected = 0u64; let mut write_set: Vec = Vec::new(); let mut returned_docs: Vec = if want_returning { @@ -76,6 +81,57 @@ impl CoreLoop { mut doc, } = row; + // `put_in_txn` addresses DOCUMENTS rows by `StorageKey` only, and + // the workspace carries no on-disk-format compatibility burden + // for a row shape that predates surrogate keying, so a row whose + // `doc_id` does not parse as one is refused rather than written + // through a raw string key. + let storage_key = match crate::engine::document::store::StorageKey::parse(&doc_id) { + Some(key) => key, + None => { + return Err(self.response_error( + task, + crate::Error::Storage { + engine: "document".into(), + detail: format!( + "UPDATE ... FROM target row '{doc_id}' in \ + '{target_collection}' has no surrogate storage key" + ), + }, + )); + } + }; + + // Period lock, both images — matching `execute_point_update`: a + // closed period must reject an edit to a row it already holds, + // and must reject an edit that assigns the period column into it. + if let Some(config) = self.doc_configs.get(&config_key) + && let Some(ref pl) = config.enforcement.period_lock + { + if let Err(e) = crate::data::executor::enforcement::period_lock::check_period_lock( + &self.sparse, + database_id, + tid, + target_collection, + &old_body, + pl, + resolved_sum_targets, + ) { + return Err(self.response_error(task, e)); + } + if let Err(e) = crate::data::executor::enforcement::period_lock::check_period_lock( + &self.sparse, + database_id, + tid, + target_collection, + &updated_bytes, + pl, + resolved_sum_targets, + ) { + return Err(self.response_error(task, e)); + } + } + // The row's body and the materialized-sum delta it owes share ONE // transaction. `ResolvedUpdateRow` already carries BOTH images — // `old_body` as stored and `body` as the post-image — so the fold @@ -89,7 +145,7 @@ impl CoreLoop { database_id, tid, target_collection, - &doc_id, + &storage_key, &updated_bytes, ); if stored.is_ok() { @@ -131,8 +187,13 @@ impl CoreLoop { // collection — this statement's redo describes only the rows of // `target_collection` it rewrote. write_set.extend(write_hook::target_write_set(&target_writes)); - self.doc_cache - .put(database_id, tid, target_collection, &doc_id, &updated_bytes); + self.doc_cache.put( + database_id, + tid, + target_collection, + &storage_key, + &updated_bytes, + ); // Emit an update event per affected row to the Event Plane, so // AFTER-UPDATE triggers and CDC/change-stream consumers see // each row `UPDATE ... FROM` touched — mirroring diff --git a/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs b/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs index 1cb534fab..00a51a679 100644 --- a/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs +++ b/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs @@ -118,7 +118,8 @@ impl CoreLoop { self.sparse .versioned_get_current(database_id, tid, collection, row_key) } else { - self.sparse.get(database_id, tid, collection, row_key) + let key = nodedb_types::StorageKey::for_surrogate(surrogate); + self.sparse.get(database_id, tid, collection, &key) }; match existing { @@ -234,7 +235,7 @@ mod tests { DB, TID, TARGET, - &surrogate_to_doc_id(T1), + &nodedb_types::StorageKey::for_surrogate(T1), &doc_format::encode_to_msgpack(&seed), ) .expect("seed target row"); @@ -253,7 +254,12 @@ mod tests { fn balance(core: &CoreLoop, surrogate: Surrogate) -> String { let stored = core .sparse - .get(DB, TID, TARGET, &surrogate_to_doc_id(surrogate)) + .get( + DB, + TID, + TARGET, + &nodedb_types::StorageKey::for_surrogate(surrogate), + ) .expect("read target") .expect("target row must exist"); doc_format::decode_document(&stored) diff --git a/nodedb/src/data/executor/handlers/upsert/exec/insert.rs b/nodedb/src/data/executor/handlers/upsert/exec/insert.rs index cf185d1fa..e5974c0d9 100644 --- a/nodedb/src/data/executor/handlers/upsert/exec/insert.rs +++ b/nodedb/src/data/executor/handlers/upsert/exec/insert.rs @@ -57,7 +57,8 @@ impl CoreLoop { strict_schema, } = ctx; - let row_identity = StorageKey::for_surrogate(surrogate).to_identity(); + let storage_key = StorageKey::for_surrogate(surrogate); + let row_identity = storage_key.to_identity(); let document_identity = RowIdentity::from_user_key(document_id); // Insert: document doesn't exist, create new (same as PointPut). @@ -112,11 +113,19 @@ impl CoreLoop { user_roles: &task.request.user_roles, enforce: true, wal_lsn: task.wal_lsn(), + resolved_targets: hook_ctx.resolved_targets, }, ) { Ok(p) => p, Err(e) => { - chain_guard::abort_after_apply(self, &chain, database_id, tid, collection, row_key); + chain_guard::abort_after_apply( + self, + &chain, + database_id, + tid, + collection, + &storage_key, + ); return self.response_error(task, e); } }; @@ -124,7 +133,14 @@ impl CoreLoop { // The advanced head lands in the SAME transaction as the row // whose hash it is. if let Err(e) = chain.persist_head(self, &txn) { - chain_guard::abort_after_apply(self, &chain, database_id, tid, collection, row_key); + chain_guard::abort_after_apply( + self, + &chain, + database_id, + tid, + collection, + &storage_key, + ); return self.response_error(task, e); } @@ -141,7 +157,14 @@ impl CoreLoop { ) { Ok(o) => o, Err(e) => { - chain_guard::abort_after_apply(self, &chain, database_id, tid, collection, row_key); + chain_guard::abort_after_apply( + self, + &chain, + database_id, + tid, + collection, + &storage_key, + ); return self.response_error(task, e); } }; @@ -152,7 +175,14 @@ impl CoreLoop { if let Err(e) = self.settle_balanced_entries(database_id, tid, collection, enforcement.balanced_entries) { - chain_guard::abort_after_apply(self, &chain, database_id, tid, collection, row_key); + chain_guard::abort_after_apply( + self, + &chain, + database_id, + tid, + collection, + &storage_key, + ); return self.response_error(task, e); } diff --git a/nodedb/src/data/executor/handlers/upsert/exec/overwrite.rs b/nodedb/src/data/executor/handlers/upsert/exec/overwrite.rs index f0c94236b..b4633d3dd 100644 --- a/nodedb/src/data/executor/handlers/upsert/exec/overwrite.rs +++ b/nodedb/src/data/executor/handlers/upsert/exec/overwrite.rs @@ -64,7 +64,8 @@ impl CoreLoop { current_bytes, } = ctx; - let row_identity = StorageKey::for_surrogate(surrogate).to_identity(); + let storage_key = StorageKey::for_surrogate(surrogate); + let row_identity = storage_key.to_identity(); let document_identity = RowIdentity::from_user_key(document_id); // Decode existing document to nodedb_types::Value. @@ -205,6 +206,7 @@ impl CoreLoop { user_roles: &task.request.user_roles, enforce: true, wal_lsn: task.wal_lsn(), + resolved_targets: hook_ctx.resolved_targets, }, ) { Ok(o) => o, @@ -214,7 +216,7 @@ impl CoreLoop { // that entry, which would then serve a body that never // committed. self.doc_cache - .invalidate(database_id, tid, collection, row_key); + .invalidate(database_id, tid, collection, &storage_key); return self.response_error(task, e); } }; @@ -235,7 +237,7 @@ impl CoreLoop { Ok(o) => o, Err(e) => { self.doc_cache - .invalidate(database_id, tid, collection, row_key); + .invalidate(database_id, tid, collection, &storage_key); return self.response_error(task, e); } }; @@ -248,7 +250,7 @@ impl CoreLoop { self.settle_balanced_entries(database_id, tid, collection, enforcement.balanced_entries) { self.doc_cache - .invalidate(database_id, tid, collection, row_key); + .invalidate(database_id, tid, collection, &storage_key); return self.response_error(task, e); } diff --git a/nodedb/src/data/executor/handlers/vector_search_exec.rs b/nodedb/src/data/executor/handlers/vector_search_exec.rs index d0f0aa513..5efb3192f 100644 --- a/nodedb/src/data/executor/handlers/vector_search_exec.rs +++ b/nodedb/src/data/executor/handlers/vector_search_exec.rs @@ -13,7 +13,6 @@ use super::vector_search_ann::{ResolvedAnnOptions, apply_ann_options, quantizati use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; use nodedb_types::Surrogate; /// Parameters for [`CoreLoop::search_ivf`]. @@ -55,8 +54,8 @@ impl CoreLoop { if !attach { return hit; } - let hex = surrogate_to_doc_id(Surrogate::new(hit.id)); - if let Ok(Some(bytes)) = self.sparse.get(database_id, tid, collection, &hex) { + let key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(hit.id)); + if let Ok(Some(bytes)) = self.sparse.get(database_id, tid, collection, &key) { let format = self.sparse_body_format( crate::types::DatabaseId::new(database_id), crate::types::TenantId::new(tid), diff --git a/nodedb/src/data/executor/handlers/vector_upsert.rs b/nodedb/src/data/executor/handlers/vector_upsert.rs index 2ff41c826..a201e59e8 100644 --- a/nodedb/src/data/executor/handlers/vector_upsert.rs +++ b/nodedb/src/data/executor/handlers/vector_upsert.rs @@ -24,7 +24,6 @@ use tracing::debug; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; /// Decode MessagePack payload bytes into `HashMap` and /// lower-case all field names so bitmap inserts agree with SELECT @@ -215,7 +214,7 @@ impl CoreLoop { // row scannable at all, so skipping it made such a row invisible to // `SELECT *` while every other path still counted it as stored. An // empty tagged map is the honest sidecar for "no non-vector columns". - let row_key = surrogate_to_doc_id(surrogate); + let storage_key = nodedb_types::StorageKey::for_surrogate(surrogate); let sidecar: std::borrow::Cow<'_, [u8]> = if payload.is_empty() { match zerompk::to_msgpack_vec(&HashMap::::new()) { Ok(bytes) => std::borrow::Cow::Owned(bytes), @@ -235,7 +234,7 @@ impl CoreLoop { task.request.database_id.as_u64(), tid, collection, - &row_key, + &storage_key, &sidecar, ) { // Roll back Steps 3 + 4 so the HNSW node and bitmap entries @@ -286,7 +285,7 @@ impl CoreLoop { task, spec, rls_filters, - &row_key, + &storage_key, &sidecar, ); } diff --git a/nodedb/src/data/executor/handlers/vector_write.rs b/nodedb/src/data/executor/handlers/vector_write.rs index bf1f0b767..315fe1125 100644 --- a/nodedb/src/data/executor/handlers/vector_write.rs +++ b/nodedb/src/data/executor/handlers/vector_write.rs @@ -11,7 +11,6 @@ use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::vector_upsert::decode_payload_lowercased; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; use crate::types::TenantId; use nodedb_types::DatabaseId; @@ -140,7 +139,7 @@ impl CoreLoop { .and_then(|c| c.get_surrogate(vector_id)); if let Some(surrogate) = surrogate_opt { - let row_key = surrogate_to_doc_id(surrogate); + let row_key = nodedb_types::StorageKey::for_surrogate(surrogate); let fields = match self .sparse diff --git a/nodedb/src/data/executor/handlers/write_batch.rs b/nodedb/src/data/executor/handlers/write_batch.rs index 25ab08868..0f95638cb 100644 --- a/nodedb/src/data/executor/handlers/write_batch.rs +++ b/nodedb/src/data/executor/handlers/write_batch.rs @@ -95,6 +95,7 @@ impl CoreLoop { collection, value, surrogate, + resolved_sum_targets, .. }) = task.plan() else { @@ -117,6 +118,7 @@ impl CoreLoop { user_roles: &task.request.user_roles, enforce: true, wal_lsn: task.wal_lsn(), + resolved_targets: resolved_sum_targets.as_slice(), }, ) .map_err(|e| { diff --git a/nodedb/src/data/executor/row_shape.rs b/nodedb/src/data/executor/row_shape.rs index 48f9d1e8b..16e89fa9d 100644 --- a/nodedb/src/data/executor/row_shape.rs +++ b/nodedb/src/data/executor/row_shape.rs @@ -67,25 +67,19 @@ pub(in crate::data::executor) fn sparse_body_to_msgpack<'a>( /// vector-primary sidecar stores the user's declared primary key, and its /// sparse key is the internal surrogate-hex, which must not displace it. /// -/// `id` is the row's storage key. When it is a minted key, the client-visible -/// identity is the surrogate's decimal string, not the hex storage key; a -/// non-minted key (a user-supplied or DDL-declared primary key) passes -/// through unchanged. Shared by the materializing scan and the streaming scan -/// so both paths produce byte-identical output. -/// -/// `id` comes off a store iterator as a plain `&str`, so this is one of the -/// few sites that reinterprets that shape: a value that parses as 8 lowercase -/// hex characters is a minted storage key, everything else is a legacy or -/// user key taken verbatim. +/// `key` is the row's storage key. The client-visible identity is its +/// surrogate's decimal string, per [`StorageKey::to_identity`]. Shared by the +/// materializing scan and the streaming scan so both paths produce +/// byte-identical output. pub(in crate::data::executor) fn sparse_row_to_doc( - id: &str, + key: &nodedb_types::StorageKey, raw: &[u8], format: SparseBodyFormatRef<'_>, ) -> (String, Vec) { - let identity = crate::engine::document::store::identity_of(id); + let identity = key.to_identity(); let mp = sparse_body_to_msgpack(raw, format); let mp = msgpack_scan::inject_str_field(&mp, "id", identity.as_str()); - (id.to_string(), mp) + (key.to_string(), mp) } /// Convert a single row from a `DecodedColumn` to a `nodedb_types::value::Value`. diff --git a/nodedb/src/data/executor/scan_normalize.rs b/nodedb/src/data/executor/scan_normalize.rs index 1ee4d66c6..4fc173ed9 100644 --- a/nodedb/src/data/executor/scan_normalize.rs +++ b/nodedb/src/data/executor/scan_normalize.rs @@ -423,6 +423,12 @@ pub(in crate::data::executor) use super::row_shape::{ #[cfg(test)] mod tests { + /// Storage key for a small test surrogate, keyed by a mnemonic letter so + /// call sites below read the same as the doc-id literals they replace. + fn key(surrogate: u32) -> nodedb_types::StorageKey { + nodedb_types::StorageKey::for_surrogate(nodedb_types::Surrogate::new(surrogate)) + } + /// Verify that `scan_collection_for_each` visits exactly the same /// `(id, bytes)` set as `scan_collection` for a sparse/document collection. /// @@ -445,9 +451,9 @@ mod tests { let raw_a = b"{\"x\":1}"; let raw_b = b"{\"x\":2}"; let raw_c = b"{\"x\":3}"; - core.sparse.put(0, tid, coll, "a", raw_a).unwrap(); - core.sparse.put(0, tid, coll, "b", raw_b).unwrap(); - core.sparse.put(0, tid, coll, "c", raw_c).unwrap(); + core.sparse.put(0, tid, coll, &key(1), raw_a).unwrap(); + core.sparse.put(0, tid, coll, &key(2), raw_b).unwrap(); + core.sparse.put(0, tid, coll, &key(3), raw_c).unwrap(); // Collect via `scan_collection` (the reference output). let mut expected = core.scan_collection(0, tid, coll, usize::MAX).unwrap(); @@ -567,10 +573,18 @@ mod tests { // Insert in non-alphabetical order so insertion order != sorted order. // If either scan path sorts internally the assertion will catch the divergence. - core.sparse.put(0, tid, coll, "d", b"{\"v\":4}").unwrap(); - core.sparse.put(0, tid, coll, "a", b"{\"v\":1}").unwrap(); - core.sparse.put(0, tid, coll, "c", b"{\"v\":3}").unwrap(); - core.sparse.put(0, tid, coll, "b", b"{\"v\":2}").unwrap(); + core.sparse + .put(0, tid, coll, &key(4), b"{\"v\":4}") + .unwrap(); + core.sparse + .put(0, tid, coll, &key(1), b"{\"v\":1}") + .unwrap(); + core.sparse + .put(0, tid, coll, &key(3), b"{\"v\":3}") + .unwrap(); + core.sparse + .put(0, tid, coll, &key(2), b"{\"v\":2}") + .unwrap(); // Reference output — NOT sorted. let expected = core.scan_collection(0, tid, coll, usize::MAX).unwrap(); @@ -718,8 +732,12 @@ mod tests { let tid: u64 = 1; let coll = "err_test"; - core.sparse.put(0, tid, coll, "a", b"{\"v\":1}").unwrap(); - core.sparse.put(0, tid, coll, "b", b"{\"v\":2}").unwrap(); + core.sparse + .put(0, tid, coll, &key(1), b"{\"v\":1}") + .unwrap(); + core.sparse + .put(0, tid, coll, &key(2), b"{\"v\":2}") + .unwrap(); let mut calls = 0usize; let result = core.scan_collection_for_each(0, tid, coll, |_id, _bytes| { diff --git a/nodedb/src/data/executor/scan_versioned.rs b/nodedb/src/data/executor/scan_versioned.rs index 79bb552a4..ab0a3be85 100644 --- a/nodedb/src/data/executor/scan_versioned.rs +++ b/nodedb/src/data/executor/scan_versioned.rs @@ -47,7 +47,16 @@ impl CoreLoop { let mut normalized = Vec::with_capacity(docs.len()); for (id, raw) in docs { - normalized.push(sparse_row_to_doc(&id, &raw, format.as_format_ref())); + // The versioned table keys every row by the same surrogate hex + // as the plain table; a shape that fails to parse is a violated + // storage invariant, not a legacy row to skip. + let key = nodedb_types::StorageKey::parse(&id).ok_or_else(|| crate::Error::Storage { + engine: "sparse".into(), + detail: format!( + "collection '{collection}' has a versioned row whose key is not a valid storage key: '{id}'" + ), + })?; + normalized.push(sparse_row_to_doc(&key, &raw, format.as_format_ref())); } Ok(normalized) } diff --git a/nodedb/src/data/executor/wal_replay/crdt.rs b/nodedb/src/data/executor/wal_replay/crdt.rs index ce47bdf96..58cc10014 100644 --- a/nodedb/src/data/executor/wal_replay/crdt.rs +++ b/nodedb/src/data/executor/wal_replay/crdt.rs @@ -667,8 +667,9 @@ mod crdt_replay_tests { Some(&LoroValue::String("retry".into())), "stale fenced record must be a no-op while matching retry applies" ); - let sparse_key = - crate::engine::document::store::surrogate_to_doc_id(nodedb_types::Surrogate::new(1)); + let sparse_key = crate::engine::document::store::StorageKey::for_surrogate( + nodedb_types::Surrogate::new(1), + ); assert!( h.core .sparse diff --git a/nodedb/src/data/executor/wal_replay_redo_document.rs b/nodedb/src/data/executor/wal_replay_redo_document.rs index 93db657fc..841db3660 100644 --- a/nodedb/src/data/executor/wal_replay_redo_document.rs +++ b/nodedb/src/data/executor/wal_replay_redo_document.rs @@ -264,6 +264,7 @@ impl CoreLoop { index_text: true, user_roles: &[], enforce: false, + resolved_targets: &[], wal_lsn: (record_lsn != 0).then(|| crate::types::Lsn::new(record_lsn)), }, ) { @@ -330,6 +331,7 @@ impl CoreLoop { surrogate, user_roles: &[], enforce: false, + resolved_targets: &[], }, ) { Ok(outcome) => match txn.commit() { @@ -485,12 +487,8 @@ mod tests { ) .expect("redo replay must succeed"); - let row_key = surrogate_to_doc_id(Surrogate::new(surrogate)); - let stored = h - .core - .sparse - .get(0, 7, "notes", row_key.as_str()) - .expect("get"); + let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); + let stored = h.core.sparse.get(0, 7, "notes", &row_key).expect("get"); assert!( stored.is_some(), "document row must be restored from redo replay" @@ -520,12 +518,8 @@ mod tests { ) .expect("redo replay must succeed"); - let row_key = surrogate_to_doc_id(Surrogate::new(surrogate)); - let stored = h - .core - .sparse - .get(0, 7, "notes", row_key.as_str()) - .expect("get"); + let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); + let stored = h.core.sparse.get(0, 7, "notes", &row_key).expect("get"); assert!(stored.is_none(), "redo delete must remove the document row"); } @@ -547,7 +541,7 @@ mod tests { 0, 7, "notes", - surrogate_to_doc_id(Surrogate::new(surrogate)).as_str(), + &nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)), ) .expect("get"); h.core @@ -560,7 +554,7 @@ mod tests { 0, 7, "notes", - surrogate_to_doc_id(Surrogate::new(surrogate)).as_str(), + &nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)), ) .expect("get"); assert_eq!( @@ -831,11 +825,11 @@ mod tests { ) .expect("redo replay must succeed"); - let row_key = surrogate_to_doc_id(Surrogate::new(doc_surrogate)); + let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(doc_surrogate)); assert!( h.core .sparse - .get(0, 7, "notes", row_key.as_str()) + .get(0, 7, "notes", &row_key) .expect("get") .is_some(), "document sub-record must be replayed" diff --git a/nodedb/src/data/executor/wal_replay_spatial.rs b/nodedb/src/data/executor/wal_replay_spatial.rs index ed0f33b5f..6d750aa5a 100644 --- a/nodedb/src/data/executor/wal_replay_spatial.rs +++ b/nodedb/src/data/executor/wal_replay_spatial.rs @@ -374,6 +374,10 @@ mod tests { format!("{SURROGATE:08x}") } + fn storage_key() -> nodedb_types::StorageKey { + nodedb_types::StorageKey::for_surrogate(nodedb_types::Surrogate::new(SURROGATE)) + } + fn point(x: f64, y: f64) -> Geometry { Geometry::Point { coordinates: [x, y], @@ -452,7 +456,7 @@ mod tests { .cloned(); let body = core .sparse - .get(DB, TENANT, COLLECTION, &doc_id()) + .get(DB, TENANT, COLLECTION, &storage_key()) .expect("sparse read") .map(|bytes| nodedb_types::value_from_msgpack(&bytes).expect("decode body")); (entries, mapped, body) diff --git a/nodedb/src/diag/context/mod.rs b/nodedb/src/diag/context/mod.rs index 276849977..075ef8d17 100644 --- a/nodedb/src/diag/context/mod.rs +++ b/nodedb/src/diag/context/mod.rs @@ -34,6 +34,6 @@ pub(in crate::diag) use recovery::{ReplayRecordUnapplied, WalArchivalFailedTrunc pub(in crate::diag) use retention::RetentionAutowireOrphaned; pub(in crate::diag) use vector::VectorIndexNotApplied; pub(in crate::diag) use write_path::{ - BatchInsertWithoutSurrogates, FtsIndexUpdateFailed, StrictRowUndecodable, - WriteAckedWithoutDurability, + BatchInsertWithoutSurrogates, FtsIndexUpdateFailed, OrphanedIndexEntryAfterDelete, + StrictRowUndecodable, WriteAckedWithoutDurability, }; diff --git a/nodedb/src/diag/context/write_path.rs b/nodedb/src/diag/context/write_path.rs index 15edcc6d5..1cf3773eb 100644 --- a/nodedb/src/diag/context/write_path.rs +++ b/nodedb/src/diag/context/write_path.rs @@ -128,6 +128,57 @@ impl DomainContext for BatchInsertWithoutSurrogates<'_> { } } +/// A committed DELETE's post-commit cascade failed to remove one row's +/// entry from a secondary structure. The row's own transaction already +/// committed, so this failure cannot roll it back — the row is gone from +/// the primary store, but its entry survives in the named index. +pub(in crate::diag) struct OrphanedIndexEntryAfterDelete<'a> { + /// Collection the deleted row belonged to. + pub collection: &'a str, + /// Which cascaded structure the entry was left behind in + /// (`"inverted"`, `"secondary"`, `"graph_edge"`). + pub index_kind: &'static str, + /// Stable class of the failure, as the index layer described it. + pub error_class: &'a str, +} + +impl DomainContext for OrphanedIndexEntryAfterDelete<'_> { + fn domain_kind(&self) -> &'static str { + "nodedb.orphaned_index_entry_after_delete" + } + + fn grouping_key(&self) -> String { + // Collection + index kind + error class name the bug; the deleted + // row's id is the occurrence, or a bulk delete over many rows would + // file one report per row instead of one growing report. + format!( + "collection={};index={};cause={}", + self.collection, self.index_kind, self.error_class + ) + } + + fn to_json(&self) -> Value { + json!({ + "collection": self.collection, + "index_kind": self.index_kind, + "error_class": self.error_class, + "why_fatal": "the row's DELETE already committed by the time this cascade \ + step runs, so the failure cannot be rolled back — the row is \ + gone from the primary store, but the named index still carries \ + an entry that points at nothing. A stale inverted-index posting \ + scores a deleted row in full-text search, a stale secondary-index \ + entry returns a document that no longer exists, and a stale graph \ + edge traverses to a removed node", + "operator_action": "read the error class: a transient cause (redb contention, \ + a full disk) clears on its own, and the orphaned entry is \ + pruned the next time the collection's index is rebuilt or \ + compacted; a structural cause (corrupt or type-mismatched \ + index table) will orphan every subsequent delete on this \ + collection until the index is rebuilt", + }) + } +} + /// A stored Binary Tuple that does not decode against its collection's /// strict schema. The bytes on disk are wrong, so the statement that read /// them is refused rather than applied over a partial row set. diff --git a/nodedb/src/diag/mod.rs b/nodedb/src/diag/mod.rs index ce7dc1332..fded5b163 100644 --- a/nodedb/src/diag/mod.rs +++ b/nodedb/src/diag/mod.rs @@ -13,9 +13,10 @@ pub use recording::{ batch_insert_without_surrogates, calvin_completion_timeout, catalog_apply_orphan_row, collection_purge_row_missing, consumer_group_offsets_retained, data_plane_response_lost, data_plane_responses_lost, entry_kind, fts_index_update_failed, history_compaction_not_applied, - ilp_invalid_utf8_drop, ilp_line_read_drop, metadata_apply_wedged, quota_row_invalid, - quota_row_undecodable, quota_row_write_failed, quota_scope_purge_incomplete, - quota_scope_replay_aborted, replay_record_unapplied, retention_autowire_orphaned, - scope_quota_not_installed, strict_row_undecodable, synonym_group_not_applied, - vector_index_not_applied, wal_archival_failed_truncation_held, write_acked_without_durability, + ilp_invalid_utf8_drop, ilp_line_read_drop, metadata_apply_wedged, + orphaned_index_entry_after_delete, quota_row_invalid, quota_row_undecodable, + quota_row_write_failed, quota_scope_purge_incomplete, quota_scope_replay_aborted, + replay_record_unapplied, retention_autowire_orphaned, scope_quota_not_installed, + strict_row_undecodable, synonym_group_not_applied, vector_index_not_applied, + wal_archival_failed_truncation_held, write_acked_without_durability, }; diff --git a/nodedb/src/diag/recording/mod.rs b/nodedb/src/diag/recording/mod.rs index 5f91b032b..3e39b978c 100644 --- a/nodedb/src/diag/recording/mod.rs +++ b/nodedb/src/diag/recording/mod.rs @@ -32,8 +32,9 @@ pub use quota::{ quota_scope_replay_aborted, scope_quota_not_installed, }; pub use recovery::{ - batch_insert_without_surrogates, fts_index_update_failed, replay_record_unapplied, - strict_row_undecodable, wal_archival_failed_truncation_held, write_acked_without_durability, + batch_insert_without_surrogates, fts_index_update_failed, orphaned_index_entry_after_delete, + replay_record_unapplied, strict_row_undecodable, wal_archival_failed_truncation_held, + write_acked_without_durability, }; pub use retention::retention_autowire_orphaned; pub use shared::entry_kind; diff --git a/nodedb/src/diag/recording/recovery.rs b/nodedb/src/diag/recording/recovery.rs index 99e1558c9..cecc91a32 100644 --- a/nodedb/src/diag/recording/recovery.rs +++ b/nodedb/src/diag/recording/recovery.rs @@ -68,6 +68,31 @@ pub fn fts_index_update_failed(err: &crate::Error, collection: &str, surrogate: .emit(); } +/// Report an index entry a committed DELETE's cascade failed to remove. +/// Called from the bulk-delete cascade's warn sites, after the row's own +/// transaction has already committed — `index_kind` names which cascaded +/// structure (`"inverted"`, `"secondary"`, `"graph_edge"`) still carries it. +pub fn orphaned_index_entry_after_delete( + err: &crate::Error, + collection: &str, + index_kind: &'static str, +) { + let class = error_class(err); + let ctx = context::OrphanedIndexEntryAfterDelete { + collection, + index_kind, + error_class: &class, + }; + let _ = Capture::new( + EventKind::InvariantViolation, + "committed delete's cascade left an index entry behind", + ) + .error_chain(error_chain_of(err)) + .domain(&ctx) + .with_backtrace() + .emit(); +} + /// Report a document batch insert refused because its rows carry no /// surrogates. Called from the batch-insert handler's parallel-length guard; /// the actual defect is in whatever produced the mismatched plan. diff --git a/nodedb/src/engine/document/store/engine/batch.rs b/nodedb/src/engine/document/store/engine/batch.rs index 03b04fdb6..de597038a 100644 --- a/nodedb/src/engine/document/store/engine/batch.rs +++ b/nodedb/src/engine/document/store/engine/batch.rs @@ -115,6 +115,9 @@ impl<'a> DocumentEngine<'a> { #[cfg(test)] mod tests { + use nodedb_types::Surrogate; + + use crate::engine::document::store::StorageKey; use crate::engine::document::store::extract::json_to_msgpack; use super::*; @@ -125,6 +128,10 @@ mod tests { (engine, dir) } + fn key(surrogate: u32) -> StorageKey { + StorageKey::for_surrogate(Surrogate::new(surrogate)) + } + #[test] fn secondary_index_extraction() { let (sparse, _dir) = make_engine(); @@ -135,14 +142,14 @@ mod tests { doc_engine .put( "users", - "u1", + &key(1), &serde_json::json!({"name": "Alice", "email": "alice@example.com"}), ) .unwrap(); doc_engine .put( "users", - "u2", + &key(2), &serde_json::json!({"name": "Bob", "email": "bob@example.com"}), ) .unwrap(); @@ -150,7 +157,7 @@ mod tests { let results = doc_engine .index_lookup("users", "$.email", "alice@example.com", false) .unwrap(); - assert_eq!(results, vec!["u1"]); + assert_eq!(results, vec![key(1).to_string()]); } #[test] @@ -163,7 +170,7 @@ mod tests { doc_engine .put( "users", - "u1", + &key(1), &serde_json::json!({"name": "Alice", "tags": ["admin", "editor"]}), ) .unwrap(); @@ -171,12 +178,12 @@ mod tests { let results = doc_engine .index_lookup("users", "$.tags", "admin", false) .unwrap(); - assert_eq!(results, vec!["u1"]); + assert_eq!(results, vec![key(1).to_string()]); let results = doc_engine .index_lookup("users", "$.tags", "editor", false) .unwrap(); - assert_eq!(results, vec!["u1"]); + assert_eq!(results, vec![key(1).to_string()]); } #[test] @@ -189,7 +196,7 @@ mod tests { doc_engine .put( "docs", - "d1", + &key(1), &serde_json::json!({"title": "Hello", "metadata": {"lang": "en"}}), ) .unwrap(); @@ -197,7 +204,7 @@ mod tests { let results = doc_engine .index_lookup("docs", "$.metadata.lang", "en", false) .unwrap(); - assert_eq!(results, vec!["d1"]); + assert_eq!(results, vec![key(1).to_string()]); } #[test] @@ -212,11 +219,11 @@ mod tests { let mut buf = Vec::new(); rmpv::encode::write_value(&mut buf, &rmpv_val).unwrap(); - doc_engine.put_raw("items", "i1", &buf).unwrap(); + doc_engine.put_raw("items", &key(1), &buf).unwrap(); let results = doc_engine .index_lookup("items", "$.category", "tools", false) .unwrap(); - assert_eq!(results, vec!["i1"]); + assert_eq!(results, vec![key(1).to_string()]); } } diff --git a/nodedb/src/engine/document/store/engine/delete.rs b/nodedb/src/engine/document/store/engine/delete.rs index de9448334..611c68c07 100644 --- a/nodedb/src/engine/document/store/engine/delete.rs +++ b/nodedb/src/engine/document/store/engine/delete.rs @@ -7,16 +7,20 @@ //! row is removed in place. use super::batch::{DocumentEngine, wall_now_ms}; +use crate::engine::document::store::StorageKey; use crate::engine::document::store::extract::extract_index_values_rmpv; impl<'a> DocumentEngine<'a> { - pub fn delete(&self, collection: &str, doc_id: &str) -> crate::Result { + pub fn delete(&self, collection: &str, doc_id: &StorageKey) -> crate::Result { + // The versioned table and the INDEXES table both take the storage + // key as text; rendered once here at the boundary. + let doc_id_str = doc_id.to_string(); if self.is_bitemporal(collection) { let prior_body = self.sparse.versioned_get_current( self.database_id, self.tenant_id, collection, - doc_id, + &doc_id_str, )?; let Some(body) = prior_body else { return Ok(false); @@ -26,7 +30,7 @@ impl<'a> DocumentEngine<'a> { self.database_id, self.tenant_id, collection, - doc_id, + &doc_id_str, sys_from, )?; if let Some(config) = self.configs.get(collection) @@ -43,7 +47,7 @@ impl<'a> DocumentEngine<'a> { coll: collection, field: &index_path.path, value: &v, - doc_id, + doc_id: &doc_id_str, sys_from_ms: sys_from, }, )?; @@ -56,7 +60,7 @@ impl<'a> DocumentEngine<'a> { self.database_id, self.tenant_id, collection, - doc_id, + &doc_id_str, )?; Ok(self .sparse @@ -67,6 +71,8 @@ impl<'a> DocumentEngine<'a> { #[cfg(test)] mod tests { + use nodedb_types::Surrogate; + use crate::engine::sparse::btree::SparseEngine; use super::*; @@ -77,14 +83,18 @@ mod tests { (engine, dir) } + fn key(surrogate: u32) -> StorageKey { + StorageKey::for_surrogate(Surrogate::new(surrogate)) + } + #[test] fn delete_document() { let (sparse, _dir) = make_engine(); let doc_engine = DocumentEngine::new(&sparse, 0, 1); let doc = serde_json::json!({"name": "Bob"}); - doc_engine.put("users", "u1", &doc).unwrap(); - assert!(doc_engine.delete("users", "u1").unwrap()); - assert!(doc_engine.get("users", "u1").unwrap().is_none()); + doc_engine.put("users", &key(1), &doc).unwrap(); + assert!(doc_engine.delete("users", &key(1)).unwrap()); + assert!(doc_engine.get("users", &key(1)).unwrap().is_none()); } } diff --git a/nodedb/src/engine/document/store/engine/get.rs b/nodedb/src/engine/document/store/engine/get.rs index 7d1b1eaa2..0ad882227 100644 --- a/nodedb/src/engine/document/store/engine/get.rs +++ b/nodedb/src/engine/document/store/engine/get.rs @@ -3,17 +3,22 @@ //! Document read paths. use super::batch::DocumentEngine; +use crate::engine::document::store::StorageKey; use crate::engine::document::store::extract::rmpv_to_json; impl<'a> DocumentEngine<'a> { /// Get a document and deserialize from MessagePack to JSON. - pub fn get(&self, collection: &str, doc_id: &str) -> crate::Result> { + pub fn get( + &self, + collection: &str, + doc_id: &StorageKey, + ) -> crate::Result> { let bytes_opt = if self.is_bitemporal(collection) { self.sparse.versioned_get_current( self.database_id, self.tenant_id, collection, - doc_id, + &doc_id.to_string(), )? } else { self.sparse @@ -34,10 +39,14 @@ impl<'a> DocumentEngine<'a> { } /// Get raw MessagePack bytes (zero-copy path for DataFusion UDFs). - pub fn get_raw(&self, collection: &str, doc_id: &str) -> crate::Result>> { + pub fn get_raw(&self, collection: &str, doc_id: &StorageKey) -> crate::Result>> { if self.is_bitemporal(collection) { - self.sparse - .versioned_get_current(self.database_id, self.tenant_id, collection, doc_id) + self.sparse.versioned_get_current( + self.database_id, + self.tenant_id, + collection, + &doc_id.to_string(), + ) } else { self.sparse .get(self.database_id, self.tenant_id, collection, doc_id) @@ -47,6 +56,8 @@ impl<'a> DocumentEngine<'a> { #[cfg(test)] mod tests { + use nodedb_types::Surrogate; + use crate::engine::sparse::btree::SparseEngine; use super::*; @@ -57,11 +68,15 @@ mod tests { (engine, dir) } + fn key(surrogate: u32) -> StorageKey { + StorageKey::for_surrogate(Surrogate::new(surrogate)) + } + #[test] fn get_nonexistent_returns_none() { let (sparse, _dir) = make_engine(); let doc_engine = DocumentEngine::new(&sparse, 0, 1); - assert!(doc_engine.get("users", "missing").unwrap().is_none()); + assert!(doc_engine.get("users", &key(1)).unwrap().is_none()); } #[test] @@ -70,14 +85,14 @@ mod tests { let doc_engine = DocumentEngine::new(&sparse, 0, 1); doc_engine - .put("users", "id1", &serde_json::json!({"type": "user"})) + .put("users", &key(1), &serde_json::json!({"type": "user"})) .unwrap(); doc_engine - .put("orders", "id1", &serde_json::json!({"type": "order"})) + .put("orders", &key(1), &serde_json::json!({"type": "order"})) .unwrap(); - let user = doc_engine.get("users", "id1").unwrap().unwrap(); - let order = doc_engine.get("orders", "id1").unwrap().unwrap(); + let user = doc_engine.get("users", &key(1)).unwrap().unwrap(); + let order = doc_engine.get("orders", &key(1)).unwrap().unwrap(); assert_eq!(user["type"], "user"); assert_eq!(order["type"], "order"); } diff --git a/nodedb/src/engine/document/store/engine/put.rs b/nodedb/src/engine/document/store/engine/put.rs index 522f5b2d9..3e4b7c1c8 100644 --- a/nodedb/src/engine/document/store/engine/put.rs +++ b/nodedb/src/engine/document/store/engine/put.rs @@ -3,6 +3,7 @@ //! Document write paths: JSON and raw-MessagePack entry points. use super::batch::{DocumentEngine, wall_now_ms}; +use crate::engine::document::store::StorageKey; use crate::engine::document::store::extract::{ extract_index_values_rmpv, json_to_msgpack, rmpv_to_json, }; @@ -12,7 +13,7 @@ impl<'a> DocumentEngine<'a> { pub fn put( &self, collection: &str, - doc_id: &str, + doc_id: &StorageKey, document: &serde_json::Value, ) -> crate::Result<()> { let msgpack = json_to_msgpack(document); @@ -34,10 +35,13 @@ impl<'a> DocumentEngine<'a> { pub fn put_raw( &self, collection: &str, - doc_id: &str, + doc_id: &StorageKey, msgpack_bytes: &[u8], ) -> crate::Result<()> { let bitemporal = self.is_bitemporal(collection); + // The versioned table and the INDEXES table both take the storage + // key as text; rendered once here at the boundary. + let doc_id_str = doc_id.to_string(); if bitemporal { let sys_from = wall_now_ms(); @@ -46,7 +50,7 @@ impl<'a> DocumentEngine<'a> { database_id: self.database_id, tenant: self.tenant_id, coll: collection, - doc_id, + doc_id: &doc_id_str, sys_from_ms: sys_from, valid_from_ms: i64::MIN, valid_until_ms: i64::MAX, @@ -78,7 +82,7 @@ impl<'a> DocumentEngine<'a> { coll: collection, field: &index_path.path, value: &v, - doc_id, + doc_id: &doc_id_str, sys_from_ms: sys_from, }, )?; @@ -89,7 +93,7 @@ impl<'a> DocumentEngine<'a> { collection, &index_path.path, &v, - doc_id, + &doc_id_str, )?; } } @@ -103,6 +107,8 @@ impl<'a> DocumentEngine<'a> { #[cfg(test)] mod tests { + use nodedb_types::Surrogate; + use crate::engine::sparse::btree::SparseEngine; use super::*; @@ -113,6 +119,10 @@ mod tests { (engine, dir) } + fn key(surrogate: u32) -> StorageKey { + StorageKey::for_surrogate(Surrogate::new(surrogate)) + } + #[test] fn put_and_get_document() { let (sparse, _dir) = make_engine(); @@ -124,8 +134,8 @@ mod tests { "age": 30 }); - doc_engine.put("users", "u1", &doc).unwrap(); - let retrieved = doc_engine.get("users", "u1").unwrap().unwrap(); + doc_engine.put("users", &key(1), &doc).unwrap(); + let retrieved = doc_engine.get("users", &key(1)).unwrap().unwrap(); assert_eq!(retrieved["name"], "Alice"); assert_eq!(retrieved["email"], "alice@example.com"); @@ -138,13 +148,13 @@ mod tests { let doc_engine = DocumentEngine::new(&sparse, 0, 1); doc_engine - .put("users", "u1", &serde_json::json!({"v": 1})) + .put("users", &key(1), &serde_json::json!({"v": 1})) .unwrap(); doc_engine - .put("users", "u1", &serde_json::json!({"v": 2})) + .put("users", &key(1), &serde_json::json!({"v": 2})) .unwrap(); - let doc = doc_engine.get("users", "u1").unwrap().unwrap(); + let doc = doc_engine.get("users", &key(1)).unwrap().unwrap(); assert_eq!(doc["v"], 2); } @@ -158,12 +168,12 @@ mod tests { let mut buf = Vec::new(); rmpv::encode::write_value(&mut buf, &rmpv_val).unwrap(); - doc_engine.put_raw("col", "id1", &buf).unwrap(); + doc_engine.put_raw("col", &key(1), &buf).unwrap(); - let raw = doc_engine.get_raw("col", "id1").unwrap().unwrap(); + let raw = doc_engine.get_raw("col", &key(1)).unwrap().unwrap(); assert_eq!(raw, buf); - let decoded = doc_engine.get("col", "id1").unwrap().unwrap(); + let decoded = doc_engine.get("col", &key(1)).unwrap().unwrap(); assert_eq!(decoded["key"], "value"); assert_eq!(decoded["num"], 42); } diff --git a/nodedb/src/engine/document/store/key.rs b/nodedb/src/engine/document/store/key.rs index e004d6aa1..6968222a3 100644 --- a/nodedb/src/engine/document/store/key.rs +++ b/nodedb/src/engine/document/store/key.rs @@ -1,215 +1,12 @@ // SPDX-License-Identifier: BUSL-1.1 -//! This module owns both encodings of a row's surrogate identity. -//! -//! [`StorageKey`] is internal: the substrate redb tables (`DOCUMENTS`, -//! `INDEXES`) use string keys, and the Document engine encodes each -//! surrogate as a fixed-width 8-character lowercase hexadecimal string -//! (e.g. `Surrogate(42)` -> `"0000002a"`). It must never reach a client. -//! -//! The format is intentionally fixed width: lexicographic ordering of the -//! hex string matches numeric ordering of the underlying surrogate, so -//! redb range scans can iterate rows in surrogate order without any -//! additional index. -//! -//! [`RowIdentity`] is what a client sees: a predicate matches it, and the -//! catalog binds it. For a minted row it is the surrogate's decimal string. -//! A row with a user-supplied or DDL-declared primary key carries a -//! different identity, the key's body value, bound via the -//! `_system.surrogate_pk{,_rev}` catalog tables. -//! -//! The two types exist so the compiler rejects passing one where the other -//! belongs. A storage key and a user-declared identity can share the same -//! 8-hex-character shape (a primary key `deadbeef` parses as a storage key), -//! so a loose `String` cannot tell them apart. `StorageKey::parse` is the -//! only place that shape gets reinterpreted as a surrogate. -use nodedb_types::Surrogate; - -/// The redb key a document row is stored under. -/// -/// Fixed-width lowercase hex, so lexicographic order matches surrogate order -/// and a range scan iterates rows in surrogate order with no extra index. -/// Internal: a storage key never reaches a client. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub struct StorageKey(Surrogate); - -impl StorageKey { - /// Wrap `surrogate` as the key it is stored under. Allocation-free: the - /// hex text is a rendering, produced only by `Display` or `to_identity`. - pub fn for_surrogate(surrogate: Surrogate) -> Self { - Self(surrogate) - } - - /// Parse a redb key back into a `StorageKey`. - /// - /// Returns `None` unless `key` is exactly 8 lowercase hex characters. - /// This handles legacy non-surrogate document IDs gracefully: a value - /// that fails to parse is not a minted storage key. Allocation-free. - pub fn parse(key: &str) -> Option { - if key.len() != 8 - || !key - .bytes() - .all(|b| b.is_ascii_hexdigit() && !b.is_ascii_uppercase()) - { - return None; - } - let raw = u32::from_str_radix(key, 16).ok()?; - Some(Self(Surrogate::new(raw))) - } - - /// Recover the surrogate this key encodes. Infallible and free: the - /// surrogate IS the key, not something re-derived from stored text. - pub fn surrogate(&self) -> Surrogate { - self.0 - } - - /// The client-visible identity of the row stored under this key. - /// - /// Allocates the decimal string — unavoidable, since a client never sees - /// the hex encoding. - pub fn to_identity(&self) -> RowIdentity { - RowIdentity::for_surrogate(self.0) - } -} - -impl std::fmt::Display for StorageKey { - /// The only `{:08x}` surrogate-to-hex formatting in the workspace. - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - write!(f, "{:08x}", self.0.as_u32()) - } -} - -/// The identity a client sees for a row. -/// -/// A minted row renders its surrogate in decimal. A row with a declared -/// `PRIMARY KEY` carries the user's own value. -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct RowIdentity(String); - -impl RowIdentity { - /// The decimal identity of a minted row. - pub fn for_surrogate(surrogate: Surrogate) -> Self { - Self(surrogate.as_u32().to_string()) - } - - /// Wrap a declared or client-supplied key as the row's identity. - /// - /// The value is taken verbatim and never interpreted as a storage key - /// or a surrogate. This is what makes it safe to call with a KV key, a - /// user-declared primary key, or any other engine-native identifier. - /// - /// Takes `impl Into` so an owned `String` moves in without a - /// copy. A `&str` caller still pays one copy, which is unavoidable. - pub fn from_user_key(key: impl Into) -> Self { - Self(key.into()) - } - - pub fn as_str(&self) -> &str { - &self.0 - } - - /// Consume this identity and return its inner `String` without copying. - pub fn into_string(self) -> String { - self.0 - } -} - -impl std::fmt::Display for RowIdentity { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.write_str(&self.0) - } -} - -/// Format a surrogate as the 8-character zero-padded lowercase hex string -/// used as the document's redb key. -/// -/// Thin wrapper over [`StorageKey::for_surrogate`], kept because 242 -/// call sites across the workspace hold the result as a plain `String` -/// (redb key params, msgpack field injection, WAL replay) rather than a -/// `StorageKey`. Converting all of them is a separate ripple from this one. -/// One allocation: the `Display` format. -pub fn surrogate_to_doc_id(surrogate: Surrogate) -> String { - StorageKey::for_surrogate(surrogate).to_string() -} - -/// Parse a hex-encoded document storage key back to a `Surrogate`. -/// -/// Returns `None` if the key is not exactly 8 lowercase hex characters — -/// this handles legacy non-surrogate document IDs gracefully. -/// -/// Thin wrapper over [`StorageKey::parse`], kept for the same reason as -/// [`surrogate_to_doc_id`]: 35 call sites hold a plain `&str` doc ID. -/// Allocation-free. -pub fn doc_id_to_surrogate(doc_id: &str) -> Option { - StorageKey::parse(doc_id).map(|key| key.surrogate()) -} - -/// The client-visible identity of a row stored under `doc_id`. -/// -/// A minted key renders its surrogate in decimal. Any other key is a user's -/// own value and passes through verbatim. -pub fn identity_of(doc_id: &str) -> RowIdentity { - StorageKey::parse(doc_id) - .map(|key| key.to_identity()) - .unwrap_or_else(|| RowIdentity::from_user_key(doc_id)) -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn formats_zero_padded_lowercase() { - assert_eq!( - StorageKey::for_surrogate(Surrogate::new(0)).to_string(), - "00000000" - ); - assert_eq!( - StorageKey::for_surrogate(Surrogate::new(42)).to_string(), - "0000002a" - ); - assert_eq!( - StorageKey::for_surrogate(Surrogate::new(0xDEAD_BEEF)).to_string(), - "deadbeef" - ); - } - - #[test] - fn lex_order_matches_numeric() { - let a = StorageKey::for_surrogate(Surrogate::new(0x10)).to_string(); - let b = StorageKey::for_surrogate(Surrogate::new(0x100)).to_string(); - assert!(a < b); - } - - #[test] - fn parse_rejects_wrong_shape() { - assert!(StorageKey::parse("").is_none()); - assert!(StorageKey::parse("abc").is_none()); - assert!(StorageKey::parse("DEADBEEF").is_none()); - assert!(StorageKey::parse("zzzzzzzz").is_none()); - assert!(StorageKey::parse("123456789").is_none()); - } - - #[test] - fn parse_roundtrips_surrogate() { - let key = StorageKey::for_surrogate(Surrogate::new(0x2a)); - let parsed = StorageKey::parse(&key.to_string()).expect("valid storage key"); - assert_eq!(parsed.surrogate(), Surrogate::new(0x2a)); - } - - #[test] - fn to_identity_is_decimal() { - let key = StorageKey::for_surrogate(Surrogate::new(42)); - assert_eq!(key.to_identity().as_str(), "42"); - } - - #[test] - fn identity_of_minted_key_is_decimal() { - assert_eq!(identity_of("0000002a").as_str(), "42"); - } - - #[test] - fn identity_of_user_key_passes_through() { - assert_eq!(identity_of("user-declared-id").as_str(), "user-declared-id"); - } -} +//! Re-exports of a row's surrogate identity types. +//! +//! The definitions live in `nodedb_types::row_identity`: `nodedb-physical` +//! and `nodedb-query` depend on `nodedb-types` but not on `nodedb`, and both +//! need these types on their own fields. This module keeps every existing +//! `crate::engine::document::store::{StorageKey, RowIdentity, identity_of, +//! surrogate_to_doc_id, doc_id_to_surrogate}` path compiling unchanged. +pub use nodedb_types::{ + RowIdentity, StorageKey, doc_id_to_surrogate, identity_of, surrogate_to_doc_id, +}; diff --git a/nodedb/src/engine/graph/pattern/executor/core/triple.rs b/nodedb/src/engine/graph/pattern/executor/core/triple.rs index 7abd29b57..626b74d1a 100644 --- a/nodedb/src/engine/graph/pattern/executor/core/triple.rs +++ b/nodedb/src/engine/graph/pattern/executor/core/triple.rs @@ -852,18 +852,19 @@ pub(in crate::engine::graph::pattern::executor) mod tests { /// Store a node-property document in collection `"col"` (matching /// `make_csr`'s `(DatabaseId::DEFAULT, TenantId::new(1))` scope), keyed by - /// `surrogate_to_doc_id(surrogate)` — the REAL document key. A graph node - /// and its same-pk document share one surrogate, so the caller assigns the - /// same surrogate to the node in the CSR via `set_node_surrogate`. + /// `StorageKey::for_surrogate(surrogate)` — the REAL document key. A graph + /// node and its same-pk document share one surrogate, so the caller + /// assigns the same surrogate to the node in the CSR via + /// `set_node_surrogate`. fn put_node_doc( sparse: &SparseEngine, surrogate: nodedb_types::Surrogate, doc: nodedb_types::Value, ) { - use crate::engine::document::store::key::surrogate_to_doc_id; + use nodedb_types::StorageKey; let bytes = nodedb_types::value_to_msgpack(&doc).unwrap(); sparse - .put(0, 1, "col", &surrogate_to_doc_id(surrogate), &bytes) + .put(0, 1, "col", &StorageKey::for_surrogate(surrogate), &bytes) .unwrap(); } diff --git a/nodedb/src/engine/graph/pattern/executor/predicates.rs b/nodedb/src/engine/graph/pattern/executor/predicates.rs index 40f35dae7..e248fd25e 100644 --- a/nodedb/src/engine/graph/pattern/executor/predicates.rs +++ b/nodedb/src/engine/graph/pattern/executor/predicates.rs @@ -252,7 +252,7 @@ fn fetch_node_doc( let Some(surrogate) = props.csr.node_surrogate(node_id) else { return Ok(None); }; - let doc_id = crate::engine::document::store::key::surrogate_to_doc_id(surrogate); + let doc_id = crate::engine::document::store::key::StorageKey::for_surrogate(surrogate); let bytes = match props .sparse .get(props.database_id, props.tenant_id, collection, &doc_id)? diff --git a/nodedb/src/engine/sparse/btree/document.rs b/nodedb/src/engine/sparse/btree/document.rs index ea7c7ba1f..6e9b29d77 100644 --- a/nodedb/src/engine/sparse/btree/document.rs +++ b/nodedb/src/engine/sparse/btree/document.rs @@ -7,6 +7,7 @@ //! `delete_in_txn`, `exists_in_txn`), where the caller owns the redb write //! transaction — plus the collection-wide byte-size estimate. +use nodedb_types::StorageKey; use redb::{ReadableDatabase, ReadableTable, WriteTransaction}; use tracing::debug; @@ -26,7 +27,7 @@ impl SparseEngine { database_id: u64, tenant_id: u64, collection: &str, - document_id: &str, + document_id: &StorageKey, value: &[u8], ) -> crate::Result>> { with_tenant_key(database_id, tenant_id, collection, document_id, |key| { @@ -45,7 +46,7 @@ impl SparseEngine { }; write_txn.commit().map_err(|e| redb_err("commit", e))?; - debug!(collection, document_id, len = value.len(), "document put"); + debug!(collection, %document_id, len = value.len(), "document put"); Ok(prior) }) } @@ -58,7 +59,7 @@ impl SparseEngine { database_id: u64, tenant_id: u64, collection: &str, - document_id: &str, + document_id: &StorageKey, value: &[u8], ) -> crate::Result>> { with_tenant_key(database_id, tenant_id, collection, document_id, |key| { @@ -87,7 +88,7 @@ impl SparseEngine { database_id: u64, tenant_id: u64, collection: &str, - document_id: &str, + document_id: &StorageKey, ) -> crate::Result { with_tenant_key(database_id, tenant_id, collection, document_id, |key| { let table = txn @@ -116,7 +117,7 @@ impl SparseEngine { database_id: u64, tenant_id: u64, collection: &str, - document_id: &str, + document_id: &StorageKey, ) -> crate::Result>> { with_tenant_key(database_id, tenant_id, collection, document_id, |key| { let table = txn @@ -136,7 +137,7 @@ impl SparseEngine { database_id: u64, tenant_id: u64, collection: &str, - documents: &[(&str, &[u8])], + documents: &[(StorageKey, &[u8])], ) -> crate::Result<()> { if documents.is_empty() { return Ok(()); @@ -180,7 +181,7 @@ impl SparseEngine { database_id: u64, tenant_id: u64, collection: &str, - document_id: &str, + document_id: &StorageKey, ) -> crate::Result>> { with_tenant_key(database_id, tenant_id, collection, document_id, |key| { let read_txn = self.db.begin_read().map_err(|e| redb_err("read txn", e))?; @@ -241,7 +242,7 @@ impl SparseEngine { database_id: u64, tenant_id: u64, collection: &str, - document_id: &str, + document_id: &StorageKey, ) -> crate::Result>> { with_tenant_key(database_id, tenant_id, collection, document_id, |key| { let write_txn = self @@ -261,7 +262,7 @@ impl SparseEngine { debug!( collection, - document_id, + %document_id, removed = prior.is_some(), "document delete" ); @@ -281,7 +282,7 @@ impl SparseEngine { database_id: u64, tenant_id: u64, collection: &str, - document_id: &str, + document_id: &StorageKey, ) -> crate::Result>> { with_tenant_key(database_id, tenant_id, collection, document_id, |key| { let mut table = txn @@ -298,6 +299,8 @@ impl SparseEngine { #[cfg(test)] mod tests { + use nodedb_types::Surrogate; + use super::*; fn open_temp() -> (SparseEngine, tempfile::TempDir) { @@ -306,33 +309,37 @@ mod tests { (engine, dir) } + fn key(surrogate: u32) -> StorageKey { + StorageKey::for_surrogate(Surrogate::new(surrogate)) + } + #[test] fn put_and_get() { let (engine, _dir) = open_temp(); - engine.put(0, 1, "users", "u1", b"alice").unwrap(); - engine.put(0, 1, "users", "u2", b"bob").unwrap(); + engine.put(0, 1, "users", &key(1), b"alice").unwrap(); + engine.put(0, 1, "users", &key(2), b"bob").unwrap(); assert_eq!( - engine.get(0, 1, "users", "u1").unwrap(), + engine.get(0, 1, "users", &key(1)).unwrap(), Some(b"alice".to_vec()) ); assert_eq!( - engine.get(0, 1, "users", "u2").unwrap(), + engine.get(0, 1, "users", &key(2)).unwrap(), Some(b"bob".to_vec()) ); - assert_eq!(engine.get(0, 1, "users", "u3").unwrap(), None); + assert_eq!(engine.get(0, 1, "users", &key(3)).unwrap(), None); } #[test] fn databases_are_isolated() { let (engine, _dir) = open_temp(); - engine.put(0, 1, "users", "u1", b"alice").unwrap(); - engine.put(7, 1, "users", "u1", b"alice-db7").unwrap(); + engine.put(0, 1, "users", &key(1), b"alice").unwrap(); + engine.put(7, 1, "users", &key(1), b"alice-db7").unwrap(); assert_eq!( - engine.get(0, 1, "users", "u1").unwrap(), + engine.get(0, 1, "users", &key(1)).unwrap(), Some(b"alice".to_vec()) ); assert_eq!( - engine.get(7, 1, "users", "u1").unwrap(), + engine.get(7, 1, "users", &key(1)).unwrap(), Some(b"alice-db7".to_vec()) ); } @@ -340,10 +347,10 @@ mod tests { #[test] fn put_overwrites() { let (engine, _dir) = open_temp(); - engine.put(0, 1, "users", "u1", b"alice").unwrap(); - engine.put(0, 1, "users", "u1", b"ALICE").unwrap(); + engine.put(0, 1, "users", &key(1), b"alice").unwrap(); + engine.put(0, 1, "users", &key(1), b"ALICE").unwrap(); assert_eq!( - engine.get(0, 1, "users", "u1").unwrap(), + engine.get(0, 1, "users", &key(1)).unwrap(), Some(b"ALICE".to_vec()) ); } @@ -351,26 +358,26 @@ mod tests { #[test] fn delete_removes() { let (engine, _dir) = open_temp(); - engine.put(0, 1, "users", "u1", b"alice").unwrap(); + engine.put(0, 1, "users", &key(1), b"alice").unwrap(); assert_eq!( - engine.delete(0, 1, "users", "u1").unwrap(), + engine.delete(0, 1, "users", &key(1)).unwrap(), Some(b"alice".to_vec()) ); - assert_eq!(engine.get(0, 1, "users", "u1").unwrap(), None); - assert_eq!(engine.delete(0, 1, "users", "u1").unwrap(), None); + assert_eq!(engine.get(0, 1, "users", &key(1)).unwrap(), None); + assert_eq!(engine.delete(0, 1, "users", &key(1)).unwrap(), None); } #[test] fn collections_are_isolated() { let (engine, _dir) = open_temp(); - engine.put(0, 1, "users", "u1", b"alice").unwrap(); - engine.put(0, 1, "orders", "u1", b"order-1").unwrap(); + engine.put(0, 1, "users", &key(1), b"alice").unwrap(); + engine.put(0, 1, "orders", &key(1), b"order-1").unwrap(); assert_eq!( - engine.get(0, 1, "users", "u1").unwrap(), + engine.get(0, 1, "users", &key(1)).unwrap(), Some(b"alice".to_vec()) ); assert_eq!( - engine.get(0, 1, "orders", "u1").unwrap(), + engine.get(0, 1, "orders", &key(1)).unwrap(), Some(b"order-1".to_vec()) ); } diff --git a/nodedb/src/engine/sparse/btree/keys.rs b/nodedb/src/engine/sparse/btree/keys.rs index 42ed72766..d6585842b 100644 --- a/nodedb/src/engine/sparse/btree/keys.rs +++ b/nodedb/src/engine/sparse/btree/keys.rs @@ -26,12 +26,14 @@ std::thread_local! { } /// Build a database/tenant-scoped composite key `"{db}:{tenant}:{a}:{b}"` -/// using a thread-local buffer. +/// in the thread-local buffer. `b` is written via `Display`, so a +/// [`nodedb_types::StorageKey`] lands as its 8 hex characters with no +/// intermediate `String`. pub(super) fn with_tenant_key( database_id: u64, tenant_id: u64, a: &str, - b: &str, + b: impl std::fmt::Display, f: impl FnOnce(&str) -> R, ) -> R { KEY_BUF.with(|buf| { @@ -44,7 +46,7 @@ pub(super) fn with_tenant_key( buf.push(':'); buf.push_str(a); buf.push(':'); - buf.push_str(b); + let _ = write!(buf, "{b}"); f(&buf) }) } diff --git a/nodedb/src/engine/sparse/btree/mod.rs b/nodedb/src/engine/sparse/btree/mod.rs index c2d070a31..49729b912 100644 --- a/nodedb/src/engine/sparse/btree/mod.rs +++ b/nodedb/src/engine/sparse/btree/mod.rs @@ -13,4 +13,4 @@ pub use engine::SparseEngine; pub(crate) use keys::coll_prefix; pub(in crate::engine::sparse) use keys::{tenant_prefix, with_tenant_key4}; pub(crate) use tables::DOCUMENTS; -pub(in crate::engine::sparse) use tables::{INDEXES, redb_err}; +pub(in crate::engine::sparse) use tables::{INDEXES, invalid_storage_key_err, redb_err}; diff --git a/nodedb/src/engine/sparse/btree/rename.rs b/nodedb/src/engine/sparse/btree/rename.rs index 95ecbe559..d3dcf176c 100644 --- a/nodedb/src/engine/sparse/btree/rename.rs +++ b/nodedb/src/engine/sparse/btree/rename.rs @@ -139,6 +139,8 @@ impl SparseEngine { #[cfg(test)] mod tests { + use nodedb_types::{StorageKey, Surrogate}; + use super::*; fn open_temp() -> (SparseEngine, tempfile::TempDir) { @@ -147,13 +149,17 @@ mod tests { (engine, dir) } + fn key(surrogate: u32) -> StorageKey { + StorageKey::for_surrogate(Surrogate::new(surrogate)) + } + /// Renaming a collection moves its chain head with its rows. Leaving the head /// under the old name would restart the renamed collection at genesis while /// its already-chained rows travelled to the new name. #[test] fn rename_moves_the_chain_head() { let (engine, _dir) = open_temp(); - engine.put(0, 1, "ledger", "0000002a", b"row").unwrap(); + engine.put(0, 1, "ledger", &key(42), b"row").unwrap(); engine.put_chain_head(0, 1, "ledger", "h1").unwrap(); engine diff --git a/nodedb/src/engine/sparse/btree/tables.rs b/nodedb/src/engine/sparse/btree/tables.rs index f6890f168..ed9c10240 100644 --- a/nodedb/src/engine/sparse/btree/tables.rs +++ b/nodedb/src/engine/sparse/btree/tables.rs @@ -21,3 +21,19 @@ pub(in crate::engine::sparse) fn redb_err(ctx: &str, e: E) detail: format!("{ctx}: {e}"), } } + +/// Report a DOCUMENTS row whose key does not parse as a [`nodedb_types::StorageKey`]. +/// +/// A non-parsing key on this table is a violated storage invariant, not a +/// legacy row to skip: every DOCUMENTS key is minted by [`StorageKey::for_surrogate`]. +pub(in crate::engine::sparse) fn invalid_storage_key_err( + collection: &str, + key: &str, +) -> crate::Error { + crate::Error::Storage { + engine: "sparse".into(), + detail: format!( + "collection '{collection}' has a DOCUMENTS row whose key is not a valid storage key: '{key}'" + ), + } +} diff --git a/nodedb/src/engine/sparse/btree_scan.rs b/nodedb/src/engine/sparse/btree_scan.rs index e32c09c52..474549f2d 100644 --- a/nodedb/src/engine/sparse/btree_scan.rs +++ b/nodedb/src/engine/sparse/btree_scan.rs @@ -2,10 +2,13 @@ //! Table scanning and import/export methods for `SparseEngine`. +use nodedb_types::StorageKey; use redb::{ReadableDatabase, ReadableTable}; use tracing::debug; -use super::btree::{DOCUMENTS, INDEXES, SparseEngine, coll_prefix, redb_err, tenant_prefix}; +use super::btree::{ + DOCUMENTS, INDEXES, SparseEngine, coll_prefix, invalid_storage_key_err, redb_err, tenant_prefix, +}; impl SparseEngine { /// Scan documents in a collection (reads DOCUMENTS table, not INDEXES). @@ -18,7 +21,7 @@ impl SparseEngine { tenant_id: u64, collection: &str, limit: usize, - ) -> crate::Result)>> { + ) -> crate::Result)>> { let prefix = coll_prefix(database_id, tenant_id, collection); let end = format!("{prefix}\u{ffff}"); @@ -37,11 +40,13 @@ impl SparseEngine { break; } let entry = entry.map_err(|e| redb_err("doc entry", e))?; - let key = entry.0.value().to_string(); + let key = entry.0.value(); // Extract document_id from key format "{database_id}:{tenant}:{collection}:{doc_id}" - let doc_id = key.strip_prefix(&prefix).unwrap_or(&key).to_string(); + let doc_id = key.strip_prefix(&prefix).unwrap_or(key); + let storage_key = StorageKey::parse(doc_id) + .ok_or_else(|| invalid_storage_key_err(collection, doc_id))?; let value = entry.1.value().to_vec(); - results.push((doc_id, value)); + results.push((storage_key, value)); } debug!(collection, count = results.len(), "document scan"); @@ -71,7 +76,7 @@ impl SparseEngine { mut f: F, ) -> crate::Result<()> where - F: FnMut(&str, &[u8]) -> crate::Result<()>, + F: FnMut(&StorageKey, &[u8]) -> crate::Result<()>, { let prefix = coll_prefix(database_id, tenant_id, collection); let end = format!("{prefix}\u{ffff}"); @@ -91,11 +96,13 @@ impl SparseEngine { break; } let entry = entry.map_err(|e| redb_err("doc entry", e))?; - let key = entry.0.value().to_string(); + let key = entry.0.value(); // Extract document_id from key format "{database_id}:{tenant}:{collection}:{doc_id}" - let doc_id = key.strip_prefix(&prefix).unwrap_or(&key); + let doc_id = key.strip_prefix(&prefix).unwrap_or(key); + let storage_key = StorageKey::parse(doc_id) + .ok_or_else(|| invalid_storage_key_err(collection, doc_id))?; let value = entry.1.value(); - f(doc_id, value)?; + f(&storage_key, value)?; count += 1; } @@ -120,7 +127,7 @@ impl SparseEngine { mut handler: F, ) -> crate::Result where - F: FnMut(&[(String, Vec)]), + F: FnMut(&[(StorageKey, Vec)]), { let prefix = coll_prefix(database_id, tenant_id, collection); let end = format!("{prefix}\u{ffff}"); @@ -142,10 +149,12 @@ impl SparseEngine { break; } let entry = entry.map_err(|e| redb_err("doc entry", e))?; - let key = entry.0.value().to_string(); - let doc_id = key.strip_prefix(&prefix).unwrap_or(&key).to_string(); + let key = entry.0.value(); + let doc_id = key.strip_prefix(&prefix).unwrap_or(key); + let storage_key = StorageKey::parse(doc_id) + .ok_or_else(|| invalid_storage_key_err(collection, doc_id))?; let value = entry.1.value().to_vec(); - chunk.push((doc_id, value)); + chunk.push((storage_key, value)); total += 1; if chunk.len() >= chunk_size { @@ -294,9 +303,9 @@ impl SparseEngine { tenant_id: u64, collection: &str, limit: usize, - predicate: &dyn Fn(&str, &[u8]) -> bool, + predicate: &dyn Fn(&StorageKey, &[u8]) -> bool, stop: &dyn Fn() -> bool, - ) -> crate::Result)>> { + ) -> crate::Result)>> { let prefix = coll_prefix(database_id, tenant_id, collection); let end = format!("{prefix}\u{ffff}"); @@ -321,13 +330,15 @@ impl SparseEngine { let value_bytes = entry.1.value(); let key = entry.0.value(); let doc_id = key.strip_prefix(&prefix).unwrap_or(key); + let storage_key = StorageKey::parse(doc_id) + .ok_or_else(|| invalid_storage_key_err(collection, doc_id))?; // Evaluate predicate on raw bytes — skip allocation if no match. - if !predicate(doc_id, value_bytes) { + if !predicate(&storage_key, value_bytes) { continue; } - results.push((doc_id.to_string(), value_bytes.to_vec())); + results.push((storage_key, value_bytes.to_vec())); } debug!(collection, count = results.len(), "filtered document scan"); @@ -478,6 +489,8 @@ impl SparseEngine { #[cfg(test)] mod tests { + use nodedb_types::Surrogate; + use super::*; fn open_temp() -> (SparseEngine, tempfile::TempDir) { @@ -486,19 +499,23 @@ mod tests { (engine, dir) } + fn key(surrogate: u32) -> StorageKey { + StorageKey::for_surrogate(Surrogate::new(surrogate)) + } + #[test] fn for_each_matches_scan_documents() { let (engine, _dir) = open_temp(); - engine.put(0, 1, "users", "u1", b"alice").unwrap(); - engine.put(0, 1, "users", "u2", b"bob").unwrap(); - engine.put(0, 1, "users", "u3", b"carol").unwrap(); + engine.put(0, 1, "users", &key(1), b"alice").unwrap(); + engine.put(0, 1, "users", &key(2), b"bob").unwrap(); + engine.put(0, 1, "users", &key(3), b"carol").unwrap(); let materialized = engine.scan_documents(0, 1, "users", usize::MAX).unwrap(); - let mut streamed: Vec<(String, Vec)> = Vec::new(); + let mut streamed: Vec<(StorageKey, Vec)> = Vec::new(); engine .scan_documents_for_each(0, 1, "users", usize::MAX, |doc_id, bytes| { - streamed.push((doc_id.to_string(), bytes.to_vec())); + streamed.push((*doc_id, bytes.to_vec())); Ok(()) }) .unwrap(); @@ -510,16 +527,16 @@ mod tests { #[test] fn for_each_respects_limit() { let (engine, _dir) = open_temp(); - engine.put(0, 1, "users", "u1", b"alice").unwrap(); - engine.put(0, 1, "users", "u2", b"bob").unwrap(); - engine.put(0, 1, "users", "u3", b"carol").unwrap(); + engine.put(0, 1, "users", &key(1), b"alice").unwrap(); + engine.put(0, 1, "users", &key(2), b"bob").unwrap(); + engine.put(0, 1, "users", &key(3), b"carol").unwrap(); let materialized = engine.scan_documents(0, 1, "users", 2).unwrap(); - let mut streamed: Vec<(String, Vec)> = Vec::new(); + let mut streamed: Vec<(StorageKey, Vec)> = Vec::new(); engine .scan_documents_for_each(0, 1, "users", 2, |doc_id, bytes| { - streamed.push((doc_id.to_string(), bytes.to_vec())); + streamed.push((*doc_id, bytes.to_vec())); Ok(()) }) .unwrap(); @@ -535,9 +552,7 @@ mod tests { fn filtered_scan_stops_on_the_stop_signal() { let (engine, _dir) = open_temp(); for i in 0..8 { - engine - .put(0, 1, "users", &format!("u{i}"), b"body") - .unwrap(); + engine.put(0, 1, "users", &key(i), b"body").unwrap(); } // Stop after the third row is visited. @@ -548,7 +563,7 @@ mod tests { 1, "users", usize::MAX, - &|_: &str, _: &[u8]| { + &|_: &StorageKey, _: &[u8]| { visited.set(visited.get() + 1); true }, @@ -564,9 +579,7 @@ mod tests { fn filtered_scan_runs_to_completion_without_a_stop() { let (engine, _dir) = open_temp(); for i in 0..8 { - engine - .put(0, 1, "users", &format!("u{i}"), b"body") - .unwrap(); + engine.put(0, 1, "users", &key(i), b"body").unwrap(); } let rows = engine @@ -575,7 +588,7 @@ mod tests { 1, "users", usize::MAX, - &|_: &str, _: &[u8]| true, + &|_: &StorageKey, _: &[u8]| true, &crate::engine::sparse::scan_stop::never_stop, ) .unwrap(); @@ -586,8 +599,8 @@ mod tests { #[test] fn for_each_propagates_callback_error() { let (engine, _dir) = open_temp(); - engine.put(0, 1, "users", "u1", b"alice").unwrap(); - engine.put(0, 1, "users", "u2", b"bob").unwrap(); + engine.put(0, 1, "users", &key(1), b"alice").unwrap(); + engine.put(0, 1, "users", &key(2), b"bob").unwrap(); let mut seen = 0usize; let result = diff --git a/nodedb/src/engine/sparse/doc_cache.rs b/nodedb/src/engine/sparse/doc_cache.rs index c94c6bb31..563bf2c2c 100644 --- a/nodedb/src/engine/sparse/doc_cache.rs +++ b/nodedb/src/engine/sparse/doc_cache.rs @@ -24,6 +24,8 @@ use std::cell::Cell; use std::collections::{HashMap, VecDeque}; +use nodedb_types::StorageKey; + /// Composite cache key: `(database_id, tenant_id, collection, document_id)`. /// /// Same `(tenant_id, collection, document_id)` in two different databases are @@ -137,7 +139,7 @@ impl DocCache { database_id: u64, tenant_id: u64, collection: &str, - document_id: &str, + document_id: &StorageKey, ) -> Option<&[u8]> { let key = Self::make_key(database_id, tenant_id, collection, document_id); match self @@ -162,7 +164,7 @@ impl DocCache { database_id: u64, tenant_id: u64, collection: &str, - document_id: &str, + document_id: &StorageKey, value: &[u8], ) { let key = Self::make_key(database_id, tenant_id, collection, document_id); @@ -238,7 +240,7 @@ impl DocCache { database_id: u64, tenant_id: u64, collection: &str, - document_id: &str, + document_id: &StorageKey, ) { let key = Self::make_key(database_id, tenant_id, collection, document_id); let removed = self @@ -322,7 +324,12 @@ impl DocCache { // ── Private helpers ─────────────────────────────────────────────────────── - fn make_key(database_id: u64, tenant_id: u64, collection: &str, document_id: &str) -> CacheKey { + fn make_key( + database_id: u64, + tenant_id: u64, + collection: &str, + document_id: &StorageKey, + ) -> CacheKey { CacheKey { database_id, tenant_id, @@ -376,50 +383,62 @@ impl DocCache { #[cfg(test)] mod tests { + use nodedb_types::Surrogate; + use super::*; + fn key(surrogate: u32) -> StorageKey { + StorageKey::for_surrogate(Surrogate::new(surrogate)) + } + // ── Basic API ───────────────────────────────────────────────────────────── #[test] fn basic_put_get() { let mut cache = DocCache::new(16); - cache.put(0, 1, "users", "u1", b"alice"); - assert_eq!(cache.get(0, 1, "users", "u1"), Some(b"alice".as_slice())); - assert_eq!(cache.get(0, 1, "users", "u2"), None); + cache.put(0, 1, "users", &key(1), b"alice"); + assert_eq!(cache.get(0, 1, "users", &key(1)), Some(b"alice".as_slice())); + assert_eq!(cache.get(0, 1, "users", &key(2)), None); } #[test] fn overwrite_updates_value() { let mut cache = DocCache::new(16); - cache.put(0, 1, "users", "u1", b"alice"); - cache.put(0, 1, "users", "u1", b"ALICE"); - assert_eq!(cache.get(0, 1, "users", "u1"), Some(b"ALICE".as_slice())); + cache.put(0, 1, "users", &key(1), b"alice"); + cache.put(0, 1, "users", &key(1), b"ALICE"); + assert_eq!(cache.get(0, 1, "users", &key(1)), Some(b"ALICE".as_slice())); } #[test] fn invalidate_removes_entry() { let mut cache = DocCache::new(16); - cache.put(0, 1, "users", "u1", b"alice"); - cache.invalidate(0, 1, "users", "u1"); - assert_eq!(cache.get(0, 1, "users", "u1"), None); + cache.put(0, 1, "users", &key(1), b"alice"); + cache.invalidate(0, 1, "users", &key(1)); + assert_eq!(cache.get(0, 1, "users", &key(1)), None); } #[test] fn tenant_isolation() { let mut cache = DocCache::new(16); - cache.put(0, 1, "users", "u1", b"tenant1"); - cache.put(0, 2, "users", "u1", b"tenant2"); - assert_eq!(cache.get(0, 1, "users", "u1"), Some(b"tenant1".as_slice())); - assert_eq!(cache.get(0, 2, "users", "u1"), Some(b"tenant2".as_slice())); + cache.put(0, 1, "users", &key(1), b"tenant1"); + cache.put(0, 2, "users", &key(1), b"tenant2"); + assert_eq!( + cache.get(0, 1, "users", &key(1)), + Some(b"tenant1".as_slice()) + ); + assert_eq!( + cache.get(0, 2, "users", &key(1)), + Some(b"tenant2".as_slice()) + ); } #[test] fn hit_rate_tracking() { let mut cache = DocCache::new(16); - cache.put(0, 1, "c", "a", b"1"); - cache.get(0, 1, "c", "a"); // hit - cache.get(0, 1, "c", "a"); // hit - cache.get(0, 1, "c", "b"); // miss + cache.put(0, 1, "c", &key(1), b"1"); + cache.get(0, 1, "c", &key(1)); // hit + cache.get(0, 1, "c", &key(1)); // hit + cache.get(0, 1, "c", &key(2)); // miss assert!((cache.hit_rate() - 0.6667).abs() < 0.01); assert_eq!(cache.total_lookups(), 3); } @@ -429,10 +448,10 @@ mod tests { #[test] fn cache_key_uniqueness_across_databases() { let mut cache = DocCache::new(16); - cache.put(1, 5, "col", "doc", b"db1"); - cache.put(2, 5, "col", "doc", b"db2"); - assert_eq!(cache.get(1, 5, "col", "doc"), Some(b"db1".as_slice())); - assert_eq!(cache.get(2, 5, "col", "doc"), Some(b"db2".as_slice())); + cache.put(1, 5, "col", &key(1), b"db1"); + cache.put(2, 5, "col", &key(1), b"db2"); + assert_eq!(cache.get(1, 5, "col", &key(1)), Some(b"db1".as_slice())); + assert_eq!(cache.get(2, 5, "col", &key(1)), Some(b"db2".as_slice())); } // ── Weighted eviction ───────────────────────────────────────────────── @@ -447,10 +466,10 @@ mod tests { cache.set_database_weight(2, 1); for i in 0..4u32 { - cache.put(1, 1, "c", &format!("db1-{i}"), b"v"); + cache.put(1, 1, "c", &key(i), b"v"); } for i in 0..4u32 { - cache.put(2, 1, "c", &format!("db2-{i}"), b"v"); + cache.put(2, 1, "c", &key(1000 + i), b"v"); } assert_eq!(cache.len(), 8); @@ -458,11 +477,11 @@ mod tests { // Since DB1 gets more new inserts, DB1's overshoot grows first and it // gets evicted. DB2 should retain all 4 entries. for i in 4..8u32 { - cache.put(1, 1, "c", &format!("db1-{i}"), b"v"); + cache.put(1, 1, "c", &key(i), b"v"); } let db2_resident: usize = (0..4u32) - .filter(|i| cache.get(2, 1, "c", &format!("db2-{i}")).is_some()) + .filter(|i| cache.get(2, 1, "c", &key(1000 + i)).is_some()) .count(); assert_eq!( db2_resident, 4, @@ -480,22 +499,22 @@ mod tests { // Fill with 5 DB1 and 5 DB2 entries. for i in 0..5u32 { - cache.put(1, 1, "c", &format!("a{i}"), b"v"); - cache.put(2, 1, "c", &format!("b{i}"), b"v"); + cache.put(1, 1, "c", &key(i), b"v"); + cache.put(2, 1, "c", &key(1000 + i), b"v"); } assert_eq!(cache.len(), capacity); // Add 5 more DB1 entries to trigger eviction. DB2 (weight=4) should // retain more entries than DB1 (weight=1). for i in 5..10u32 { - cache.put(1, 1, "c", &format!("a{i}"), b"v"); + cache.put(1, 1, "c", &key(i), b"v"); } let db1_count = (0..10u32) - .filter(|i| cache.get(1, 1, "c", &format!("a{i}")).is_some()) + .filter(|i| cache.get(1, 1, "c", &key(*i)).is_some()) .count(); let db2_count = (0..5u32) - .filter(|i| cache.get(2, 1, "c", &format!("b{i}")).is_some()) + .filter(|i| cache.get(2, 1, "c", &key(1000 + i)).is_some()) .count(); assert!( db2_count > db1_count, @@ -506,26 +525,26 @@ mod tests { #[test] fn evict_collection_removes_correct_entries() { let mut cache = DocCache::new(16); - cache.put(1, 1, "col_a", "d1", b"1"); - cache.put(1, 1, "col_b", "d1", b"2"); - cache.put(2, 1, "col_a", "d1", b"3"); + cache.put(1, 1, "col_a", &key(1), b"1"); + cache.put(1, 1, "col_b", &key(1), b"2"); + cache.put(2, 1, "col_a", &key(1), b"3"); cache.evict_collection(1, 1, "col_a"); - assert_eq!(cache.get(1, 1, "col_a", "d1"), None); - assert_eq!(cache.get(1, 1, "col_b", "d1"), Some(b"2".as_slice())); - assert_eq!(cache.get(2, 1, "col_a", "d1"), Some(b"3".as_slice())); + assert_eq!(cache.get(1, 1, "col_a", &key(1)), None); + assert_eq!(cache.get(1, 1, "col_b", &key(1)), Some(b"2".as_slice())); + assert_eq!(cache.get(2, 1, "col_a", &key(1)), Some(b"3".as_slice())); } #[test] fn evict_tenant_removes_correct_entries() { let mut cache = DocCache::new(16); - cache.put(1, 1, "col", "d1", b"t1"); - cache.put(1, 2, "col", "d1", b"t2"); - cache.put(2, 1, "col", "d1", b"db2"); + cache.put(1, 1, "col", &key(1), b"t1"); + cache.put(1, 2, "col", &key(1), b"t2"); + cache.put(2, 1, "col", &key(1), b"db2"); cache.evict_tenant(1, 1); - assert_eq!(cache.get(1, 1, "col", "d1"), None); - assert_eq!(cache.get(1, 2, "col", "d1"), Some(b"t2".as_slice())); - assert_eq!(cache.get(2, 1, "col", "d1"), Some(b"db2".as_slice())); + assert_eq!(cache.get(1, 1, "col", &key(1)), None); + assert_eq!(cache.get(1, 2, "col", &key(1)), Some(b"t2".as_slice())); + assert_eq!(cache.get(2, 1, "col", &key(1)), Some(b"db2".as_slice())); } } diff --git a/nodedb/src/error/types.rs b/nodedb/src/error/types.rs index a4ed8d330..dcd1d5bbf 100644 --- a/nodedb/src/error/types.rs +++ b/nodedb/src/error/types.rs @@ -112,6 +112,20 @@ pub enum Error { #[error("period locked on {collection}: {detail}")] PeriodLocked { collection: String, detail: String }, + /// A period-lock reference row exists but does not carry the + /// configured `status_column` — a misconfigured column name, not a + /// locked period. + #[error( + "period lock on {collection} misconfigured: reference table '{ref_table}' row \ + '{row_identity}' has no column '{status_column}'" + )] + PeriodLockMisconfigured { + collection: String, + ref_table: String, + status_column: String, + row_identity: String, + }, + #[error("retention violation on {collection}: {detail}")] RetentionViolation { collection: String, detail: String }, diff --git a/nodedb/src/error_classify.rs b/nodedb/src/error_classify.rs index aaf58663c..1b08e588d 100644 --- a/nodedb/src/error_classify.rs +++ b/nodedb/src/error_classify.rs @@ -74,6 +74,17 @@ pub(crate) fn classify(e: &Error) -> NodeDbError { Error::PeriodLocked { collection, detail, .. } => NodeDbError::period_locked(collection.clone(), detail), + Error::PeriodLockMisconfigured { + collection, + ref_table, + status_column, + row_identity, + } => NodeDbError::period_lock_misconfigured( + collection.clone(), + ref_table.clone(), + status_column.clone(), + row_identity, + ), Error::RetentionViolation { collection, detail, .. } => NodeDbError::retention_violation(collection.clone(), detail), diff --git a/nodedb/src/error_from_data_plane.rs b/nodedb/src/error_from_data_plane.rs index 74f271a99..46ac98e1f 100644 --- a/nodedb/src/error_from_data_plane.rs +++ b/nodedb/src/error_from_data_plane.rs @@ -78,6 +78,17 @@ pub(crate) fn data_plane_code_to_public(code: ErrorCode) -> NodeDbError { ErrorCode::PeriodLocked { collection } => { NodeDbError::period_locked(collection, "writes rejected") } + ErrorCode::PeriodLockMisconfigured { + collection, + ref_table, + status_column, + row_identity, + } => NodeDbError::period_lock_misconfigured( + collection, + ref_table, + status_column, + row_identity, + ), ErrorCode::RetentionViolation { collection } => { NodeDbError::retention_violation(collection, "retention period has not expired") } diff --git a/nodedb/tests/inproc/cases/collection_purge_persistent.rs b/nodedb/tests/inproc/cases/collection_purge_persistent.rs index 0edbe37a7..ef08a0d14 100644 --- a/nodedb/tests/inproc/cases/collection_purge_persistent.rs +++ b/nodedb/tests/inproc/cases/collection_purge_persistent.rs @@ -8,6 +8,7 @@ //! the cross-engine regression gate that catches a future refactor //! where only some engines honor the scoped purge. +use nodedb::engine::document::store::StorageKey; use nodedb::engine::kv::{KvEngine, KvPutParams}; use nodedb::engine::sparse::btree::SparseEngine; use nodedb::engine::sparse::inverted::InvertedIndex; @@ -25,16 +26,20 @@ fn open_sparse() -> (tempfile::TempDir, SparseEngine) { (tmp, sparse) } +fn key(n: u32) -> StorageKey { + StorageKey::for_surrogate(Surrogate::new(n)) +} + #[test] fn sparse_engine_purge_leaves_no_documents_for_collection() { let (_tmp, sparse) = open_sparse(); let doc_bytes = b"{\"k\":1}".to_vec(); - sparse.put(DB, TENANT, "keep", "d1", &doc_bytes).unwrap(); + sparse.put(DB, TENANT, "keep", &key(1), &doc_bytes).unwrap(); sparse - .put(DB, TENANT, "purge_me", "d1", &doc_bytes) + .put(DB, TENANT, "purge_me", &key(1), &doc_bytes) .unwrap(); sparse - .put(DB, TENANT, "purge_me", "d2", &doc_bytes) + .put(DB, TENANT, "purge_me", &key(2), &doc_bytes) .unwrap(); let (docs_removed, _idx_removed) = sparse @@ -42,23 +47,33 @@ fn sparse_engine_purge_leaves_no_documents_for_collection() { .unwrap(); assert_eq!(docs_removed, 2); - assert!(sparse.get(DB, TENANT, "purge_me", "d1").unwrap().is_none()); - assert!(sparse.get(DB, TENANT, "purge_me", "d2").unwrap().is_none()); - assert!(sparse.get(DB, TENANT, "keep", "d1").unwrap().is_some()); + assert!( + sparse + .get(DB, TENANT, "purge_me", &key(1)) + .unwrap() + .is_none() + ); + assert!( + sparse + .get(DB, TENANT, "purge_me", &key(2)) + .unwrap() + .is_none() + ); + assert!(sparse.get(DB, TENANT, "keep", &key(1)).unwrap().is_some()); } #[test] fn sparse_engine_cross_tenant_isolation() { let (_tmp, sparse) = open_sparse(); let doc_bytes = b"{\"k\":1}".to_vec(); - sparse.put(DB, 1, "docs", "d1", &doc_bytes).unwrap(); - sparse.put(DB, 2, "docs", "d1", &doc_bytes).unwrap(); + sparse.put(DB, 1, "docs", &key(1), &doc_bytes).unwrap(); + sparse.put(DB, 2, "docs", &key(1), &doc_bytes).unwrap(); let (removed, _) = sparse.delete_all_for_collection(DB, 1, "docs").unwrap(); assert_eq!(removed, 1); - assert!(sparse.get(DB, 1, "docs", "d1").unwrap().is_none()); + assert!(sparse.get(DB, 1, "docs", &key(1)).unwrap().is_none()); assert!( - sparse.get(DB, 2, "docs", "d1").unwrap().is_some(), + sparse.get(DB, 2, "docs", &key(1)).unwrap().is_some(), "tenant 2's same-named collection must survive tenant 1's purge" ); } diff --git a/nodedb/tests/inproc/cases/document_bitemporal_dml.rs b/nodedb/tests/inproc/cases/document_bitemporal_dml.rs index 94a07b028..c1899b9a4 100644 --- a/nodedb/tests/inproc/cases/document_bitemporal_dml.rs +++ b/nodedb/tests/inproc/cases/document_bitemporal_dml.rs @@ -11,8 +11,13 @@ //! prior versions from current-state reads only" — are the correctness //! contract the handlers rely on. -use nodedb::engine::document::store::{CollectionConfig, DocumentEngine}; +use nodedb::engine::document::store::{CollectionConfig, DocumentEngine, StorageKey}; use nodedb::engine::sparse::btree::SparseEngine; +use nodedb_types::Surrogate; + +fn key(n: u32) -> StorageKey { + StorageKey::for_surrogate(Surrogate::new(n)) +} fn open() -> (SparseEngine, tempfile::TempDir) { let dir = tempfile::tempdir().unwrap(); @@ -38,17 +43,21 @@ fn update_via_put_creates_new_version_and_preserves_old() { let (sparse, _d) = open(); let engine = register(&sparse); - engine.put("c", "k", &serde_json::json!({"v": 1})).unwrap(); + engine + .put("c", &key(1), &serde_json::json!({"v": 1})) + .unwrap(); std::thread::sleep(std::time::Duration::from_millis(5)); let t_mid = wall_ms(); std::thread::sleep(std::time::Duration::from_millis(5)); - engine.put("c", "k", &serde_json::json!({"v": 2})).unwrap(); + engine + .put("c", &key(1), &serde_json::json!({"v": 2})) + .unwrap(); - assert_eq!(engine.get("c", "k").unwrap().unwrap()["v"], 2); + assert_eq!(engine.get("c", &key(1)).unwrap().unwrap()["v"], 2); // History at t_mid still sees v=1. let body = sparse - .versioned_get_as_of(0, 1, "c", "k", Some(t_mid), None) + .versioned_get_as_of(0, 1, "c", &key(1).to_string(), Some(t_mid), None) .unwrap() .expect("historical version"); let rmpv_val = rmpv::decode::read_value(&mut body.as_slice()).unwrap(); @@ -62,19 +71,19 @@ fn delete_appends_tombstone_but_prior_version_still_visible_as_of() { let engine = register(&sparse); engine - .put("c", "k", &serde_json::json!({"name": "Alice"})) + .put("c", &key(1), &serde_json::json!({"name": "Alice"})) .unwrap(); std::thread::sleep(std::time::Duration::from_millis(5)); let t_before_delete = wall_ms(); std::thread::sleep(std::time::Duration::from_millis(5)); - assert!(engine.delete("c", "k").unwrap()); + assert!(engine.delete("c", &key(1)).unwrap()); // Current-state read: None. - assert!(engine.get("c", "k").unwrap().is_none()); + assert!(engine.get("c", &key(1)).unwrap().is_none()); // Historical read at t_before_delete: still Alice. let body = sparse - .versioned_get_as_of(0, 1, "c", "k", Some(t_before_delete), None) + .versioned_get_as_of(0, 1, "c", &key(1).to_string(), Some(t_before_delete), None) .unwrap() .expect("pre-delete version still reachable"); let rmpv_val = rmpv::decode::read_value(&mut body.as_slice()).unwrap(); @@ -87,14 +96,16 @@ fn ten_sequential_updates_produce_ten_reachable_versions() { let engine = register(&sparse); let mut cutoffs: Vec<(i64, i64)> = Vec::new(); // (cutoff_ms, expected_v) for i in 1..=10 { - engine.put("c", "k", &serde_json::json!({"v": i})).unwrap(); + engine + .put("c", &key(1), &serde_json::json!({"v": i})) + .unwrap(); std::thread::sleep(std::time::Duration::from_millis(3)); cutoffs.push((wall_ms(), i)); std::thread::sleep(std::time::Duration::from_millis(3)); } for (cutoff, expected) in &cutoffs { let body = sparse - .versioned_get_as_of(0, 1, "c", "k", Some(*cutoff), None) + .versioned_get_as_of(0, 1, "c", &key(1).to_string(), Some(*cutoff), None) .unwrap() .unwrap_or_else(|| panic!("missing version at cutoff {cutoff}")); let rmpv_val = rmpv::decode::read_value(&mut body.as_slice()).unwrap(); @@ -114,29 +125,29 @@ fn secondary_index_reflects_each_version_independently() { ); engine - .put("c", "u1", &serde_json::json!({"email": "a@x.com"})) + .put("c", &key(1), &serde_json::json!({"email": "a@x.com"})) .unwrap(); std::thread::sleep(std::time::Duration::from_millis(5)); let t_mid = wall_ms(); std::thread::sleep(std::time::Duration::from_millis(5)); engine - .put("c", "u1", &serde_json::json!({"email": "b@x.com"})) + .put("c", &key(1), &serde_json::json!({"email": "b@x.com"})) .unwrap(); // Past: "a@x.com" → u1 at t_mid. let ids_a_mid = sparse .versioned_index_lookup_as_of(0, 1, "c", "$.email", "a@x.com", Some(t_mid)) .unwrap(); - assert_eq!(ids_a_mid, vec!["u1"]); + assert_eq!(ids_a_mid, vec![key(1).to_string()]); // Current: "b@x.com" → u1. let ids_b_now = sparse .versioned_index_lookup_as_of(0, 1, "c", "$.email", "b@x.com", None) .unwrap(); - assert_eq!(ids_b_now, vec!["u1"]); + assert_eq!(ids_b_now, vec![key(1).to_string()]); // After delete → no current entry for b@x.com either. - engine.delete("c", "u1").unwrap(); + engine.delete("c", &key(1)).unwrap(); let ids_b_after = sparse .versioned_index_lookup_as_of(0, 1, "c", "$.email", "b@x.com", None) .unwrap(); @@ -145,18 +156,22 @@ fn secondary_index_reflects_each_version_independently() { let ids_a_still = sparse .versioned_index_lookup_as_of(0, 1, "c", "$.email", "a@x.com", Some(t_mid)) .unwrap(); - assert_eq!(ids_a_still, vec!["u1"]); + assert_eq!(ids_a_still, vec![key(1).to_string()]); } #[test] fn re_put_after_tombstone_is_a_live_resurrection() { let (sparse, _d) = open(); let engine = register(&sparse); - engine.put("c", "k", &serde_json::json!({"v": 1})).unwrap(); - engine.delete("c", "k").unwrap(); - assert!(engine.get("c", "k").unwrap().is_none()); - engine.put("c", "k", &serde_json::json!({"v": 2})).unwrap(); - let now = engine.get("c", "k").unwrap().unwrap(); + engine + .put("c", &key(1), &serde_json::json!({"v": 1})) + .unwrap(); + engine.delete("c", &key(1)).unwrap(); + assert!(engine.get("c", &key(1)).unwrap().is_none()); + engine + .put("c", &key(1), &serde_json::json!({"v": 2})) + .unwrap(); + let now = engine.get("c", &key(1)).unwrap().unwrap(); assert_eq!(now["v"], 2); } diff --git a/nodedb/tests/inproc/cases/document_bitemporal_store.rs b/nodedb/tests/inproc/cases/document_bitemporal_store.rs index d03778495..9a5cadc21 100644 --- a/nodedb/tests/inproc/cases/document_bitemporal_store.rs +++ b/nodedb/tests/inproc/cases/document_bitemporal_store.rs @@ -10,9 +10,14 @@ //! Ceiling-at-head, and arbitrary system-time cutoffs via //! `versioned_get_as_of`. -use nodedb::engine::document::store::{CollectionConfig, DocumentEngine}; +use nodedb::engine::document::store::{CollectionConfig, DocumentEngine, StorageKey}; use nodedb::engine::sparse::btree::SparseEngine; use nodedb::engine::sparse::btree_versioned::VersionedScanParams; +use nodedb_types::Surrogate; + +fn key(n: u32) -> StorageKey { + StorageKey::for_surrogate(Surrogate::new(n)) +} fn open() -> (SparseEngine, tempfile::TempDir) { let dir = tempfile::tempdir().unwrap(); @@ -27,10 +32,10 @@ fn bitemporal_put_and_current_get_roundtrip() { engine.register_collection(CollectionConfig::new("users").with_bitemporal(true)); engine - .put("users", "u1", &serde_json::json!({"name": "Alice"})) + .put("users", &key(1), &serde_json::json!({"name": "Alice"})) .unwrap(); - let got = engine.get("users", "u1").unwrap().unwrap(); + let got = engine.get("users", &key(1)).unwrap().unwrap(); assert_eq!(got["name"], "Alice"); } @@ -40,12 +45,14 @@ fn bitemporal_delete_appends_tombstone_so_current_get_is_none() { let mut engine = DocumentEngine::new(&sparse, 0, 1); engine.register_collection(CollectionConfig::new("c").with_bitemporal(true)); - engine.put("c", "a", &serde_json::json!({"v": 1})).unwrap(); - assert!(engine.get("c", "a").unwrap().is_some()); - let removed = engine.delete("c", "a").unwrap(); + engine + .put("c", &key(1), &serde_json::json!({"v": 1})) + .unwrap(); + assert!(engine.get("c", &key(1)).unwrap().is_some()); + let removed = engine.delete("c", &key(1)).unwrap(); assert!(removed, "delete should report row was live"); assert!( - engine.get("c", "a").unwrap().is_none(), + engine.get("c", &key(1)).unwrap().is_none(), "after tombstone current-state get is None" ); } @@ -58,15 +65,19 @@ fn bitemporal_multiple_puts_retain_history_via_versioned_get_as_of() { // Three puts create three versions. The versioned API is the way // callers query history — engine.get() returns current state only. - engine.put("c", "k", &serde_json::json!({"v": 1})).unwrap(); + engine + .put("c", &key(1), &serde_json::json!({"v": 1})) + .unwrap(); std::thread::sleep(std::time::Duration::from_millis(5)); let t_mid = wall_ms(); std::thread::sleep(std::time::Duration::from_millis(5)); - engine.put("c", "k", &serde_json::json!({"v": 2})).unwrap(); + engine + .put("c", &key(1), &serde_json::json!({"v": 2})) + .unwrap(); // A cutoff before the second write should surface v=1. let body = sparse - .versioned_get_as_of(0, 1, "c", "k", Some(t_mid), None) + .versioned_get_as_of(0, 1, "c", &key(1).to_string(), Some(t_mid), None) .unwrap() .expect("version at cutoff"); let val: serde_json::Value = { @@ -76,7 +87,7 @@ fn bitemporal_multiple_puts_retain_history_via_versioned_get_as_of() { assert_eq!(val["v"], 1); // Current state reflects the latest write. - let now = engine.get("c", "k").unwrap().unwrap(); + let now = engine.get("c", &key(1)).unwrap().unwrap(); assert_eq!(now["v"], 2); } @@ -85,7 +96,9 @@ fn non_bitemporal_collection_uses_legacy_storage() { let (sparse, _d) = open(); let mut engine = DocumentEngine::new(&sparse, 0, 1); engine.register_collection(CollectionConfig::new("c")); - engine.put("c", "k", &serde_json::json!({"v": 1})).unwrap(); + engine + .put("c", &key(1), &serde_json::json!({"v": 1})) + .unwrap(); // The versioned table must be empty for this collection because // writes went to the legacy path. @@ -108,7 +121,7 @@ fn non_bitemporal_collection_uses_legacy_storage() { "non-bitemporal collection should not populate the versioned table" ); // And the legacy read still works. - let got = engine.get("c", "k").unwrap().unwrap(); + let got = engine.get("c", &key(1)).unwrap().unwrap(); assert_eq!(got["v"], 1); } diff --git a/nodedb/tests/wire/cases/enforcement_period_lock.rs b/nodedb/tests/wire/cases/enforcement_period_lock.rs new file mode 100644 index 000000000..d2056eed8 --- /dev/null +++ b/nodedb/tests/wire/cases/enforcement_period_lock.rs @@ -0,0 +1,368 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Integration tests for the PERIOD LOCK constraint across every DML shape +//! that can rewrite a row: `UPDATE` (point), `MERGE`, `INSERT ... SELECT`, +//! `BulkUpdate`, and `BulkDelete`. +//! +//! A period lock refuses a write whose row names a period (via +//! `config.period_column`) whose reference-table status is not in the +//! declared allowed set. Every collection here uses `document_schemaless` +//! so its stored rows stay in MessagePack, matching the format the +//! Data-Plane period-lock check reads. + +use crate::harness::TestServer; + +/// Declare a reference table and a period-locked entries collection, joined +/// on `fiscal_period` / `period_key`, allowing writes only in status `OPEN`. +async fn setup(server: &TestServer, periods: &str, entries: &str) { + server + .exec(&format!( + "CREATE COLLECTION {periods} (period_key TEXT PRIMARY KEY, status TEXT) \ + WITH (engine='document_schemaless')" + )) + .await + .unwrap(); + server + .exec(&format!( + "CREATE COLLECTION {entries} (id TEXT PRIMARY KEY, fiscal_period TEXT, amount INT) \ + WITH (engine='document_schemaless')" + )) + .await + .unwrap(); + server + .exec(&format!( + "ALTER COLLECTION {entries} ADD PERIOD LOCK ON fiscal_period \ + REFERENCES {periods}(period_key) STATUS status \ + ALLOW WRITE WHEN status IN ('OPEN')" + )) + .await + .unwrap(); +} + +async fn seed_period(server: &TestServer, periods: &str, key: &str, status: &str) { + server + .exec(&format!( + "INSERT INTO {periods} (period_key, status) VALUES ('{key}', '{status}')" + )) + .await + .unwrap(); +} + +fn assert_period_locked(result: &Result<(), String>) { + let message = match result { + Ok(()) => panic!("expected the write to be refused as a period lock violation"), + Err(message) => message, + }; + assert!( + message.to_lowercase().contains("period locked"), + "expected a period-locked refusal, got: {message}" + ); +} + +/// A misconfigured `status_column` refuses as a config error — distinct from +/// both a locked period and an admitted write. +fn assert_period_lock_misconfigured(result: &Result<(), String>) { + let message = match result { + Ok(()) => panic!("expected the write to be refused as a period-lock config error"), + Err(message) => message, + }; + let lower = message.to_lowercase(); + assert!( + lower.contains("misconfigured"), + "expected a period-lock misconfiguration refusal, got: {message}" + ); + assert!( + !lower.contains("period locked"), + "a misconfigured status column must not be reported as a locked period, got: {message}" + ); +} + +/// A point `UPDATE` that assigns the period column into a CLOSED period is +/// refused, even though the row started in an OPEN one. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn point_update_moving_a_row_into_a_locked_period_is_refused() { + let server = TestServer::start().await; + setup(&server, "plk_upd_ref_periods", "plk_upd_ref_entries").await; + seed_period(&server, "plk_upd_ref_periods", "P-OPEN", "OPEN").await; + seed_period(&server, "plk_upd_ref_periods", "P-CLOSED", "CLOSED").await; + + server + .exec("INSERT INTO plk_upd_ref_entries (id, fiscal_period, amount) VALUES ('e1', 'P-OPEN', 100)") + .await + .expect("seeding into an open period must succeed"); + + let result = server + .exec("UPDATE plk_upd_ref_entries SET fiscal_period = 'P-CLOSED' WHERE id = 'e1'") + .await; + assert_period_locked(&result); + + let rows = server + .query_text("SELECT fiscal_period FROM plk_upd_ref_entries WHERE id = 'e1'") + .await + .unwrap(); + assert_eq!( + rows, + vec!["P-OPEN".to_string()], + "a refused update must leave the row in its original period" + ); +} + +/// A point `UPDATE` that moves a row between two OPEN periods is admitted. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn point_update_moving_a_row_into_an_open_period_is_admitted() { + let server = TestServer::start().await; + setup(&server, "plk_upd_ok_periods", "plk_upd_ok_entries").await; + seed_period(&server, "plk_upd_ok_periods", "P-OPEN-1", "OPEN").await; + seed_period(&server, "plk_upd_ok_periods", "P-OPEN-2", "OPEN").await; + + server + .exec("INSERT INTO plk_upd_ok_entries (id, fiscal_period, amount) VALUES ('e1', 'P-OPEN-1', 100)") + .await + .expect("seeding into an open period must succeed"); + + server + .exec("UPDATE plk_upd_ok_entries SET fiscal_period = 'P-OPEN-2' WHERE id = 'e1'") + .await + .expect("moving a row between two open periods must be admitted"); + + let rows = server + .query_text("SELECT fiscal_period FROM plk_upd_ok_entries WHERE id = 'e1'") + .await + .unwrap(); + assert_eq!(rows, vec!["P-OPEN-2".to_string()]); +} + +/// A `MERGE ... WHEN NOT MATCHED THEN INSERT` into an OPEN period is +/// admitted — the orchestrator must resolve the period-lock target before +/// it dispatches the apply pass, or the row is refused as an unknown period. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn merge_insert_into_an_open_period_is_admitted() { + let server = TestServer::start().await; + setup(&server, "plk_merge_periods", "plk_merge_entries").await; + seed_period(&server, "plk_merge_periods", "P-OPEN", "OPEN").await; + + server + .exec( + "CREATE COLLECTION plk_merge_source (id TEXT PRIMARY KEY, fiscal_period TEXT, amount INT) \ + WITH (engine='document_schemaless')", + ) + .await + .unwrap(); + server + .exec( + "INSERT INTO plk_merge_source (id, fiscal_period, amount) VALUES ('m1', 'P-OPEN', 50)", + ) + .await + .unwrap(); + + server + .exec( + "MERGE INTO plk_merge_entries t USING plk_merge_source s ON t.id = s.id \ + WHEN NOT MATCHED THEN INSERT (id, fiscal_period, amount) \ + VALUES (s.id, s.fiscal_period, s.amount)", + ) + .await + .expect("a MERGE insert into an open period must be admitted"); + + let rows = server + .query_text("SELECT amount FROM plk_merge_entries WHERE id = 'm1'") + .await + .unwrap(); + assert_eq!(rows, vec!["50".to_string()]); +} + +/// An `INSERT ... SELECT` into an OPEN period is admitted — the orchestrator +/// resolves the period-lock target for each paged `BatchInsert` it ships. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn insert_select_into_an_open_period_is_admitted() { + let server = TestServer::start().await; + setup(&server, "plk_is_periods", "plk_is_entries").await; + seed_period(&server, "plk_is_periods", "P-OPEN", "OPEN").await; + + server + .exec( + "CREATE COLLECTION plk_is_source (id TEXT PRIMARY KEY, fiscal_period TEXT, amount INT) \ + WITH (engine='document_schemaless')", + ) + .await + .unwrap(); + server + .exec("INSERT INTO plk_is_source (id, fiscal_period, amount) VALUES ('s1', 'P-OPEN', 77)") + .await + .unwrap(); + + server + .exec("INSERT INTO plk_is_entries SELECT * FROM plk_is_source") + .await + .expect("an INSERT ... SELECT into an open period must be admitted"); + + let rows = server + .query_text("SELECT amount FROM plk_is_entries WHERE id = 's1'") + .await + .unwrap(); + assert_eq!(rows, vec!["77".to_string()]); +} + +/// A `BulkUpdate` (`UPDATE ... WHERE `) touching rows in +/// an OPEN period is admitted. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn bulk_update_touching_an_open_period_is_admitted() { + let server = TestServer::start().await; + setup(&server, "plk_bu_ok_periods", "plk_bu_ok_entries").await; + seed_period(&server, "plk_bu_ok_periods", "P-OPEN", "OPEN").await; + + server + .exec( + "INSERT INTO plk_bu_ok_entries (id, fiscal_period, amount) VALUES ('e1', 'P-OPEN', 10)", + ) + .await + .unwrap(); + server + .exec( + "INSERT INTO plk_bu_ok_entries (id, fiscal_period, amount) VALUES ('e2', 'P-OPEN', 20)", + ) + .await + .unwrap(); + + server + .exec("UPDATE plk_bu_ok_entries SET amount = amount + 1 WHERE fiscal_period = 'P-OPEN'") + .await + .expect("a bulk update over an open period must be admitted"); + + let rows = server + .query_text("SELECT amount FROM plk_bu_ok_entries ORDER BY id") + .await + .unwrap(); + assert_eq!(rows, vec!["11".to_string(), "21".to_string()]); +} + +/// A `BulkUpdate` touching rows whose period was closed after they were +/// written is refused. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn bulk_update_touching_a_locked_period_is_refused() { + let server = TestServer::start().await; + setup(&server, "plk_bu_lock_periods", "plk_bu_lock_entries").await; + seed_period(&server, "plk_bu_lock_periods", "P-SOON-CLOSED", "OPEN").await; + + server + .exec("INSERT INTO plk_bu_lock_entries (id, fiscal_period, amount) VALUES ('e1', 'P-SOON-CLOSED', 10)") + .await + .expect("seeding while the period is still open must succeed"); + server + .exec("INSERT INTO plk_bu_lock_entries (id, fiscal_period, amount) VALUES ('e2', 'P-SOON-CLOSED', 20)") + .await + .unwrap(); + + server + .exec("UPDATE plk_bu_lock_periods SET status = 'CLOSED' WHERE period_key = 'P-SOON-CLOSED'") + .await + .expect("closing the reference period must succeed"); + + let result = server + .exec("UPDATE plk_bu_lock_entries SET amount = amount + 1 WHERE fiscal_period = 'P-SOON-CLOSED'") + .await; + assert_period_locked(&result); + + let rows = server + .query_text("SELECT amount FROM plk_bu_lock_entries ORDER BY id") + .await + .unwrap(); + assert_eq!( + rows, + vec!["10".to_string(), "20".to_string()], + "a refused bulk update must leave every matched row unchanged" + ); +} + +/// A `BulkDelete` removing rows from an OPEN period is admitted. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn bulk_delete_from_an_open_period_is_admitted() { + let server = TestServer::start().await; + setup(&server, "plk_bd_ok_periods", "plk_bd_ok_entries").await; + seed_period(&server, "plk_bd_ok_periods", "P-OPEN", "OPEN").await; + + server + .exec( + "INSERT INTO plk_bd_ok_entries (id, fiscal_period, amount) VALUES ('e1', 'P-OPEN', 10)", + ) + .await + .unwrap(); + + server + .exec("DELETE FROM plk_bd_ok_entries WHERE fiscal_period = 'P-OPEN'") + .await + .expect("a bulk delete over an open period must be admitted"); + + let rows = server + .query_text("SELECT id FROM plk_bd_ok_entries") + .await + .unwrap(); + assert!(rows.is_empty()); +} + +/// A `BulkDelete` removing rows whose period was closed after they were +/// written is refused. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn bulk_delete_from_a_locked_period_is_refused() { + let server = TestServer::start().await; + setup(&server, "plk_bd_lock_periods", "plk_bd_lock_entries").await; + seed_period(&server, "plk_bd_lock_periods", "P-SOON-CLOSED", "OPEN").await; + + server + .exec("INSERT INTO plk_bd_lock_entries (id, fiscal_period, amount) VALUES ('e1', 'P-SOON-CLOSED', 10)") + .await + .expect("seeding while the period is still open must succeed"); + + server + .exec("UPDATE plk_bd_lock_periods SET status = 'CLOSED' WHERE period_key = 'P-SOON-CLOSED'") + .await + .expect("closing the reference period must succeed"); + + let result = server + .exec("DELETE FROM plk_bd_lock_entries WHERE fiscal_period = 'P-SOON-CLOSED'") + .await; + assert_period_locked(&result); + + let rows = server + .query_text("SELECT id FROM plk_bd_lock_entries") + .await + .unwrap(); + assert_eq!( + rows, + vec!["e1".to_string()], + "a refused bulk delete must leave every matched row in place" + ); +} + +/// A reference row that resolves but carries no `status` column at all is a +/// misconfigured `status_column`, not a locked period — a typo in the +/// configured column name must not read as every period being closed. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_reference_row_missing_the_status_column_is_a_config_error() { + let server = TestServer::start().await; + setup(&server, "plk_cfg_periods", "plk_cfg_entries").await; + + // Seed the reference row through the real DDL surface, omitting `status` + // entirely — a schemaless collection admits the row with no such field. + server + .exec("INSERT INTO plk_cfg_periods (period_key) VALUES ('P-NO-STATUS')") + .await + .expect("seeding a reference row without a status column must succeed"); + + let result = server + .exec( + "INSERT INTO plk_cfg_entries (id, fiscal_period, amount) \ + VALUES ('e1', 'P-NO-STATUS', 10)", + ) + .await; + assert_period_lock_misconfigured(&result); + + let rows = server + .query_text("SELECT id FROM plk_cfg_entries") + .await + .unwrap(); + assert!( + rows.is_empty(), + "a write refused as misconfigured must leave no row behind" + ); +} diff --git a/nodedb/tests/wire/cases/mod.rs b/nodedb/tests/wire/cases/mod.rs index 49316db90..30173a7ee 100644 --- a/nodedb/tests/wire/cases/mod.rs +++ b/nodedb/tests/wire/cases/mod.rs @@ -60,6 +60,7 @@ mod drop_consumer_group_if_exists; mod drop_recreate_bitemporal_no_resurrection; mod drop_rls_policy_if_exists; mod enforcement_balanced; +mod enforcement_period_lock; mod engine_surface_array; mod engine_surface_columnar; mod engine_surface_crdt_document; From cc4aaa553db25b95a867dde18edccad86a512d76 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 12 Sep 2026 05:09:28 +0800 Subject: [PATCH 06/17] refactor(materialized-sum): dedupe resolved targets in O(1) Replace the free function resolve_one_target and its Vec accumulator with a ResolvedTargets struct. Dedup on (target, value) used to scan the accumulated Vec linearly per candidate; ResolvedTargets tracks seen values per target in a HashMap>, so a page or scan that touches the same target row many times resolves it in O(1) instead of O(n). Materialized-sum and period-lock resolution share the same struct. --- .../control/planner/materialized_sum/mod.rs | 2 +- .../planner/materialized_sum/predicate.rs | 18 ++-- .../planner/materialized_sum/resolve.rs | 29 +++---- .../materialized_sum/resolve_target.rs | 83 ++++++++++++++----- .../planner/materialized_sum/stored.rs | 22 ++--- .../planner/period_lock/point_update.rs | 28 +++---- .../control/planner/period_lock/predicate.rs | 37 +++++---- .../control/planner/period_lock/singular.rs | 22 +++-- 8 files changed, 138 insertions(+), 103 deletions(-) diff --git a/nodedb/src/control/planner/materialized_sum/mod.rs b/nodedb/src/control/planner/materialized_sum/mod.rs index 637cef9cb..c4ea272cd 100644 --- a/nodedb/src/control/planner/materialized_sum/mod.rs +++ b/nodedb/src/control/planner/materialized_sum/mod.rs @@ -31,4 +31,4 @@ pub use index::MaterializedSumIndex; pub use resolve::{ resolve_materialized_sum_targets, resolve_sum_targets_for_bodies, source_drives_bindings, }; -pub use resolve_target::resolve_one_target; +pub use resolve_target::ResolvedTargets; diff --git a/nodedb/src/control/planner/materialized_sum/predicate.rs b/nodedb/src/control/planner/materialized_sum/predicate.rs index 17071cc87..767198aef 100644 --- a/nodedb/src/control/planner/materialized_sum/predicate.rs +++ b/nodedb/src/control/planner/materialized_sum/predicate.rs @@ -21,7 +21,7 @@ use nodedb_types::id::TxnId; use super::recon::recon_scan_rows; use super::resolve::{lookup_join_value, source_drives_bindings}; -use super::resolve_target::resolve_one_target; +use super::resolve_target::ResolvedTargets; use super::settle::{ SettleInput, Settlement, co_resident_target_keys, omit_shipped, settle_cross_shard_images, }; @@ -154,23 +154,19 @@ async fn resolve_scanned_rows( database_id: DatabaseId, trace_id: TraceId, ) -> crate::Result> { - let mut resolved: Vec = Vec::new(); + let mut resolved = ResolvedTargets::new(); for binding in bindings.iter() { for join_value in crate::query::binding_join_keys(binding, updates, rows)? { - resolve_one_target( - &mut resolved, - &binding.target_collection, - join_value, - async |v| { + resolved + .resolve(&binding.target_collection, join_value, async |v| { lookup_join_value(state, binding, v, tenant_id, database_id, trace_id) .await .map(Some) - }, - ) - .await?; + }) + .await?; } } - Ok(resolved) + Ok(resolved.into_vec()) } /// The scan inputs of a predicate-driven write, or `None` for every other op. diff --git a/nodedb/src/control/planner/materialized_sum/resolve.rs b/nodedb/src/control/planner/materialized_sum/resolve.rs index 38848361a..c36dc0e61 100644 --- a/nodedb/src/control/planner/materialized_sum/resolve.rs +++ b/nodedb/src/control/planner/materialized_sum/resolve.rs @@ -12,7 +12,7 @@ use nodedb_physical::physical_task::PhysicalTask; use nodedb_types::Surrogate; use super::extract::join_value_from_body; -use super::resolve_target::resolve_one_target; +use super::resolve_target::ResolvedTargets; use super::settle::{ SettleInput, co_resident_target_keys, omit_shipped, settle_cross_shard_images, }; @@ -111,14 +111,14 @@ pub async fn resolve_materialized_sum_targets( resolve_bodies(state, &bindings, bodies, tenant_id, database_id, trace_id) .await? } - None => Vec::new(), + None => ResolvedTargets::new(), }; // A point write that rewrites a stored row reads it here — for the // join values it addresses AND for the pre-image a cross-shard // delta is folded from. One read, one snapshot: settling from a // second read would total a different one. match &stored { - None => (resolved, None), + None => (resolved.into_vec(), None), Some(scope) => { let images = super::stored::extend_with_stored_row( state, @@ -130,6 +130,7 @@ pub async fn resolve_materialized_sum_targets( trace_id, ) .await?; + let mut resolved = resolved.into_vec(); let input = SettleInput { source_collection: collection, images: &images.images, @@ -309,7 +310,11 @@ pub async fn resolve_sum_targets_for_bodies( else { return Ok(Vec::new()); }; - resolve_bodies(state, &bindings, bodies, tenant_id, database_id, trace_id).await + Ok( + resolve_bodies(state, &bindings, bodies, tenant_id, database_id, trace_id) + .await? + .into_vec(), + ) } /// Whether `source_collection` drives any materialized-sum binding. @@ -381,8 +386,8 @@ async fn resolve_bodies( tenant_id: TenantId, database_id: DatabaseId, trace_id: TraceId, -) -> crate::Result> { - let mut resolved: Vec = Vec::new(); +) -> crate::Result { + let mut resolved = ResolvedTargets::new(); for binding in bindings.iter() { for body in bodies { let Some(join_value) = join_value_from_body(body, &binding.join_column) else { @@ -391,17 +396,13 @@ async fn resolve_bodies( // and nothing to add a delta to. continue; }; - resolve_one_target( - &mut resolved, - &binding.target_collection, - join_value, - async |v| { + resolved + .resolve(&binding.target_collection, join_value, async |v| { lookup_join_value(state, binding, v, tenant_id, database_id, trace_id) .await .map(Some) - }, - ) - .await?; + }) + .await?; } } Ok(resolved) diff --git a/nodedb/src/control/planner/materialized_sum/resolve_target.rs b/nodedb/src/control/planner/materialized_sum/resolve_target.rs index 77a8fae9a..d24e13367 100644 --- a/nodedb/src/control/planner/materialized_sum/resolve_target.rs +++ b/nodedb/src/control/planner/materialized_sum/resolve_target.rs @@ -4,32 +4,71 @@ //! period-lock resolution loop performs once per candidate `(target, value)` //! pair. +use std::collections::{HashMap, HashSet}; + use nodedb_physical::physical_plan::ResolvedSumTarget; use nodedb_types::Surrogate; -/// Resolve one `(target, value)` pair into `resolved`, deduping on the pair -/// and skipping the push when `lookup` names no row. -/// -/// One entry per DISTINCT `(target, value)` pair: a page or scan that touches -/// the same target row many times resolves it once, so every caller checks -/// `resolved` before spending a lookup on a value it already holds. +/// The resolution a statement builds up, one entry per DISTINCT +/// `(target, value)` pair. /// -/// `lookup` differs by caller: a period-lock lookup can name no reference row -/// (`Ok(None)` skips the push, leaving the value unresolved for -/// `check_period_lock` to refuse), while a materialized-sum lookup always -/// resolves or fails outright — its `Err` propagates through the `?` below -/// before `Ok(None)` could ever apply. -pub async fn resolve_one_target( - resolved: &mut Vec, - target: &str, - value: String, - lookup: impl AsyncFnOnce(&str) -> crate::Result>, -) -> crate::Result<()> { - if resolved.iter().any(|entry| entry.addresses(target, &value)) { - return Ok(()); +/// A page or scan that touches the same target row many times resolves it +/// once: `seen` answers the dedupe in O(1) per candidate, and `entries` keeps +/// first-resolution order, which is the order the plan carries. +#[derive(Default)] +pub struct ResolvedTargets { + entries: Vec, + /// Every value a lookup already ran for, per target. Values the lookup + /// left unresolved count too, so a repeated value never re-runs it. Keyed + /// target-first so the membership check borrows both halves. + seen: HashMap>, +} + +impl ResolvedTargets { + pub fn new() -> Self { + Self::default() } - if let Some(surrogate) = lookup(&value).await? { - resolved.push(ResolvedSumTarget::new(target, value, surrogate)); + + /// The resolved entries in first-resolution order. + pub fn into_vec(self) -> Vec { + self.entries + } + + /// Resolve one `(target, value)` pair, skipping the push when `lookup` + /// names no row. + /// + /// `lookup` differs by caller: a period-lock lookup can name no reference + /// row (`Ok(None)` skips the push, leaving the value unresolved for + /// `check_period_lock` to refuse), while a materialized-sum lookup always + /// resolves or fails outright — its `Err` propagates through the `?` + /// below before `Ok(None)` could ever apply. + pub async fn resolve( + &mut self, + target: &str, + value: String, + lookup: impl AsyncFnOnce(&str) -> crate::Result>, + ) -> crate::Result<()> { + if self + .seen + .get(target) + .is_some_and(|values| values.contains(value.as_str())) + { + return Ok(()); + } + let surrogate = lookup(&value).await?; + match self.seen.get_mut(target) { + Some(values) => { + values.insert(value.clone()); + } + None => { + self.seen + .insert(target.to_string(), HashSet::from([value.clone()])); + } + } + if let Some(surrogate) = surrogate { + self.entries + .push(ResolvedSumTarget::new(target, value, surrogate)); + } + Ok(()) } - Ok(()) } diff --git a/nodedb/src/control/planner/materialized_sum/stored.rs b/nodedb/src/control/planner/materialized_sum/stored.rs index 3701a7f88..84ede6e42 100644 --- a/nodedb/src/control/planner/materialized_sum/stored.rs +++ b/nodedb/src/control/planner/materialized_sum/stored.rs @@ -22,14 +22,12 @@ use std::sync::Arc; -use nodedb_physical::physical_plan::{ - DocumentOp, MaterializedSumBinding, ResolvedSumTarget, UpdateValue, -}; +use nodedb_physical::physical_plan::{DocumentOp, MaterializedSumBinding, UpdateValue}; use nodedb_types::Surrogate; use super::recon::recon_point_row; use super::resolve::lookup_join_value; -use super::resolve_target::resolve_one_target; +use super::resolve_target::ResolvedTargets; use crate::control::state::SharedState; use crate::types::{DatabaseId, Lsn, TenantId, TraceId}; @@ -191,7 +189,7 @@ pub(super) async fn extend_with_stored_row( state: &SharedState, bindings: &Arc>, scope: &StoredRowScope<'_>, - resolved: &mut Vec, + resolved: &mut ResolvedTargets, tenant_id: TenantId, database_id: DatabaseId, trace_id: TraceId, @@ -234,20 +232,16 @@ pub(super) async fn extend_with_stored_row( // body-driven resolution: a write whose old and new join keys are the // same resolves that target once, while two bindings that share a join // column and name different targets each keep their own entry. - // `resolve_one_target` enforces the dedupe against `resolved`. + // `ResolvedTargets` enforces the dedupe. for binding in bindings.iter() { for join_value in crate::query::binding_join_keys(binding, &[], &rows)? { - resolve_one_target( - resolved, - &binding.target_collection, - join_value, - async |v| { + resolved + .resolve(&binding.target_collection, join_value, async |v| { lookup_join_value(state, binding, v, tenant_id, database_id, trace_id) .await .map(Some) - }, - ) - .await?; + }) + .await?; } } Ok(outcome) diff --git a/nodedb/src/control/planner/period_lock/point_update.rs b/nodedb/src/control/planner/period_lock/point_update.rs index 8f641a50c..e64fb8c1f 100644 --- a/nodedb/src/control/planner/period_lock/point_update.rs +++ b/nodedb/src/control/planner/period_lock/point_update.rs @@ -6,8 +6,8 @@ use nodedb_physical::physical_plan::{ResolvedSumTarget, UpdateValue}; use super::lookup::{PeriodLockScope, lookup_period_surrogate}; +use crate::control::planner::materialized_sum::ResolvedTargets; use crate::control::planner::materialized_sum::recon::recon_point_row; -use crate::control::planner::materialized_sum::resolve_one_target; use crate::control::security::catalog::PeriodLockDef; /// Resolve the period value(s) a `PointUpdate` addresses: the stored row's @@ -39,22 +39,19 @@ pub(super) async fn resolve_update_period_values( }; let new_row = crate::query::apply_update_assignments(&old_row, updates)?; - // `resolve_one_target` dedupes on `(target, value)` itself, so the old - // and new images resolve through the same call whether or not they carry - // the same period value. - let mut resolved: Vec = Vec::new(); + // `ResolvedTargets` dedupes on `(target, value)` itself, so the old and + // new images resolve through the same call whether or not they carry the + // same period value. + let mut resolved = ResolvedTargets::new(); for row in [&old_row, &new_row] { let Some(value) = row.get(def.period_column.as_str()).and_then(|v| v.as_str()) else { continue; }; - // No reference row names this period — `resolve_one_target` leaves - // it unresolved so `check_period_lock` reports the unknown-period + // No reference row names this period — the resolve leaves it + // unresolved so `check_period_lock` reports the unknown-period // refusal for it. - resolve_one_target( - &mut resolved, - &def.ref_table, - value.to_string(), - async |key| { + resolved + .resolve(&def.ref_table, value.to_string(), async |key| { lookup_period_surrogate( scope.state, &def.ref_table, @@ -64,9 +61,8 @@ pub(super) async fn resolve_update_period_values( scope.trace_id, ) .await - }, - ) - .await?; + }) + .await?; } - Ok(resolved) + Ok(resolved.into_vec()) } diff --git a/nodedb/src/control/planner/period_lock/predicate.rs b/nodedb/src/control/planner/period_lock/predicate.rs index b62bd8d7a..ccf79a68d 100644 --- a/nodedb/src/control/planner/period_lock/predicate.rs +++ b/nodedb/src/control/planner/period_lock/predicate.rs @@ -7,8 +7,8 @@ use nodedb_physical::physical_plan::{ResolvedSumTarget, UpdateValue}; use super::lookup::{PeriodLockScope, lookup_period_surrogate}; +use crate::control::planner::materialized_sum::ResolvedTargets; use crate::control::planner::materialized_sum::recon::recon_scan_rows; -use crate::control::planner::materialized_sum::resolve_one_target; use crate::control::security::catalog::PeriodLockDef; /// What a predicate-driven statement does to each row it matches — decides @@ -45,10 +45,10 @@ pub(super) async fn resolve_predicate_period_values( ) .await?; - // `resolve_one_target` dedupes on `(target, value)` itself, so every - // row's pre- and post-image resolves through the same call regardless of - // how many rows or images share a period value. - let mut resolved: Vec = Vec::new(); + // `ResolvedTargets` dedupes on `(target, value)` itself, so every row's + // pre- and post-image resolves through the same call regardless of how + // many rows or images share a period value. + let mut resolved = ResolvedTargets::new(); for row in &read.rows { resolve_period_value(scope, def, row, &mut resolved).await?; if matches!(effect, PeriodLockEffect::Assign) { @@ -56,7 +56,7 @@ pub(super) async fn resolve_predicate_period_values( resolve_period_value(scope, def, &new_row, &mut resolved).await?; } } - Ok(resolved) + Ok(resolved.into_vec()) } /// Resolve one row's period value into `resolved`, when it carries the @@ -67,21 +67,22 @@ async fn resolve_period_value( scope: &PeriodLockScope<'_>, def: &PeriodLockDef, row: &serde_json::Value, - resolved: &mut Vec, + resolved: &mut ResolvedTargets, ) -> crate::Result<()> { let Some(value) = row.get(def.period_column.as_str()).and_then(|v| v.as_str()) else { return Ok(()); }; - resolve_one_target(resolved, &def.ref_table, value.to_string(), async |key| { - lookup_period_surrogate( - scope.state, - &def.ref_table, - key, - scope.tenant_id, - scope.database_id, - scope.trace_id, - ) + resolved + .resolve(&def.ref_table, value.to_string(), async |key| { + lookup_period_surrogate( + scope.state, + &def.ref_table, + key, + scope.tenant_id, + scope.database_id, + scope.trace_id, + ) + .await + }) .await - }) - .await } diff --git a/nodedb/src/control/planner/period_lock/singular.rs b/nodedb/src/control/planner/period_lock/singular.rs index a56207f4e..e7ddc91c2 100644 --- a/nodedb/src/control/planner/period_lock/singular.rs +++ b/nodedb/src/control/planner/period_lock/singular.rs @@ -7,7 +7,7 @@ use nodedb_physical::physical_plan::{DocumentOp, ResolvedSumTarget}; use super::lookup::lookup_period_surrogate; use crate::control::planner::materialized_sum::recon::recon_point_row; -use crate::control::planner::materialized_sum::{join_value_from_body, resolve_one_target}; +use crate::control::planner::materialized_sum::{ResolvedTargets, join_value_from_body}; use crate::control::security::catalog::PeriodLockDef; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId, TraceId}; @@ -93,16 +93,24 @@ pub(super) async fn resolve_batch_period_values( database_id: DatabaseId, trace_id: TraceId, ) -> crate::Result> { - let mut resolved: Vec = Vec::new(); + let mut resolved = ResolvedTargets::new(); for body in bodies { let Some(period_key) = join_value_from_body(body, &def.period_column) else { continue; }; - resolve_one_target(&mut resolved, &def.ref_table, period_key, async |key| { - lookup_period_surrogate(state, &def.ref_table, key, tenant_id, database_id, trace_id) + resolved + .resolve(&def.ref_table, period_key, async |key| { + lookup_period_surrogate( + state, + &def.ref_table, + key, + tenant_id, + database_id, + trace_id, + ) .await - }) - .await?; + }) + .await?; } - Ok(resolved) + Ok(resolved.into_vec()) } From 3b0c9898dcae9f12ccdd368b17c908e8383c338b Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 12 Sep 2026 05:09:42 +0800 Subject: [PATCH 07/17] fix(document): remove the document scan's engine fallback A current-mode document fetch no longer falls back to scan_collection when the sparse store holds nothing for the collection: it now only ever returns sparse rows, each keyed by its rendered storage key, dropping the Fetched/ RowOrigin distinction between sparse and foreign rows entirely. That fallback was the only thing letting VERIFY_HASH_CHAIN, TEMPORAL_LOOKUP, CONVERT_CURRENCY, VERIFY_BALANCE, BALANCE_AS_OF, and CREATE GRAPH INDEX answer over a KV or columnar-family collection at all, silently over zero rows. Each now fails closed via a new CollectionReadGate::require_document_engine check (SQLSTATE 0A000), or the equivalent check in CREATE GRAPH INDEX, instead of reporting an empty result as if it were a true one. Native dispatch's build_scan now routes a plain collection scan to the collection's own engine (KV, timeseries, columnar, spatial) instead of always emitting a document scan, and collection_type propagates catalog errors and resolves against the caller's database id instead of the default one. --- .../native/dispatch/plan_builder/document.rs | 21 +- .../native/dispatch/plan_builder/helpers.rs | 16 +- .../neutral/query_functions/balance_as_of.rs | 2 + .../convert_currency_lookup.rs | 1 + .../query_functions/temporal_lookup.rs | 1 + .../neutral/query_functions/verify_balance.rs | 2 + .../query_functions/verify_hash_chain.rs | 1 + .../server/shared/ddl/neutral/read_gate.rs | 38 +++- .../ddl/neutral/tree_ops/create_index.rs | 16 +- .../data/executor/dispatch/document_admit.rs | 2 +- .../executor/handlers/document/read/fetch.rs | 191 +++++------------- .../handlers/document/read/fetch_types.rs | 31 ++- .../executor/handlers/document/read/mod.rs | 5 +- .../executor/handlers/document/read/scan.rs | 49 ++--- .../test_cross_type_join/basic_scans.rs | 95 +-------- .../test_cross_type_join/multi_core_joins.rs | 48 ++--- nodedb/tests/wire/cases/mod.rs | 1 + .../wire/cases/query_function_engine_gate.rs | 113 +++++++++++ 18 files changed, 297 insertions(+), 336 deletions(-) create mode 100644 nodedb/tests/wire/cases/query_function_engine_gate.rs diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/document.rs b/nodedb/src/control/server/native/dispatch/plan_builder/document.rs index b89a0a6ea..da7ecdcc5 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/document.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/document.rs @@ -20,7 +20,7 @@ pub(crate) fn build_point_get( collection: &str, ) -> crate::Result { let doc_id = require_doc_id(fields)?; - match collection_type(ctx, collection) { + match collection_type(ctx, collection)? { Some(CollectionType::KeyValue(_)) => Ok(PhysicalPlan::Kv(KvOp::Get { collection: QualifiedCollection::new(ctx.database_id(), collection), key: doc_id.into_bytes(), @@ -66,7 +66,7 @@ pub(crate) fn build_point_put( ) -> crate::Result { let doc_id = require_doc_id(fields)?; let value = fields.data.clone().unwrap_or_default(); - match collection_type(ctx, collection) { + match collection_type(ctx, collection)? { Some(CollectionType::KeyValue(_)) => { let key = doc_id.into_bytes(); let surrogate = ctx.state.surrogate_assigner.assign( @@ -135,7 +135,7 @@ pub(crate) fn build_point_delete( collection: &str, ) -> crate::Result { let doc_id = require_doc_id(fields)?; - match collection_type(ctx, collection) { + match collection_type(ctx, collection)? { Some(CollectionType::KeyValue(_)) => Ok(PhysicalPlan::Kv(KvOp::Delete { collection: QualifiedCollection::new(ctx.database_id(), collection), keys: vec![doc_id.into_bytes()], @@ -284,11 +284,26 @@ pub(crate) fn build_update( })) } +/// A collection scan, routed by the collection's engine like the point ops +/// above: a document scan reads the sparse store only, so every other engine +/// takes its own scan builder. A spatial collection's plain scan is the +/// columnar scan; the geometry query is `SpatialScan`. pub(crate) fn build_scan( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, ) -> crate::Result { + match collection_type(ctx, collection)? { + Some(CollectionType::KeyValue(_)) => return super::kv::build_scan(ctx, fields, collection), + Some(CollectionType::Columnar(ColumnarProfile::Timeseries { .. })) => { + return super::timeseries::build_scan(ctx, fields, collection); + } + Some(CollectionType::Columnar(ColumnarProfile::Plain)) + | Some(CollectionType::Columnar(ColumnarProfile::Spatial { .. })) => { + return super::columnar::build_scan(ctx, fields, collection); + } + Some(CollectionType::Document(_)) | None => {} + } let limit = fields.limit.unwrap_or(1000) as usize; let filters = fields.filters.clone().unwrap_or_default(); Ok(PhysicalPlan::Document(DocumentOp::Scan { diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/helpers.rs b/nodedb/src/control/server/native/dispatch/plan_builder/helpers.rs index 8c38cc532..78a5976cf 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/helpers.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/helpers.rs @@ -2,28 +2,26 @@ //! Shared helpers used across per-engine plan builders. -use nodedb_types::DatabaseId; use nodedb_types::protocol::TextFields; use super::super::DispatchCtx; /// Single catalog lookup returning the collection's storage type. /// -/// Returns `None` when: no catalog available, collection not found, -/// or catalog read error. Callers treat `None` as "default to document". +/// `Ok(None)` means the catalog holds no such collection; callers treat that +/// as "default to document". A catalog read error propagates. pub(in crate::control::server::native::dispatch) fn collection_type( ctx: &DispatchCtx<'_>, collection: &str, -) -> Option { +) -> crate::Result> { let catalog = ctx.state.credentials.catalog(); - let coll = catalog + Ok(catalog .get_collection( - DatabaseId::DEFAULT, + ctx.database_id(), ctx.identity.tenant_id.as_u64(), collection, - ) - .ok()??; - Some(coll.collection_type.clone()) + )? + .map(|coll| coll.collection_type)) } /// `collection`'s DDL-declared `PRIMARY KEY` column name, for the apply-time diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs index b48cc2eae..221924de9 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs @@ -47,6 +47,7 @@ pub async fn balance_as_of( // redaction rule on that column has no honest answer — masking it would // report a number no row holds. let gate = CollectionReadGate::open(state, identity, database_id, &collection)?; + gate.require_document_engine(&collection, "BALANCE_AS_OF")?; gate.refuse_if_field_redacted(&collection, &column, "the as-of balance")?; // Read current balance from the target document. @@ -109,6 +110,7 @@ pub async fn balance_as_of( // `value_expr` can name any of its columns, so a redaction rule anywhere on // it is refused rather than silently summed over hidden values. gate.authorize(&mat_def.source_collection)?; + gate.require_document_engine(&mat_def.source_collection, "BALANCE_AS_OF")?; gate.refuse_if_any_redaction(&mat_def.source_collection, "the as-of balance")?; // Scan the source collection for rows where join_column = key AND created_at > as_of. diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs index 95c802717..d96d7df50 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs @@ -65,6 +65,7 @@ pub async fn convert_currency_lookup( // so a redaction rule on that column is refused rather than answered with a // figure derived from a value the caller may not see. let gate = CollectionReadGate::open(state, identity, database_id, &rate_table)?; + gate.require_document_engine(&rate_table, "CONVERT_CURRENCY")?; gate.refuse_if_field_redacted(&rate_table, &rate_column, "the converted amount")?; // Build the composite key: "{from}/{to}" for the rate table lookup. diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs index 3044d3b80..f5955747d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs @@ -43,6 +43,7 @@ pub async fn temporal_lookup( // authorized, row-filtered, and redacted here — nothing downstream of the // hand-built plan does any of it. let gate = CollectionReadGate::open(state, identity, database_id, &table)?; + gate.require_document_engine(&table, "TEMPORAL_LOOKUP")?; // Scan the table. let vshard = VShardId::from_collection_in_database(database_id, &table); diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs index 73b575ba9..e1b213dfd 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs @@ -44,6 +44,7 @@ pub async fn verify_balance( // rows, so a redaction rule over either side is refused: a count computed // from masked values would call a consistent ledger broken. let gate = CollectionReadGate::open(state, identity, database_id, &collection)?; + gate.require_document_engine(&collection, "VERIFY_BALANCE")?; gate.refuse_if_field_redacted(&collection, &column, "the balance verification")?; // Find the materialized sum definition. @@ -65,6 +66,7 @@ pub async fn verify_balance( }; gate.authorize(&mat_def.source_collection)?; + gate.require_document_engine(&mat_def.source_collection, "VERIFY_BALANCE")?; gate.refuse_if_any_redaction(&mat_def.source_collection, "the balance verification")?; // Scan all target rows. diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs index 4b2471275..4acda9e41 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs @@ -40,6 +40,7 @@ pub async fn verify_hash_chain( // so any redaction rule on the collection is refused: hashing a masked row // would report an intact chain as broken. let gate = CollectionReadGate::open(state, identity, database_id, &collection)?; + gate.require_document_engine(&collection, "VERIFY_HASH_CHAIN")?; gate.refuse_if_any_redaction(&collection, "the hash chain")?; // Scan all documents. diff --git a/nodedb/src/control/server/shared/ddl/neutral/read_gate.rs b/nodedb/src/control/server/shared/ddl/neutral/read_gate.rs index 403dabd28..7d37e20ab 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/read_gate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/read_gate.rs @@ -45,8 +45,11 @@ use super::super::result::DdlError; /// SQLSTATE for a denied RBAC check. const INSUFFICIENT_PRIVILEGE: &str = "42501"; -/// SQLSTATE for a policy this delivery shape cannot express. +/// SQLSTATE for a policy this delivery shape cannot express, or a feature +/// the collection's engine does not carry. const FEATURE_NOT_SUPPORTED: &str = "0A000"; +/// SQLSTATE for a collection the catalog does not hold. +const UNDEFINED_TABLE: &str = "42P01"; fn gate_err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) @@ -168,6 +171,39 @@ impl<'a> CollectionReadGate<'a> { }) } + /// Fail closed unless `collection` is a document collection. + /// + /// For handlers that hand-build a `DocumentOp::Scan` over a caller-named + /// collection: the document scan reads the sparse store only, so a KV or + /// columnar-family collection answers with no rows. `what` names the + /// feature in the refusal. + pub fn require_document_engine(&self, collection: &str, what: &str) -> Result<(), DdlError> { + let stored = self + .state + .credentials + .catalog() + .get_collection( + self.scope.database_id(), + self.tenant_id().as_u64(), + collection, + ) + .map_err(|e| gate_err("XX000", e.to_string()))?; + match stored { + None => Err(gate_err( + UNDEFINED_TABLE, + format!("{what}: collection '{collection}' does not exist"), + )), + Some(stored) if stored.collection_type.is_document() => Ok(()), + Some(stored) => Err(gate_err( + FEATURE_NOT_SUPPORTED, + format!( + "{what} reads document collections; '{collection}' is a {} collection", + stored.collection_type.as_str() + ), + )), + } + } + /// Fail closed when a read policy exists on `collection`. /// /// For plans with no filter slot to inject into, where a policy cannot be diff --git a/nodedb/src/control/server/shared/ddl/neutral/tree_ops/create_index.rs b/nodedb/src/control/server/shared/ddl/neutral/tree_ops/create_index.rs index 0a52b2cb7..dce6deaf5 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/tree_ops/create_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/tree_ops/create_index.rs @@ -79,15 +79,21 @@ pub async fn create_graph_index( let (parent_col, id_col) = parse_edge_columns(sql)?; + // The index is built from a document scan, which reads the sparse store + // only, so a KV or columnar-family collection is refused rather than + // indexed from zero rows. let catalog = state.credentials.catalog(); - if catalog + let stored = catalog .get_collection(database_id, tenant_id.as_u64(), &collection) .map_err(|e| ddl_err("XX000", e.to_string()))? - .is_none() - { + .ok_or_else(|| ddl_err("42P01", format!("collection '{collection}' not found")))?; + if !stored.collection_type.is_document() { return Err(ddl_err( - "42P01", - format!("collection '{collection}' not found"), + "0A000", + format!( + "CREATE GRAPH INDEX reads document collections; '{collection}' is a {} collection", + stored.collection_type.as_str() + ), )); } diff --git a/nodedb/src/data/executor/dispatch/document_admit.rs b/nodedb/src/data/executor/dispatch/document_admit.rs index f489f6450..6bdb388b3 100644 --- a/nodedb/src/data/executor/dispatch/document_admit.rs +++ b/nodedb/src/data/executor/dispatch/document_admit.rs @@ -11,7 +11,7 @@ use nodedb_physical::physical_plan::DocumentOp; use nodedb_types::SystemTimeScope; -use crate::data::executor::handlers::document::read::fetch::DocScanMode; +use crate::data::executor::handlers::document::read::DocScanMode; /// Whether the op mutates stored state. /// diff --git a/nodedb/src/data/executor/handlers/document/read/fetch.rs b/nodedb/src/data/executor/handlers/document/read/fetch.rs index 5d9b18a93..8c09192d9 100644 --- a/nodedb/src/data/executor/handlers/document/read/fetch.rs +++ b/nodedb/src/data/executor/handlers/document/read/fetch.rs @@ -22,15 +22,11 @@ use std::cell::Cell; -use tracing::warn; - use nodedb_types::StorageKey; use super::audit_body::{inject_temporal_columns, strict_audit_body}; -use super::fetch_types::Fetched; -pub(in crate::data::executor) use super::fetch_types::{ - DocFetchParams, DocScanMode, FetchedRows, RowOrigin, -}; +use super::fetch_types::FetchedRows; +use super::{DocFetchParams, DocScanMode, parse_fetched_key}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::core_loop::filter_match::matches_with_resolved_schema; use crate::data::executor::scan_normalize::{sparse_body_to_msgpack, sparse_row_to_doc}; @@ -123,7 +119,6 @@ impl CoreLoop { .collect(); Ok(FetchedRows { rows, - origin: RowOrigin::Sparse, effective_schema: None, deadline_expired: deadline.tripped(), }) @@ -178,7 +173,6 @@ impl CoreLoop { } Ok(FetchedRows { rows, - origin: RowOrigin::Sparse, effective_schema: None, deadline_expired: deadline.tripped(), }) @@ -202,7 +196,7 @@ impl CoreLoop { /// Newest live version per document (current-time read). Bitemporal /// collections read current state from the versioned store; plain - /// collections from the live table with a `scan_collection` fallback. + /// collections from the live table. fn fetch_current( &mut self, task: &ExecutionTask, @@ -231,7 +225,7 @@ impl CoreLoop { SparseBodyFormat::VectorSidecar ); - // `scan_documents_filtered`/`versioned_scan_as_of`/`scan_collection` + // `scan_documents_filtered`/`versioned_scan_as_of` // take an infallible `Fn(&str, &[u8]) -> bool` predicate (a // storage-engine primitive out of scope for this fix), so a // division/modulo-by-zero is captured via this `Cell` side-channel @@ -262,9 +256,8 @@ impl CoreLoop { } }; // `scan_documents_filtered` hands the predicate a typed `StorageKey`; - // `matches` (and `versioned_scan_as_of`'s predicate, and the - // `scan_collection` fallback filter) still take the row's storage - // key as text, so this renders it once per candidate row. + // `matches` (and `versioned_scan_as_of`'s predicate) still take the + // row's storage key as text, so this renders it once per candidate row. let matches_by_key = |key: &StorageKey, value: &[u8]| matches(&key.to_string(), value); // `versioned_scan_as_of` hands back a rendered storage key, parsed @@ -272,98 +265,11 @@ impl CoreLoop { // shape that fails to parse names a fetch-pipeline bug, not a row to // skip. let parse_row_key = |id: String, body: Vec| -> crate::Result<(StorageKey, Vec)> { - let key = StorageKey::parse(&id).ok_or_else(|| crate::Error::Storage { - engine: "sparse".into(), - detail: format!( - "collection '{collection}' fetched a row whose id is not a valid storage key: '{id}'" - ), - })?; - Ok((key, body)) + Ok((parse_fetched_key(collection, &id)?, body)) }; - let rows: Fetched = if filter_predicates.is_empty() { - if bitemporal { - Fetched::Sparse( - self.sparse - .versioned_scan_as_of( - crate::engine::sparse::btree_versioned::VersionedScanParams { - database_id, - tenant: tid, - coll: collection, - sys_cutoff_ms: None, - valid_at_ms: None, - limit: fetch_limit, - }, - &|_, _| true, - &stop, - )? - .into_iter() - .map(|(id, body)| parse_row_key(id, body)) - .collect::>>()?, - ) - } else { - // Routed through the filtered scan with an always-true - // predicate rather than `scan_documents`: the unfiltered full - // scan is the longest-running read shape there is, and only - // this entry point takes the stop signal. Rows, order and the - // `limit` cutoff are identical. - let sparse_result = self.sparse.scan_documents_filtered( - database_id, - tid, - collection, - fetch_limit, - &|_: &StorageKey, _: &[u8]| true, - &stop, - ); - match sparse_result { - Ok(docs) if docs.is_empty() => { - let fallback = - self.scan_collection(database_id, tid, collection, fetch_limit)?; - if !fallback.is_empty() { - warn!( - core = self.core_id, - %collection, - count = fallback.len(), - "document scan fallback to scan_collection" - ); - } - Fetched::Foreign(fallback) - } - other => Fetched::Sparse(other?), - } - } - } else if strict_schema.is_some() { + let rows: Vec<(StorageKey, Vec)> = if filter_predicates.is_empty() { if bitemporal { - Fetched::Sparse( - self.sparse - .versioned_scan_as_of( - crate::engine::sparse::btree_versioned::VersionedScanParams { - database_id, - tenant: tid, - coll: collection, - sys_cutoff_ms: None, - valid_at_ms: None, - limit: fetch_limit, - }, - &matches, - &stop, - )? - .into_iter() - .map(|(id, body)| parse_row_key(id, body)) - .collect::>>()?, - ) - } else { - Fetched::Sparse(self.sparse.scan_documents_filtered( - database_id, - tid, - collection, - fetch_limit, - &matches_by_key, - &stop, - )?) - } - } else if bitemporal { - Fetched::Sparse( self.sparse .versioned_scan_as_of( crate::engine::sparse::btree_versioned::VersionedScanParams { @@ -374,31 +280,53 @@ impl CoreLoop { valid_at_ms: None, limit: fetch_limit, }, - &matches, + &|_, _| true, &stop, )? .into_iter() .map(|(id, body)| parse_row_key(id, body)) - .collect::>>()?, - ) + .collect::>>()? + } else { + // Routed through the filtered scan with an always-true + // predicate rather than `scan_documents`: the unfiltered full + // scan is the longest-running read shape there is, and only + // this entry point takes the stop signal. Rows, order and the + // `limit` cutoff are identical. + self.sparse.scan_documents_filtered( + database_id, + tid, + collection, + fetch_limit, + &|_: &StorageKey, _: &[u8]| true, + &stop, + )? + } + } else if bitemporal { + self.sparse + .versioned_scan_as_of( + crate::engine::sparse::btree_versioned::VersionedScanParams { + database_id, + tenant: tid, + coll: collection, + sys_cutoff_ms: None, + valid_at_ms: None, + limit: fetch_limit, + }, + &matches, + &stop, + )? + .into_iter() + .map(|(id, body)| parse_row_key(id, body)) + .collect::>>()? } else { - let sparse_result = self.sparse.scan_documents_filtered( + self.sparse.scan_documents_filtered( database_id, tid, collection, fetch_limit, &matches_by_key, &stop, - ); - match sparse_result { - Ok(docs) if docs.is_empty() => Fetched::Foreign( - self.scan_collection(database_id, tid, collection, fetch_limit)? - .into_iter() - .filter(|(id, data)| matches(id, data)) - .collect(), - ), - other => Fetched::Sparse(other?), - } + )? }; if let Some(e) = predicate_err.take() { @@ -413,31 +341,20 @@ impl CoreLoop { // it sees for every other collection. Without it the tagged values pass // through untouched and reach the client as `[4,"alice"]`. The key // stays typed from the scan above, so this needs no re-parse. - let (rows, origin): (Vec<(String, Vec)>, RowOrigin) = match rows { - Fetched::Sparse(rows) if is_vector_sidecar => ( - rows.into_iter() - .map(|(key, body)| { - sparse_row_to_doc(&key, &body, SparseBodyFormatRef::VectorSidecar) - }) - .collect(), - RowOrigin::Sparse, - ), - Fetched::Sparse(rows) => ( - rows.into_iter() - .map(|(key, body)| (key.to_string(), body)) - .collect(), - RowOrigin::Sparse, - ), - // An empty fallback proves nothing about which engine owns the - // collection, and the rows a transaction overlay merges in later - // are sparse-shaped, so an empty fetch reports the sparse origin. - Fetched::Foreign(rows) if rows.is_empty() => (rows, RowOrigin::Sparse), - Fetched::Foreign(rows) => (rows, RowOrigin::Foreign), + let rows: Vec<(String, Vec)> = if is_vector_sidecar { + rows.into_iter() + .map(|(key, body)| { + sparse_row_to_doc(&key, &body, SparseBodyFormatRef::VectorSidecar) + }) + .collect() + } else { + rows.into_iter() + .map(|(key, body)| (key.to_string(), body)) + .collect() }; Ok(FetchedRows { rows, - origin, effective_schema: strict_schema.cloned(), deadline_expired: deadline.tripped(), }) diff --git a/nodedb/src/data/executor/handlers/document/read/fetch_types.rs b/nodedb/src/data/executor/handlers/document/read/fetch_types.rs index d7e297e7a..0f9d4946c 100644 --- a/nodedb/src/data/executor/handlers/document/read/fetch_types.rs +++ b/nodedb/src/data/executor/handlers/document/read/fetch_types.rs @@ -32,12 +32,6 @@ impl DocScanMode { } } -/// Rows a current-mode fetch produced, typed by [`RowOrigin`]. -pub(super) enum Fetched { - Sparse(Vec<(StorageKey, Vec)>), - Foreign(Vec<(String, Vec)>), -} - /// Borrowed inputs for [`crate::data::executor::core_loop::CoreLoop::document_scan_fetch`]. pub(in crate::data::executor) struct DocFetchParams<'a> { pub collection: &'a str, @@ -54,24 +48,23 @@ pub(in crate::data::executor) struct DocFetchParams<'a> { pub full_fetch: bool, } -/// Which engine keyed the rows a fetch produced. A fetch never mixes the -/// two: the foreign fallback runs only when the sparse store holds nothing -/// for the collection, and an empty fetch always reports `Sparse`. -#[derive(Clone, Copy)] -pub(in crate::data::executor) enum RowOrigin { - /// Sparse-store rows. Every id is a rendered storage key. - Sparse, - /// `scan_collection` fallback rows. A KV row is keyed by its user key and - /// a columnar row by its `id` column, so the id is the engine's own - /// identity text, never a storage key, and the body is already a - /// standard msgpack map. - Foreign, +/// Parse a fetched row's id back into the storage key it was minted as. +/// +/// Every row a document fetch produces is keyed by a rendered surrogate, so +/// a shape that fails to parse names a fetch-pipeline bug, never a row to +/// skip. +pub(super) fn parse_fetched_key(collection: &str, id: &str) -> crate::Result { + StorageKey::parse(id).ok_or_else(|| crate::Error::Storage { + engine: "sparse".into(), + detail: format!( + "collection '{collection}' fetched a row whose id is not a valid storage key: '{id}'" + ), + }) } /// Raw rows plus the schema the downstream should decode them with. pub(in crate::data::executor) struct FetchedRows { pub rows: Vec<(String, Vec)>, - pub origin: RowOrigin, pub effective_schema: Option, /// The statement's deadline passed while the storage scan was running, so /// `rows` holds an arbitrary prefix of the answer. The caller MUST fail the diff --git a/nodedb/src/data/executor/handlers/document/read/mod.rs b/nodedb/src/data/executor/handlers/document/read/mod.rs index d628bb04b..f2fd5054c 100644 --- a/nodedb/src/data/executor/handlers/document/read/mod.rs +++ b/nodedb/src/data/executor/handlers/document/read/mod.rs @@ -6,7 +6,10 @@ mod audit_body; pub mod decode; pub mod emit; pub mod fetch; -pub mod fetch_types; +mod fetch_types; pub mod materialize_scan; pub mod projection; pub mod scan; + +use fetch_types::parse_fetched_key; +pub(in crate::data::executor) use fetch_types::{DocFetchParams, DocScanMode}; diff --git a/nodedb/src/data/executor/handlers/document/read/scan.rs b/nodedb/src/data/executor/handlers/document/read/scan.rs index 8cb4aae9a..e8a25c1e8 100644 --- a/nodedb/src/data/executor/handlers/document/read/scan.rs +++ b/nodedb/src/data/executor/handlers/document/read/scan.rs @@ -4,8 +4,9 @@ use tracing::{debug, warn}; -use super::fetch::{DocFetchParams, DocScanMode, RowOrigin}; +use super::fetch_types::parse_fetched_key; use super::projection::{apply_projection, apply_projection_msgpack}; +use super::{DocFetchParams, DocScanMode}; use crate::bridge::envelope::{ErrorCode, Response}; use crate::bridge::scan_filter::ScanFilter; use crate::data::executor::core_loop::CoreLoop; @@ -17,31 +18,17 @@ use crate::data::executor::task::ExecutionTask; /// Shape one fetched row into `(identity, standard msgpack body)`. /// -/// A sparse row's id reaches this handler as rendered storage-key text -/// several calls removed from the scan that produced it (`document_scan_fetch` -/// unifies every source — current, `AS OF`, bitemporal — to -/// `(String, Vec)`), so a shape that fails to parse names a -/// fetch-pipeline bug rather than a row to skip. A foreign row already -/// carries its engine's identity and a msgpack body, so it passes through. +/// The id reaches this handler as rendered storage-key text several calls +/// removed from the scan that produced it: `document_scan_fetch` unifies every +/// source — current, `AS OF`, bitemporal — to `(String, Vec)`. fn fetched_row_to_doc( - origin: RowOrigin, collection: &str, - id: String, - body: Vec, + id: &str, + body: &[u8], body_format: SparseBodyFormatRef<'_>, ) -> crate::Result<(String, Vec)> { - match origin { - RowOrigin::Sparse => { - let key = nodedb_types::StorageKey::parse(&id).ok_or_else(|| crate::Error::Storage { - engine: "sparse".into(), - detail: format!( - "collection '{collection}' scan fetched a row whose id is not a valid storage key: '{id}'" - ), - })?; - Ok(sparse_row_to_doc(&key, &body, body_format)) - } - RowOrigin::Foreign => Ok((id, body)), - } + let key = parse_fetched_key(collection, id)?; + Ok(sparse_row_to_doc(&key, body, body_format)) } /// Parameters for [`CoreLoop::execute_document_scan`]. @@ -176,7 +163,6 @@ impl CoreLoop { return self.response_error(task, ErrorCode::DeadlineExceeded); } let mut filtered = fetched.rows; - let origin = fetched.origin; let effective_schema = fetched.effective_schema; // The encoding the rows arrive in from the fetch stage. It is // NOT the collection's stored encoding: the fetch stage has @@ -268,9 +254,7 @@ impl CoreLoop { let filtered = if !sort_keys.is_empty() || !projection.is_empty() { match filtered .into_iter() - .map(|(id, bytes)| { - fetched_row_to_doc(origin, collection, id, bytes, body_format) - }) + .map(|(id, bytes)| fetched_row_to_doc(collection, &id, &bytes, body_format)) .collect::>>() { Ok(rows) => rows, @@ -323,7 +307,7 @@ impl CoreLoop { .into_iter() .map(|(doc_id, val)| { let (doc_id, mp) = - fetched_row_to_doc(origin, collection, doc_id, val, body_format)?; + fetched_row_to_doc(collection, &doc_id, &val, body_format)?; let projected = apply_projection_msgpack(&mp, &computed_cols, projection)?; Ok((doc_id, projected)) @@ -354,7 +338,7 @@ impl CoreLoop { .into_iter() .map(|(id, val)| { let (doc_id, mp) = - fetched_row_to_doc(origin, collection, id, val, body_format)?; + fetched_row_to_doc(collection, &id, &val, body_format)?; crate::data::executor::doc_format::decode_document(&mp) .map(|doc| (doc_id, doc)) }) @@ -408,13 +392,8 @@ impl CoreLoop { let projected_rows: Vec<_> = match sorted .into_iter() .map(|(doc_id, value)| { - let (doc_id, mp) = fetched_row_to_doc( - origin, - collection, - doc_id, - value, - body_format, - )?; + let (doc_id, mp) = + fetched_row_to_doc(collection, &doc_id, &value, body_format)?; let projected = apply_projection_msgpack(&mp, &computed_cols, projection)?; Ok((doc_id, projected)) diff --git a/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/basic_scans.rs b/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/basic_scans.rs index 58477129e..d5a07d7f0 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/basic_scans.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/basic_scans.rs @@ -5,10 +5,7 @@ use super::super::helpers::{make_ctx, send_ok}; use nodedb::data::executor::handlers::join; use nodedb::data::executor::response_codec; -use nodedb_physical::physical_plan::{ - DocumentOp, EnforcementOptions, KvOp, PhysicalPlan, StorageMode, -}; -use nodedb_types::columnar::{ColumnDef, ColumnType, StrictSchema}; +use nodedb_physical::physical_plan::{DocumentOp, KvOp, PhysicalPlan}; pub(super) fn build_msgpack_map(fields: &[(&str, &str)]) -> Vec { let mut map = serde_json::Map::new(); @@ -82,96 +79,6 @@ fn kv_put_scan_roundtrip() { } } -#[test] -fn document_scan_preserves_kv_rows_when_collection_has_strict_config() { - let mut ctx = make_ctx(); - - send_ok( - &mut ctx.core, - &mut ctx.tx, - &mut ctx.rx, - PhysicalPlan::Document(DocumentOp::Register { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "prefs", - ), - indexes: Vec::new(), - crdt_enabled: false, - storage_mode: StorageMode::Strict { - schema: StrictSchema { - columns: vec![ - ColumnDef::required("key", ColumnType::String).with_primary_key(), - ColumnDef::required("theme", ColumnType::String), - ColumnDef::nullable("lang", ColumnType::String), - ], - version: 1, - dropped_columns: Vec::new(), - bitemporal: false, - }, - }, - enforcement: Box::new(EnforcementOptions::default()), - bitemporal: false, - conflict_policy: None, - timeseries: None, - vector_primary: None, - }), - ); - - let value = build_msgpack_map(&[("theme", "dark"), ("lang", "en")]); - send_ok( - &mut ctx.core, - &mut ctx.tx, - &mut ctx.rx, - PhysicalPlan::Kv(KvOp::Put { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "prefs", - ), - key: b"d1".to_vec(), - value, - ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, - returning: None, - rls_filters: Vec::new(), - }), - ); - - let payload = send_ok( - &mut ctx.core, - &mut ctx.tx, - &mut ctx.rx, - PhysicalPlan::Document(DocumentOp::Scan { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "prefs", - ), - filters: Vec::new(), - limit: 100, - offset: 0, - sort_keys: Vec::new(), - distinct: false, - projection: Vec::new(), - computed_columns: Vec::new(), - window_functions: Vec::new(), - system_time: nodedb_types::SystemTimeScope::Current, - valid_at_ms: None, - prefilter: None, - }), - ); - - let json = response_codec::decode_payload_to_json(&payload); - let parsed: Vec = - serde_json::from_str(&json).unwrap_or_else(|e| panic!("invalid JSON: {e}\nraw: {json}")); - assert_eq!(parsed.len(), 1, "expected 1 row, got {json}"); - - let data = parsed[0]["data"] - .as_object() - .unwrap_or_else(|| panic!("expected object data, got {}", parsed[0]["data"])); - assert_eq!(data.get("key").and_then(|v| v.as_str()), Some("d1")); - assert_eq!(data.get("theme").and_then(|v| v.as_str()), Some("dark")); - assert_eq!(data.get("lang").and_then(|v| v.as_str()), Some("en")); -} - #[test] fn schemaless_put_scan_roundtrip() { let mut ctx = make_ctx(); diff --git a/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/multi_core_joins.rs b/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/multi_core_joins.rs index 4c99430cf..0452152fc 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/multi_core_joins.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/multi_core_joins.rs @@ -63,27 +63,23 @@ fn multi_core_broadcast_inner_join() { ); } - // Phase 1: Scan prefs from core 1 via DocumentScan (same as broadcast_raw). + // Phase 1: scan prefs from core 1 with the KV scan the coordinator's + // build-side gather emits for a KV collection. let phase1_payload = send_ok( &mut core1.core, &mut core1.tx, &mut core1.rx, - PhysicalPlan::Document(DocumentOp::Scan { + PhysicalPlan::Kv(KvOp::Scan { collection: nodedb_types::QualifiedCollection::new( nodedb_types::DatabaseId::DEFAULT, "prefs", ), + cursor: Vec::new(), + count: 100, filters: Vec::new(), - limit: 100, - offset: 0, sort_keys: Vec::new(), - distinct: false, - projection: Vec::new(), - computed_columns: Vec::new(), - window_functions: Vec::new(), - system_time: nodedb_types::SystemTimeScope::Current, - valid_at_ms: None, - prefilter: None, + match_pattern: None, + surrogate_ceiling: None, }), ); @@ -227,22 +223,17 @@ fn multi_core_broadcast_left_join() { &mut core1.core, &mut core1.tx, &mut core1.rx, - PhysicalPlan::Document(DocumentOp::Scan { + PhysicalPlan::Kv(KvOp::Scan { collection: nodedb_types::QualifiedCollection::new( nodedb_types::DatabaseId::DEFAULT, "prefs", ), + cursor: Vec::new(), + count: 100, filters: Vec::new(), - limit: 100, - offset: 0, sort_keys: Vec::new(), - distinct: false, - projection: Vec::new(), - computed_columns: Vec::new(), - window_functions: Vec::new(), - system_time: nodedb_types::SystemTimeScope::Current, - valid_at_ms: None, - prefilter: None, + match_pattern: None, + surrogate_ceiling: None, }), ); @@ -399,22 +390,17 @@ fn multi_core_broadcast_merge_simulation() { } // Phase 1: scan prefs from BOTH cores and concatenate raw payloads. - let scan_plan = PhysicalPlan::Document(DocumentOp::Scan { + let scan_plan = PhysicalPlan::Kv(KvOp::Scan { collection: nodedb_types::QualifiedCollection::new( nodedb_types::DatabaseId::DEFAULT, "prefs", ), + cursor: Vec::new(), + count: 100, filters: Vec::new(), - limit: 100, - offset: 0, sort_keys: Vec::new(), - distinct: false, - projection: Vec::new(), - computed_columns: Vec::new(), - window_functions: Vec::new(), - system_time: nodedb_types::SystemTimeScope::Current, - valid_at_ms: None, - prefilter: None, + match_pattern: None, + surrogate_ceiling: None, }); let payload0 = send_ok( &mut core0.core, diff --git a/nodedb/tests/wire/cases/mod.rs b/nodedb/tests/wire/cases/mod.rs index 30173a7ee..e47f21052 100644 --- a/nodedb/tests/wire/cases/mod.rs +++ b/nodedb/tests/wire/cases/mod.rs @@ -127,6 +127,7 @@ mod pgwire_txn_overlay_teardown_reclaim; mod point_get_after_miss_then_insert; mod procedure_e2e; mod query_function_authorization; +mod query_function_engine_gate; mod quota_bitemporal_composition; mod quota_sieve_routing_composition; mod redaction_policy_database_scope; diff --git a/nodedb/tests/wire/cases/query_function_engine_gate.rs b/nodedb/tests/wire/cases/query_function_engine_gate.rs new file mode 100644 index 000000000..a55ee9616 --- /dev/null +++ b/nodedb/tests/wire/cases/query_function_engine_gate.rs @@ -0,0 +1,113 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Engine gate for the SQL functions and DDL that hand-build a document scan +//! over a caller-named collection. +//! +//! `VERIFY_HASH_CHAIN`, `TEMPORAL_LOOKUP`, `CONVERT_CURRENCY`, +//! `VERIFY_BALANCE`, `BALANCE_AS_OF` and `CREATE GRAPH INDEX` read the sparse +//! store, which holds document rows only. On a KV or columnar-family +//! collection that scan answers with no rows, so each of them refuses such a +//! collection with SQLSTATE `0A000` instead of reporting over an empty set. + +use crate::harness::TestServer; + +async fn seed_kv(server: &TestServer, collection: &str) { + server + .exec(&format!( + "CREATE COLLECTION {collection} (id STRING PRIMARY KEY, parent STRING, amount INT) \ + WITH (engine='kv')" + )) + .await + .unwrap_or_else(|e| panic!("create {collection}: {e}")); + server + .exec(&format!( + "INSERT INTO {collection} {{ id: 'a', parent: 'root', amount: 5 }}" + )) + .await + .unwrap_or_else(|e| panic!("seed {collection}: {e}")); +} + +async fn seed_columnar(server: &TestServer, collection: &str) { + server + .exec(&format!( + "CREATE COLLECTION {collection} (id STRING, pair STRING, ts STRING) \ + WITH (engine='columnar')" + )) + .await + .unwrap_or_else(|e| panic!("create {collection}: {e}")); + server + .exec(&format!( + "INSERT INTO {collection} (id, pair, ts) VALUES ('r1', 'k1', '2024-01-01')" + )) + .await + .unwrap_or_else(|e| panic!("seed {collection}: {e}")); +} + +fn assert_engine_refusal(what: &str, result: Result, String>) { + match result { + Err(message) => assert!( + message.contains("SQLSTATE 0A000") && message.contains("reads document collections"), + "{what}: expected the engine refusal, got: {message}" + ), + Ok(rows) => panic!("{what}: answered over a non-document collection: {rows:?}"), + } +} + +/// `VERIFY_HASH_CHAIN` on a KV collection is refused, never "valid over zero +/// entries". +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn verify_hash_chain_refuses_a_kv_collection() { + let server = TestServer::start().await; + seed_kv(&server, "qfe_hash_kv").await; + + let result = server + .query_text("SELECT VERIFY_HASH_CHAIN('qfe_hash_kv')") + .await; + + assert_engine_refusal("VERIFY_HASH_CHAIN", result); +} + +/// `TEMPORAL_LOOKUP` on a columnar collection is refused, never "no row". +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn temporal_lookup_refuses_a_columnar_collection() { + let server = TestServer::start().await; + seed_columnar(&server, "qfe_tl_col").await; + + let result = server + .query_text("SELECT TEMPORAL_LOOKUP('qfe_tl_col', 'k1', '2024-12-31', 'pair', 'ts')") + .await; + + assert_engine_refusal("TEMPORAL_LOOKUP", result); +} + +/// `CREATE GRAPH INDEX` on a KV collection is refused, never built empty. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn create_graph_index_refuses_a_kv_collection() { + let server = TestServer::start().await; + seed_kv(&server, "qfe_graph_kv").await; + + let result = server + .query_text("CREATE GRAPH INDEX qfe_graph_kv_idx ON qfe_graph_kv (parent -> id)") + .await; + + assert_engine_refusal("CREATE GRAPH INDEX", result); +} + +/// A document collection passes the gate: the same call answers. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn verify_hash_chain_answers_on_a_document_collection() { + let server = TestServer::start().await; + server + .exec("CREATE COLLECTION qfe_hash_doc") + .await + .expect("create document collection"); + server + .exec("INSERT INTO qfe_hash_doc { id: 'a', amount: 5 }") + .await + .expect("seed row"); + + server + .query_text("SELECT VERIFY_HASH_CHAIN('qfe_hash_doc')") + .await + .expect("document collection passes the engine gate"); +} From 342c2d750efdecaa03b45a3d6cde4455021b181c Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 12 Sep 2026 06:33:11 +0800 Subject: [PATCH 08/17] refactor(sparse): thread StorageKey through secondary index ops Secondary-index reads and writes on the sparse btree engine took document ids as raw &str/String, forcing a StorageKey -> String -> StorageKey round-trip at every index_put/delete/lookup call site. IndexEntryTxn, index_put, delete_index_entries_for_field, and range_scan now carry a StorageKey directly, and with_tenant_key4 renders its last segment via Display instead of requiring a pre-formatted &str. invalid_storage_key_err takes the source table name (DOCUMENTS, INDEXES, INDEXES_VERSIONED) so a malformed-key error reports which table produced it. Propagate the StorageKey-typed document id through the document store's put/delete/batch paths and every executor handler that builds or consumes a SecondaryIndexInputs, bulk delete/update, undo, or truncate call. --- .../data/executor/core_loop/maintenance.rs | 3 +- .../materialized_sum/divergence.rs | 12 +- .../data/executor/enforcement/statement.rs | 13 +-- .../executor/handlers/bulk_dml/admission.rs | 21 ++-- .../data/executor/handlers/bulk_dml/delete.rs | 69 +++++------ .../handlers/bulk_dml/delete_cascade.rs | 107 ++++++++---------- .../data/executor/handlers/bulk_dml/scan.rs | 50 ++++---- .../handlers/bulk_dml/update_project.rs | 31 ++--- .../control/calvin_overlay_stage_bulk.rs | 11 +- .../executor/handlers/document/index_fetch.rs | 26 +++-- .../handlers/document/index_maintenance.rs | 12 +- .../handlers/document/resolve/bulk.rs | 11 +- nodedb/src/data/executor/handlers/facet.rs | 15 +-- .../executor/handlers/point/apply_delete.rs | 2 +- .../executor/handlers/point/apply_put/core.rs | 4 +- .../handlers/point/apply_put/unique.rs | 4 +- .../executor/handlers/point/update_reindex.rs | 2 +- .../transaction/stage_write/constraint.rs | 9 +- .../stage_write/stage_bulk_update.rs | 10 +- .../handlers/transaction/undo/document.rs | 14 ++- .../handlers/transaction/undo/rollback.rs | 2 +- nodedb/src/data/executor/handlers/truncate.rs | 86 ++++++-------- .../src/engine/document/store/engine/batch.rs | 38 +++++-- .../engine/document/store/engine/delete.rs | 2 +- .../src/engine/document/store/engine/put.rs | 2 +- nodedb/src/engine/sparse/btree/keys.rs | 6 +- nodedb/src/engine/sparse/btree/mod.rs | 4 +- nodedb/src/engine/sparse/btree/tables.rs | 14 +-- nodedb/src/engine/sparse/btree_index.rs | 55 ++++++--- nodedb/src/engine/sparse/btree_scan.rs | 8 +- 30 files changed, 303 insertions(+), 340 deletions(-) diff --git a/nodedb/src/data/executor/core_loop/maintenance.rs b/nodedb/src/data/executor/core_loop/maintenance.rs index e220927f0..f30487875 100644 --- a/nodedb/src/data/executor/core_loop/maintenance.rs +++ b/nodedb/src/data/executor/core_loop/maintenance.rs @@ -3,6 +3,7 @@ use std::sync::Arc; use super::CoreLoop; +use crate::engine::document::store::StorageKey; use crate::engine::sparse::doc_cache::DocCache; /// (added, removed) secondary-index (field, value) tuples. @@ -15,7 +16,7 @@ pub(in crate::data::executor) struct SecondaryIndexInputs<'a> { pub collection: &'a str, pub old_doc: Option<&'a serde_json::Value>, pub new_doc: &'a serde_json::Value, - pub doc_id: &'a str, + pub doc_id: &'a StorageKey, pub index_paths: &'a [crate::engine::document::store::IndexPath], } diff --git a/nodedb/src/data/executor/enforcement/materialized_sum/divergence.rs b/nodedb/src/data/executor/enforcement/materialized_sum/divergence.rs index d60890647..e8f8136e2 100644 --- a/nodedb/src/data/executor/enforcement/materialized_sum/divergence.rs +++ b/nodedb/src/data/executor/enforcement/materialized_sum/divergence.rs @@ -145,22 +145,16 @@ impl CoreLoop { pub(in crate::data::executor) fn sum_targets_diverged_for_ids( &self, check: &SumTargetCheck<'_>, - doc_ids: &[String], + doc_ids: &[nodedb_types::StorageKey], ) -> bool { if !self.ollp_is_group_leader || !self.declares_materialized_sums(check) { return false; } let mut rows: Vec = Vec::with_capacity(doc_ids.len()); - for doc_id in doc_ids { - // `doc_id` is a bare string from a raw-table scan; a shape that - // fails to parse as a storage key contributes no row, same as a - // `get` miss right below. - let Some(key) = nodedb_types::StorageKey::parse(doc_id) else { - continue; - }; + for key in doc_ids { let Ok(Some(bytes)) = self.sparse - .get(check.database_id, check.tid, check.collection, &key) + .get(check.database_id, check.tid, check.collection, key) else { continue; }; diff --git a/nodedb/src/data/executor/enforcement/statement.rs b/nodedb/src/data/executor/enforcement/statement.rs index 1b6736151..1b1858f20 100644 --- a/nodedb/src/data/executor/enforcement/statement.rs +++ b/nodedb/src/data/executor/enforcement/statement.rs @@ -92,21 +92,14 @@ impl CoreLoop { database_id: u64, tid: u64, collection: &str, - document_ids: &[String], + document_ids: &[nodedb_types::StorageKey], ) -> crate::Result> { let Some(def) = self.balanced_def(database_id, tid, collection) else { return Ok(Vec::new()); }; let mut entries = Vec::new(); - for document_id in document_ids { - // `document_id` arrives as a bare string several calls removed - // from the scan that produced it. A shape that fails to parse - // as a storage key is treated the same as a row that is not - // there: this check contributes nothing for a row it cannot read. - let Some(key) = nodedb_types::StorageKey::parse(document_id) else { - continue; - }; - let Some(stored) = self.sparse.get(database_id, tid, collection, &key)? else { + for key in document_ids { + let Some(stored) = self.sparse.get(database_id, tid, collection, key)? else { continue; }; // A stored row of a collection that declares constraints over its diff --git a/nodedb/src/data/executor/handlers/bulk_dml/admission.rs b/nodedb/src/data/executor/handlers/bulk_dml/admission.rs index 2cde59983..3980f0cb2 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/admission.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/admission.rs @@ -23,6 +23,7 @@ use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::doc_format; use crate::data::executor::enforcement::materialized_sum::divergence::SumTargetCheck; use nodedb_physical::physical_plan::{OllpPredictedEdge, ResolvedSumTarget, UpdateValue}; +use nodedb_types::StorageKey; /// The predictions one bulk statement carries. pub(in crate::data::executor) struct BulkAdmission<'a> { @@ -52,10 +53,10 @@ impl CoreLoop { &self, database_id: u64, tid: u64, - matching_ids: Vec, + matching_ids: Vec, admission: &BulkAdmission<'_>, - ) -> Result, ErrorCode> { - let apply_ids: Vec = match admission.predicted_surrogates { + ) -> Result, ErrorCode> { + let apply_ids: Vec = match admission.predicted_surrogates { Some(predicted) => { // The set comparison is deterministic: both sides are sorted. if self.ollp_is_group_leader @@ -110,9 +111,8 @@ impl CoreLoop { /// Compute the sorted ACTUAL implicit-edge set for the matched docs. /// - /// For each matched `doc_id`, parse its surrogate (same `len()==8` hex - /// parse as [`ollp_actual_surrogates`]), fetch the stored doc bytes via the - /// SAME `sparse.get` path the delete loop uses, decode it, and — only when + /// For each matched storage key, fetch the stored doc bytes via the SAME + /// `sparse.get` path the delete loop uses, decode it, and — only when /// it carries BOTH `_from` and `_to` as strings — record an /// [`OllpPredictedEdge`] with the raw `_type` as `label`. A matched doc /// without both endpoints is not an edge and is skipped; if it gained an @@ -129,16 +129,13 @@ impl CoreLoop { database_id: u64, tid: u64, collection: &str, - matching_ids: &[String], + matching_ids: &[StorageKey], ) -> Vec { // `decode_document` returns `serde_json::Value`, whose `get`/`as_str` // are inherent methods — no extra trait import needed. let mut edges: Vec = Vec::new(); - for doc_id in matching_ids { - let Some(key) = crate::engine::document::store::StorageKey::parse(doc_id) else { - continue; - }; - let Ok(Some(bytes)) = self.sparse.get(database_id, tid, collection, &key) else { + for key in matching_ids { + let Ok(Some(bytes)) = self.sparse.get(database_id, tid, collection, key) else { continue; }; let Ok(doc) = doc_format::decode_document(&bytes) else { diff --git a/nodedb/src/data/executor/handlers/bulk_dml/delete.rs b/nodedb/src/data/executor/handlers/bulk_dml/delete.rs index 72aa20df3..890fff20a 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/delete.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/delete.rs @@ -175,19 +175,13 @@ impl CoreLoop { rls_write_check.decision(), nodedb_types::WriteGateDecision::AdmitAll ) { - for doc_id in &apply_ids { - // `doc_id` is a bare string from a raw-table scan; a shape - // that fails to parse as a storage key is treated the same - // as the row-already-gone case right below it. - let Some(key) = crate::engine::document::store::StorageKey::parse(doc_id) else { - continue; - }; - let stored = match self.sparse.get(database_id, tid, collection, &key) { + for key in &apply_ids { + let stored = match self.sparse.get(database_id, tid, collection, key) { Ok(Some(bytes)) => bytes, Ok(None) => continue, Err(e) => return self.response_error(task, e), }; - let identity = crate::engine::document::store::identity_of(doc_id); + let identity = key.to_identity(); if let Err(e) = rls_write_gate::admit_stored_row( rls_write_check, &stored, @@ -232,13 +226,8 @@ impl CoreLoop { } else { Vec::new() }; - for doc_id in &apply_ids { - // `doc_id` is a bare string from a raw-table scan several calls - // removed from `SparseEngine`'s typed scan methods. A shape that - // fails to parse as a storage key can hold no row in DOCUMENTS - // either way, so every call below that needs the typed key treats - // it exactly like the "row already gone" case it already handles. - let storage_key = crate::engine::document::store::StorageKey::parse(doc_id); + for storage_key in &apply_ids { + let doc_id = storage_key.to_string(); // Capture pre-deletion snapshot if RETURNING was requested, or if // the collection is indexed (needed to recompute the removed @@ -251,18 +240,19 @@ impl CoreLoop { let pre_delete_doc: Option = if returning.is_some() || !index_paths.is_empty() { - match storage_key.and_then(|key| { - self.sparse - .get(task.request.database_id.as_u64(), tid, collection, &key) - .ok() - .flatten() - }) { + match self + .sparse + .get( + task.request.database_id.as_u64(), + tid, + collection, + storage_key, + ) + .ok() + .flatten() + { Some(bytes) => { - // `doc_id` is the storage key from the scan. `RETURNING` - // reports the row's client-visible identity, not the - // storage key. A value that fails to parse as a minted - // key is a legacy or user key, taken verbatim. - let identity = crate::engine::document::store::identity_of(doc_id); + let identity = storage_key.to_identity(); match returning_doc::from_stored(&bytes, &identity, strict_schema.as_ref()) { Ok(doc) => Some(doc), @@ -285,18 +275,17 @@ impl CoreLoop { Ok(txn) => txn, Err(e) => return self.response_error(task, e), }; - let deleted_bytes = storage_key.and_then(|key| { - self.sparse - .delete_in_txn( - &row_txn, - task.request.database_id.as_u64(), - tid, - collection, - &key, - ) - .ok() - .flatten() - }); + let deleted_bytes = self + .sparse + .delete_in_txn( + &row_txn, + task.request.database_id.as_u64(), + tid, + collection, + storage_key, + ) + .ok() + .flatten(); // Period lock, the pre-deletion image — a delete has no other. // Checked before `write_hook::run` and before commit: dropping // `row_txn` un-committed on a refusal reverses the removal. @@ -362,7 +351,7 @@ impl CoreLoop { tid, collection, doc_id: doc_id.as_str(), - storage_key, + storage_key: *storage_key, deleted_bytes: bytes, has_vectors, index_paths: &index_paths, diff --git a/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs b/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs index 1e12010d2..da7551ae1 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs @@ -13,7 +13,7 @@ use tracing::warn; use crate::bridge::envelope::WriteSetEntry; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::{IndexPath, RowIdentity, StorageKey}; +use crate::engine::document::store::{IndexPath, StorageKey}; /// Borrowed + owned inputs for [`CoreLoop::bulk_delete_row_cascade`], grouped /// so the call stays within the argument-count budget. @@ -23,7 +23,7 @@ pub(in crate::data::executor) struct BulkDeleteRowCascade<'a> { pub tid: u64, pub collection: &'a str, pub doc_id: &'a str, - pub storage_key: Option, + pub storage_key: StorageKey, /// The row's pre-deletion bytes, as `sparse.delete` returned them — /// never re-read. pub deleted_bytes: &'a [u8], @@ -60,35 +60,26 @@ impl CoreLoop { returning, } = cascade; - // Cascade: inverted index. doc_id is the hex-encoded surrogate - // (the redb storage key). `storage_key` already holds the parsed - // form; reuse it for the write version + write-set entry below. - let row_surrogate = storage_key.map(|key| key.surrogate()); - match row_surrogate { - Some(surrogate) => { - if let Err(e) = self.inverted.remove_document( - task.request.database_id.as_u64(), - crate::types::TenantId::new(tid), - collection, - surrogate, - ) { - // Recorded here, at the detection site: the row's own - // transaction has already committed, so this cleanup - // failure cannot roll it back. - crate::diag::orphaned_index_entry_after_delete(&e, collection, "inverted"); - warn!(core = self.core_id, %collection, %doc_id, error = %e, "bulk delete: inverted index removal failed"); - } - } - None => { - warn!(core = self.core_id, %collection, %doc_id, "bulk delete: doc_id is not a valid surrogate; FTS entry may be orphaned"); - } + // Cascade: inverted index. + let row_surrogate = storage_key.surrogate(); + if let Err(e) = self.inverted.remove_document( + task.request.database_id.as_u64(), + crate::types::TenantId::new(tid), + collection, + row_surrogate, + ) { + // Recorded here, at the detection site: the row's own + // transaction has already committed, so this cleanup + // failure cannot roll it back. + crate::diag::orphaned_index_entry_after_delete(&e, collection, "inverted"); + warn!(core = self.core_id, %collection, %doc_id, error = %e, "bulk delete: inverted index removal failed"); } // Cascade: secondary indexes. if let Err(e) = self.sparse.delete_indexes_for_document( task.request.database_id.as_u64(), tid, collection, - doc_id, + &storage_key, ) { crate::diag::orphaned_index_entry_after_delete(&e, collection, "secondary"); warn!(core = self.core_id, %collection, %doc_id, error = %e, "bulk delete: secondary index cascade failed"); @@ -117,39 +108,39 @@ impl CoreLoop { if has_vectors { self.remove_document_vector_indexes(database_id, tid, collection, doc_id); } - if let Some(key) = storage_key { - self.doc_cache - .invalidate(task.request.database_id.as_u64(), tid, collection, &key); - } + self.doc_cache.invalidate( + task.request.database_id.as_u64(), + tid, + collection, + &storage_key, + ); // Record the committed delete's write version against its // surrogate + collection. - if let Some(surrogate) = row_surrogate { - self.note_surrogate_write_lsn(task, tid, collection, surrogate.as_u32()); - // Record the removed secondary-index tuples into the - // per-index write-value substrate, recomputed from the - // pre-delete document (see `index_paths` comment above). - if let (Some(lsn), Some(doc)) = (task.wal_lsn(), pre_delete_doc.as_ref()) { - let tuples = self.index_tuples_for_doc(doc, index_paths); - self.note_index_write_values( - task.request.database_id, - crate::types::TenantId::new(tid), - collection, - &tuples, - lsn, - ); - } - // Carry the surrogate back for a post-apply `Delete` redo so - // the removed vector node does not resurrect on a WAL-only - // restart. Gated on `has_vectors` — a non-vector collection - // pays nothing. A delete carries no post-image body. - if has_vectors { - write_set.push(WriteSetEntry { - surrogate: surrogate.as_u32(), - is_delete: true, - value: Vec::new(), - collection: None, - }); - } + self.note_surrogate_write_lsn(task, tid, collection, row_surrogate.as_u32()); + // Record the removed secondary-index tuples into the + // per-index write-value substrate, recomputed from the + // pre-delete document (see `index_paths` comment above). + if let (Some(lsn), Some(doc)) = (task.wal_lsn(), pre_delete_doc.as_ref()) { + let tuples = self.index_tuples_for_doc(doc, index_paths); + self.note_index_write_values( + task.request.database_id, + crate::types::TenantId::new(tid), + collection, + &tuples, + lsn, + ); + } + // Carry the surrogate back for a post-apply `Delete` redo so + // the removed vector node does not resurrect on a WAL-only + // restart. Gated on `has_vectors` — a non-vector collection + // pays nothing. A delete carries no post-image body. + if has_vectors { + write_set.push(WriteSetEntry { + surrogate: row_surrogate.as_u32(), + is_delete: true, + value: Vec::new(), + collection: None, + }); } // Emit a delete event per affected row to the Event Plane, so // AFTER-DELETE triggers and CDC/change-stream consumers see @@ -167,9 +158,7 @@ impl CoreLoop { collection, deleted_bytes, ); - let event_identity = storage_key - .map(|key| key.to_identity()) - .unwrap_or_else(|| RowIdentity::from_user_key(doc_id)); + let event_identity = storage_key.to_identity(); self.emit_document_delete_event( task, collection, diff --git a/nodedb/src/data/executor/handlers/bulk_dml/scan.rs b/nodedb/src/data/executor/handlers/bulk_dml/scan.rs index a700d1792..93cf53b65 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/scan.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/scan.rs @@ -2,19 +2,20 @@ use crate::bridge::scan_filter::ScanFilter; use crate::data::executor::core_loop::CoreLoop; +use nodedb_types::StorageKey; use redb::{ReadableDatabase, ReadableTable}; impl CoreLoop { /// Scan documents in a collection matching the given filters. /// - /// Returns document IDs of all matching documents. + /// Returns the storage keys of all matching documents. pub(in crate::data::executor) fn scan_matching_documents( &self, database_id: u64, tid: u64, collection: &str, filters: &[ScanFilter], - ) -> crate::Result> { + ) -> crate::Result> { let prefix = crate::engine::sparse::btree::coll_prefix(database_id, tid, collection); let end = format!("{prefix}\u{ffff}"); @@ -46,7 +47,14 @@ impl CoreLoop { if let Some(doc_id) = key.strip_prefix(&prefix) && matches(doc_id, value_bytes)? { - ids.push(doc_id.to_string()); + let doc_id = StorageKey::parse(doc_id).ok_or_else(|| { + crate::engine::sparse::btree::invalid_storage_key_err( + "DOCUMENTS", + collection, + doc_id, + ) + })?; + ids.push(doc_id); } } } @@ -54,44 +62,30 @@ impl CoreLoop { } } -/// Compute the sorted list of surrogates from scanned document IDs. -/// -/// Document storage keys are 8-character hex-encoded u32 surrogates -/// (see `engine::document::store::key`). Ids that cannot be parsed are -/// silently skipped — they represent legacy non-surrogate documents that -/// do not participate in OLLP verification. -/// -/// The output is sorted ascending, matching the contract expected by the -/// OLLP verification comparison on both sides (Data Plane and Control -/// Plane pre-exec). /// Convert the carried OLLP predicted surrogate set into the sorted list of -/// document storage keys (8-char hex doc-ids) to apply the bulk mutation to. +/// storage keys to apply the bulk mutation to. /// /// This is the determinism anchor for multi-replica OLLP: every replica — /// leader and follower — mutates EXACTLY this set, derived from the leader's /// verified prediction carried in the plan, rather than from a per-replica /// local scan (which can differ when a follower's redb snapshot lags). Output /// is sorted ascending by surrogate so the apply order is identical on every -/// replica (`surrogate_to_doc_id` is monotonic in the surrogate, so sorting the -/// surrogates sorts the doc-ids). -pub(in crate::data::executor) fn ollp_predicted_doc_ids(predicted: &[u32]) -> Vec { +/// replica. +pub(in crate::data::executor) fn ollp_predicted_doc_ids(predicted: &[u32]) -> Vec { let mut surrogates: Vec = predicted.to_vec(); surrogates.sort_unstable(); surrogates .into_iter() - .map(|s| { - crate::engine::document::store::surrogate_to_doc_id(nodedb_types::Surrogate::new(s)) - }) + .map(|s| StorageKey::for_surrogate(nodedb_types::Surrogate::new(s))) .collect() } -pub(in crate::data::executor) fn ollp_actual_surrogates(doc_ids: &[String]) -> Vec { - let mut surrogates: Vec = doc_ids - .iter() - .filter_map(|id| { - crate::engine::document::store::doc_id_to_surrogate(id).map(|s| s.as_u32()) - }) - .collect(); +/// Compute the sorted list of surrogates from scanned storage keys. +/// +/// Feeds the OLLP verification comparison on both sides: Data Plane and +/// Control Plane pre-exec. +pub(in crate::data::executor) fn ollp_actual_surrogates(doc_ids: &[StorageKey]) -> Vec { + let mut surrogates: Vec = doc_ids.iter().map(|k| k.surrogate().as_u32()).collect(); surrogates.sort_unstable(); surrogates } @@ -102,7 +96,7 @@ pub(in crate::data::executor) fn ollp_actual_surrogates(doc_ids: &[String]) -> V /// the Calvin active-stage OLLP verifier so the `actual == predicted` guard /// lives in exactly one place. pub(in crate::data::executor) fn ollp_surrogates_match( - matching_ids: &[String], + matching_ids: &[StorageKey], predicted: &[u32], ) -> bool { let actual = ollp_actual_surrogates(matching_ids); diff --git a/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs b/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs index f9a3a65e4..8fd71ab39 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs @@ -39,7 +39,7 @@ pub(in crate::data::executor) struct ProjectUpdateRows<'a> { pub(in crate::data::executor) tid: u64, pub(in crate::data::executor) collection: &'a str, /// The settled apply set, in statement order. - pub(in crate::data::executor) doc_ids: &'a [String], + pub(in crate::data::executor) doc_ids: &'a [crate::engine::document::store::StorageKey], pub(in crate::data::executor) updates: &'a [(String, UpdateValue)], /// `Some` for a strict collection, whose bodies are Binary Tuples. pub(in crate::data::executor) strict_schema: Option<&'a StrictSchema>, @@ -72,15 +72,10 @@ impl CoreLoop { ); let mut projected = Vec::with_capacity(doc_ids.len()); - for doc_id in doc_ids { - // `doc_id` is a bare string from the settled apply set, several - // calls removed from any typed scan. A shape that fails to parse - // as a storage key is skipped the same as a row deleted between - // the match and this pass. - let Some(key) = crate::engine::document::store::StorageKey::parse(doc_id) else { - continue; - }; - let Some(current_bytes) = self.sparse.get(database_id, tid, collection, &key)? else { + for key in doc_ids { + let doc_id_owned = key.to_string(); + let doc_id = doc_id_owned.as_str(); + let Some(current_bytes) = self.sparse.get(database_id, tid, collection, key)? else { continue; }; @@ -97,19 +92,17 @@ impl CoreLoop { ) .ok_or_else(|| { crate::diag::strict_row_undecodable(collection, doc_id, "bulk_update_project"); - let identity = crate::engine::document::store::identity_of(doc_id); + let identity = key.to_identity(); crate::data::executor::strict_format::undecodable_strict_row( collection, identity.as_str(), ) })?, None => { - // `doc_id` is the storage key from the apply set. The - // decoded document's `id` must be the row's client-visible - // identity, not the storage key. A value that fails to - // parse as a minted key is a legacy or user key, taken - // verbatim. - let identity = crate::engine::document::store::identity_of(doc_id); + // `key` is the storage key from the apply set. The + // decoded document's `id` is the row's client-visible + // identity, not the storage key. + let identity = key.to_identity(); crate::data::executor::handlers::returning_doc::from_stored( ¤t_bytes, &identity, @@ -185,8 +178,8 @@ impl CoreLoop { }; projected.push(ProjectedUpdateRow { - doc_id: doc_id.clone(), - storage_key: key, + doc_id: doc_id_owned, + storage_key: *key, current_bytes, old_doc, doc, diff --git a/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs b/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs index 8e707996f..6a384466c 100644 --- a/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs +++ b/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs @@ -111,15 +111,12 @@ impl CoreLoop { rls_write_check.decision(), nodedb_types::WriteGateDecision::AdmitAll ) { - for (&surrogate, doc_id) in predicted_sorted.iter().zip(&doc_ids) { - let key = nodedb_types::StorageKey::for_surrogate(nodedb_types::Surrogate::new( - surrogate, - )); + for doc_id in &doc_ids { if let Some(body) = self.sparse - .get(task.request.database_id.as_u64(), tid, collection, &key)? + .get(task.request.database_id.as_u64(), tid, collection, doc_id)? { - let identity = crate::engine::document::store::identity_of(doc_id); + let identity = doc_id.to_identity(); self.stage_admit_write( rls_write_check, &body, @@ -134,7 +131,7 @@ impl CoreLoop { let overlay = self.txn_overlay_mut(txn_id); for (surrogate, doc_id) in predicted_sorted.into_iter().zip(doc_ids) { - overlay.insert_tombstone(coll_key.clone(), surrogate, &doc_id); + overlay.insert_tombstone(coll_key.clone(), surrogate, &doc_id.to_string()); } Ok(()) } diff --git a/nodedb/src/data/executor/handlers/document/index_fetch.rs b/nodedb/src/data/executor/handlers/document/index_fetch.rs index 072119355..e99f87356 100644 --- a/nodedb/src/data/executor/handlers/document/index_fetch.rs +++ b/nodedb/src/data/executor/handlers/document/index_fetch.rs @@ -72,7 +72,8 @@ impl CoreLoop { tid, ); match doc_engine.index_lookup(collection, path, value, bitemporal) { - Ok(mut doc_ids) => { + Ok(doc_ids) => { + let mut doc_ids: Vec = doc_ids.into_iter().map(|k| k.to_string()).collect(); if let Some(txn_id) = task.request.txn_id { let config_key = ( task.request.database_id, @@ -175,17 +176,18 @@ impl CoreLoop { let bitemporal = self.is_bitemporal(database_id, tid, collection); let doc_engine = crate::engine::document::store::DocumentEngine::new(&self.sparse, database_id, tid); - let mut doc_ids = match doc_engine.index_lookup(collection, path, value, bitemporal) { - Ok(ids) => ids, - Err(e) => { - return self.response_error( - task, - ErrorCode::Internal { - detail: format!("indexed fetch: {e}"), - }, - ); - } - }; + let mut doc_ids: Vec = + match doc_engine.index_lookup(collection, path, value, bitemporal) { + Ok(ids) => ids.into_iter().map(|k| k.to_string()).collect(), + Err(e) => { + return self.response_error( + task, + ErrorCode::Internal { + detail: format!("indexed fetch: {e}"), + }, + ); + } + }; // Strict collections store Binary Tuple bytes; the response codec // expects msgpack maps. Decode-then-encode here so cross-engine diff --git a/nodedb/src/data/executor/handlers/document/index_maintenance.rs b/nodedb/src/data/executor/handlers/document/index_maintenance.rs index ee08e0c7e..3eb549e04 100644 --- a/nodedb/src/data/executor/handlers/document/index_maintenance.rs +++ b/nodedb/src/data/executor/handlers/document/index_maintenance.rs @@ -113,7 +113,10 @@ impl CoreLoop { // Deduplicate-unique-as-we-go: track `(normalized_value → doc_id)` // so a dup within the existing set is flagged before we ever // touch the index table. - let mut seen: std::collections::HashMap = std::collections::HashMap::new(); + let mut seen: std::collections::HashMap< + String, + crate::engine::document::store::StorageKey, + > = std::collections::HashMap::new(); let txn = match self.sparse.begin_write() { Ok(t) => t, @@ -132,11 +135,6 @@ impl CoreLoop { let mut pending_keys: Vec = Vec::with_capacity(docs.len()); for (doc_id, bytes) in &docs { - // `IndexEntryTxn` and the dedup map below are INDEXES-table - // concerns, out of this unit's typed scope, so the storage key - // is rendered once here at the boundary. - let doc_id = doc_id.to_string(); - let doc_id = doc_id.as_str(); // A row skipped here is a row the finished index permanently omits, // and the index is then reported as built — every later lookup on // that row's value silently misses it. @@ -175,7 +173,7 @@ impl CoreLoop { ); } if unique { - seen.insert(stored.clone(), doc_id.to_string()); + seen.insert(stored.clone(), *doc_id); } pending_keys.push(crate::engine::sparse::btree_index::index_key_for( crate::engine::sparse::btree_index::IndexEntryTxn { diff --git a/nodedb/src/data/executor/handlers/document/resolve/bulk.rs b/nodedb/src/data/executor/handlers/document/resolve/bulk.rs index 57d901745..afe95b99c 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/bulk.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/bulk.rs @@ -169,14 +169,7 @@ impl CoreLoop { let mut mutations = Vec::with_capacity(doc_ids.len()); let mut rows: Vec<(RowIdentity, Vec)> = Vec::new(); - for doc_id in doc_ids { - // `doc_id` is a bare string from the raw-table scan. A shape - // that fails to parse as a storage key can hold no row in - // DOCUMENTS either way, so it is skipped the same as a row that - // vanished between the scan and this read. - let Some(key) = crate::engine::document::store::StorageKey::parse(&doc_id) else { - continue; - }; + for key in doc_ids { let Some(stored) = self.doc_resolve_read(&ctx, collection, &key)? else { continue; }; @@ -195,7 +188,7 @@ impl CoreLoop { let surrogate = key.surrogate(); mutations.push(delete_mutation( collection, - &doc_id, + &key.to_string(), surrogate, Some(stored.clone()), resolved_sum_targets, diff --git a/nodedb/src/data/executor/handlers/facet.rs b/nodedb/src/data/executor/handlers/facet.rs index 37eeb8c43..f19130888 100644 --- a/nodedb/src/data/executor/handlers/facet.rs +++ b/nodedb/src/data/executor/handlers/facet.rs @@ -67,7 +67,7 @@ impl CoreLoop { } }; - let matching_set: HashSet = matching_ids.iter().cloned().collect(); + let matching_set: HashSet = matching_ids.iter().map(|k| k.to_string()).collect(); // Step 2: For each facet field, count values. let mut facet_result = serde_json::Map::new(); @@ -133,7 +133,7 @@ impl CoreLoop { collection: &str, field: &str, matching_set: &HashSet, - matching_ids: &[String], + matching_ids: &[nodedb_types::StorageKey], ) -> Vec<(String, usize)> { // Fast path: index-backed counting with filtered doc set. if let Ok(groups) = self.sparse.scan_index_groups_filtered( @@ -161,15 +161,8 @@ impl CoreLoop { collection, ); let mut counts: HashMap = HashMap::new(); - for doc_id in matching_ids { - // `doc_id` is a bare string from a raw-table scan several calls - // removed from `SparseEngine`'s typed scan methods; a shape that - // fails to parse as a storage key contributes no row, the same as - // a `get` miss below. - let Some(key) = nodedb_types::StorageKey::parse(doc_id) else { - continue; - }; - if let Ok(Some(bytes)) = self.sparse.get(database_id, tid, collection, &key) { + for key in matching_ids { + if let Ok(Some(bytes)) = self.sparse.get(database_id, tid, collection, key) { let mp = crate::data::executor::scan_normalize::sparse_body_to_msgpack( &bytes, body_format.as_format_ref(), diff --git a/nodedb/src/data/executor/handlers/point/apply_delete.rs b/nodedb/src/data/executor/handlers/point/apply_delete.rs index a62728e95..31a59b640 100644 --- a/nodedb/src/data/executor/handlers/point/apply_delete.rs +++ b/nodedb/src/data/executor/handlers/point/apply_delete.rs @@ -325,7 +325,7 @@ impl CoreLoop { database_id, tid, collection, - row_key, + &storage_key, ) { warn!(core = self.core_id, %collection, %document_id, error = %e, "secondary index cascade failed; rejecting the delete"); return Err(e); diff --git a/nodedb/src/data/executor/handlers/point/apply_put/core.rs b/nodedb/src/data/executor/handlers/point/apply_put/core.rs index 9249e1e8c..b05d26086 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/core.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/core.rs @@ -238,7 +238,7 @@ impl CoreLoop { tid, collection, doc: &doc, - document_id, + document_id: &storage_key, paths: &paths, bitemporal, })?; @@ -288,7 +288,7 @@ impl CoreLoop { collection, old_doc: old_doc_for_index.as_ref(), new_doc: &doc, - doc_id: document_id, + doc_id: &storage_key, index_paths: &paths, }, )?; diff --git a/nodedb/src/data/executor/handlers/point/apply_put/unique.rs b/nodedb/src/data/executor/handlers/point/apply_put/unique.rs index 1723115d5..f7824843d 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/unique.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/unique.rs @@ -16,7 +16,7 @@ pub(in crate::data::executor) struct UniqueCheck<'a> { pub tid: u64, pub collection: &'a str, pub doc: &'a serde_json::Value, - pub document_id: &'a str, + pub document_id: &'a crate::engine::document::store::StorageKey, pub paths: &'a [crate::engine::document::store::IndexPath], /// Bitemporal collections keep secondary-index entries in the versioned /// index only; the uniqueness probe must read that index, not the empty @@ -59,7 +59,7 @@ pub(in crate::data::executor) fn check_unique_constraints(c: UniqueCheck<'_>) -> } else { raw }; - let existing = doc_engine + let existing: Vec = doc_engine .index_lookup(collection, &path.path, &needle, bitemporal) .unwrap_or_default(); if existing.iter().any(|id| id != document_id) { diff --git a/nodedb/src/data/executor/handlers/point/update_reindex.rs b/nodedb/src/data/executor/handlers/point/update_reindex.rs index 353d79440..677cb7999 100644 --- a/nodedb/src/data/executor/handlers/point/update_reindex.rs +++ b/nodedb/src/data/executor/handlers/point/update_reindex.rs @@ -208,7 +208,7 @@ impl CoreLoop { collection: p.collection, old_doc: Some(p.old_doc), new_doc: p.new_doc, - doc_id: p.doc_id, + doc_id: p.storage_key, index_paths: p.index_paths, }, )? diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/constraint.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/constraint.rs index c08307159..7e61bbd35 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/constraint.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/constraint.rs @@ -14,7 +14,7 @@ use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::point::apply_put::unique::{ UniqueCheck, check_unique_constraints, }; -use crate::engine::document::store::{CollectionConfig, extract_index_values}; +use crate::engine::document::store::{CollectionConfig, StorageKey, extract_index_values}; /// The overlay's verdict on a primary key within the current transaction. pub(super) enum OverlayPk { @@ -32,7 +32,7 @@ impl CoreLoop { &self, ctx: &StageCtx<'_>, row_key: &str, - storage_key: &crate::engine::document::store::StorageKey, + storage_key: &StorageKey, bitemporal: bool, overlay: OverlayPk, ) -> crate::Result { @@ -80,13 +80,16 @@ impl CoreLoop { ) -> crate::Result<()> { let collection = ctx.collection; // BASE: another durable row already owning one of the unique values. + // The row's own index entries are keyed by its storage key. The + // self-match exclusion compares that key, not the plan's document id. + let storage_key = StorageKey::for_surrogate(ctx.surrogate); check_unique_constraints(UniqueCheck { sparse: &self.sparse, database_id: ctx.database_id, tid: ctx.tid, collection, doc: incoming_doc, - document_id: &ctx.document_id, + document_id: &storage_key, paths: &config.index_paths, bitemporal: config.bitemporal, })?; diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs index 52bd68a80..2fd72344c 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs @@ -196,15 +196,9 @@ impl CoreLoop { .scan_matching_documents(database_id, tid, collection, filters) .map_err(|e| self.response_error(task, e))?; let mut rows: Vec<(String, Vec)> = Vec::with_capacity(matching_ids.len()); - for doc_id in matching_ids { - // `doc_id` is a bare string from a raw-table scan; a shape that - // fails to parse as a storage key contributes no row, same as a - // `get` miss right below. - let Some(key) = crate::engine::document::store::StorageKey::parse(&doc_id) else { - continue; - }; + for key in matching_ids { if let Ok(Some(bytes)) = self.sparse.get(database_id, tid, collection, &key) { - rows.push((doc_id, bytes)); + rows.push((key.to_string(), bytes)); } } Ok(rows) diff --git a/nodedb/src/data/executor/handlers/transaction/undo/document.rs b/nodedb/src/data/executor/handlers/transaction/undo/document.rs index de44923fe..ae5ee632e 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/document.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/document.rs @@ -94,7 +94,12 @@ impl CoreLoop { // restore the stale entries this put removed. Empty on the // bitemporal path (its index reversal happened in // `undo_bitemporal_write` above), so this is a no-op there. - self.undo_secondary_index(ctx, &secondary_index_added, &secondary_index_removed)?; + self.undo_secondary_index( + ctx, + &storage_key, + &secondary_index_added, + &secondary_index_removed, + )?; // Revert inverted index: remove the postings this rolled-back // put wrote. FATAL on failure — a rollback that leaves stale FTS // postings behind is the same silent-partial-success corruption @@ -174,7 +179,7 @@ impl CoreLoop { // Restore the plain secondary-index entries the forward delete // cascade removed. Empty on the bitemporal path (no plain // INDEXES entries there), so this is a no-op for it. - self.undo_secondary_index(ctx, &[], &secondary_index_tuples)?; + self.undo_secondary_index(ctx, &storage_key, &[], &secondary_index_tuples)?; // Re-index the restored document into the full-text inverted // index. The forward delete cascade removed its postings // unconditionally, so a rollback that restored the row but not @@ -264,6 +269,7 @@ impl CoreLoop { fn undo_secondary_index( &self, ctx: UndoDocumentContext<'_>, + storage_key: &crate::engine::document::store::StorageKey, to_remove: &[(String, String)], to_restore: &[(String, String)], ) -> Result<(), (usize, String)> { @@ -290,12 +296,12 @@ impl CoreLoop { }; for (field, value) in to_remove { self.sparse - .index_remove(database_id, tid, collection, field, value, document_id) + .index_remove(database_id, tid, collection, field, value, storage_key) .map_err(|e| map_err("remove", e.to_string()))?; } for (field, value) in to_restore { self.sparse - .index_put(database_id, tid, collection, field, value, document_id) + .index_put(database_id, tid, collection, field, value, storage_key) .map_err(|e| map_err("restore", e.to_string()))?; } Ok(()) diff --git a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs index 91ac69351..43a3d4591 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs @@ -329,7 +329,7 @@ mod tests { .scan_index_values(DB, TID, COLL, "status", 100) .unwrap() .into_iter() - .map(|(doc_id, _value)| doc_id) + .map(|(doc_id, _value)| doc_id.to_string()) .collect(); v.sort(); v diff --git a/nodedb/src/data/executor/handlers/truncate.rs b/nodedb/src/data/executor/handlers/truncate.rs index e285dc1a2..065ff4d93 100644 --- a/nodedb/src/data/executor/handlers/truncate.rs +++ b/nodedb/src/data/executor/handlers/truncate.rs @@ -106,13 +106,8 @@ impl CoreLoop { // `Put` record and resurrect its HNSW vector — mirrors // `execute_bulk_delete`'s `write_set` cascade. let mut write_set: Vec = Vec::new(); - for doc_id in &all_ids { - // `doc_id` is a bare string from a raw-table scan several calls - // removed from `SparseEngine`'s typed scan methods. A shape that - // fails to parse as a storage key can hold no row in DOCUMENTS - // either way, so `delete_in_txn` below sees the same "nothing to - // remove" outcome a lookup miss would have produced. - let storage_key = crate::engine::document::store::StorageKey::parse(doc_id); + for storage_key in &all_ids { + let doc_id = storage_key.to_string(); // One transaction per removed row, shared with the materialized-sum // delta that row owes — identical to `execute_bulk_delete`, so a @@ -121,12 +116,11 @@ impl CoreLoop { Ok(txn) => txn, Err(e) => return self.response_error(task, e), }; - let deleted_bytes = storage_key.and_then(|key| { - self.sparse - .delete_in_txn(&row_txn, database_id, tid, collection, &key) - .ok() - .flatten() - }); + let deleted_bytes = self + .sparse + .delete_in_txn(&row_txn, database_id, tid, collection, storage_key) + .ok() + .flatten(); let mut target_writes = Vec::new(); if let Some(bytes) = deleted_bytes.as_deref() { match write_hook::run( @@ -161,23 +155,21 @@ impl CoreLoop { } write_set.extend(write_hook::target_write_set(&target_writes)); if let Some(deleted_bytes) = deleted_bytes.as_deref() { - // doc_id is the hex-encoded surrogate (the redb storage key). - // `storage_key` already holds the parsed form. Non-hex keys - // (legacy non-surrogate docs) hold `None` and skip FTS. - if let Some(surrogate) = storage_key.map(|key| key.surrogate()) - && let Err(e) = self.inverted.remove_document( - database_id, - crate::types::TenantId::new(tid), - collection, - surrogate, - ) - { + let surrogate = storage_key.surrogate(); + if let Err(e) = self.inverted.remove_document( + database_id, + crate::types::TenantId::new(tid), + collection, + surrogate, + ) { warn!(core = self.core_id, %collection, %doc_id, error = %e, "truncate: inverted removal failed"); } - if let Err(e) = - self.sparse - .delete_indexes_for_document(database_id, tid, collection, doc_id) - { + if let Err(e) = self.sparse.delete_indexes_for_document( + database_id, + tid, + collection, + storage_key, + ) { warn!(core = self.core_id, %collection, %doc_id, error = %e, "truncate: index cascade failed"); } // Cascade: secondary HNSW vector index. The put path indexed @@ -186,38 +178,34 @@ impl CoreLoop { // the leaked vector keeps scoring in KNN search in the same // process (mirrors `execute_bulk_delete`'s vector cascade). if has_vectors { - self.remove_document_vector_indexes(database_id, tid, collection, doc_id); - if let Some(surrogate) = storage_key.map(|key| key.surrogate()) { - write_set.push(WriteSetEntry { - surrogate: surrogate.as_u32(), - is_delete: true, - value: Vec::new(), - collection: None, - }); - } + self.remove_document_vector_indexes(database_id, tid, collection, &doc_id); + write_set.push(WriteSetEntry { + surrogate: surrogate.as_u32(), + is_delete: true, + value: Vec::new(), + collection: None, + }); } let edges = self .csr_partition_mut(database_id, tid) - .remove_node_edges(doc_id); + .remove_node_edges(&doc_id); let cascade_ord = self.hlc.next_ordinal(); if edges > 0 && let Err(e) = self.edge_store.delete_edges_for_node( database_id, nodedb_types::TenantId::new(tid), - doc_id, + &doc_id, cascade_ord, ) { warn!(core = self.core_id, %doc_id, error = %e, "truncate: edge cascade failed"); } - if let Some(key) = storage_key { - self.doc_cache.invalidate( - task.request.database_id.as_u64(), - tid, - collection, - &key, - ); - } + self.doc_cache.invalidate( + task.request.database_id.as_u64(), + tid, + collection, + storage_key, + ); // Emit a delete event per removed row to the Event Plane, so // AFTER-DELETE triggers and CDC/change-stream consumers see // each row TRUNCATE removed — mirroring `execute_point_delete` @@ -235,9 +223,7 @@ impl CoreLoop { collection, deleted_bytes, ); - let identity = storage_key.map(|key| key.to_identity()).unwrap_or_else(|| { - crate::engine::document::store::RowIdentity::from_user_key(doc_id.as_str()) - }); + let identity = storage_key.to_identity(); self.emit_document_delete_event( task, collection, diff --git a/nodedb/src/engine/document/store/engine/batch.rs b/nodedb/src/engine/document/store/engine/batch.rs index de597038a..050ea7624 100644 --- a/nodedb/src/engine/document/store/engine/batch.rs +++ b/nodedb/src/engine/document/store/engine/batch.rs @@ -5,6 +5,7 @@ use std::collections::HashMap; use std::time::{SystemTime, UNIX_EPOCH}; +use crate::engine::document::store::StorageKey; use crate::engine::document::store::config::CollectionConfig; use crate::engine::sparse::btree::SparseEngine; @@ -73,16 +74,29 @@ impl<'a> DocumentEngine<'a> { path: &str, value: &str, bitemporal: bool, - ) -> crate::Result> { + ) -> crate::Result> { if bitemporal { - return self.sparse.versioned_index_lookup_as_of( + // The versioned index still yields text, so this parses at the boundary. + let ids = self.sparse.versioned_index_lookup_as_of( self.database_id, self.tenant_id, collection, path, value, None, - ); + )?; + return ids + .into_iter() + .map(|id| { + StorageKey::parse(&id).ok_or_else(|| { + crate::engine::sparse::btree::invalid_storage_key_err( + "INDEXES_VERSIONED", + collection, + &id, + ) + }) + }) + .collect(); } let prefix_with_value = format!("{value}:"); let results = @@ -105,7 +119,12 @@ impl<'a> DocumentEngine<'a> { self.database_id, self.tenant_id ); if key.starts_with(&expected_prefix) { - doc_ids.push(doc_id.to_string()); + let doc_id = StorageKey::parse(doc_id).ok_or_else(|| { + crate::engine::sparse::btree::invalid_storage_key_err( + "INDEXES", collection, doc_id, + ) + })?; + doc_ids.push(doc_id); } } } @@ -117,7 +136,6 @@ impl<'a> DocumentEngine<'a> { mod tests { use nodedb_types::Surrogate; - use crate::engine::document::store::StorageKey; use crate::engine::document::store::extract::json_to_msgpack; use super::*; @@ -157,7 +175,7 @@ mod tests { let results = doc_engine .index_lookup("users", "$.email", "alice@example.com", false) .unwrap(); - assert_eq!(results, vec![key(1).to_string()]); + assert_eq!(results, vec![key(1)]); } #[test] @@ -178,12 +196,12 @@ mod tests { let results = doc_engine .index_lookup("users", "$.tags", "admin", false) .unwrap(); - assert_eq!(results, vec![key(1).to_string()]); + assert_eq!(results, vec![key(1)]); let results = doc_engine .index_lookup("users", "$.tags", "editor", false) .unwrap(); - assert_eq!(results, vec![key(1).to_string()]); + assert_eq!(results, vec![key(1)]); } #[test] @@ -204,7 +222,7 @@ mod tests { let results = doc_engine .index_lookup("docs", "$.metadata.lang", "en", false) .unwrap(); - assert_eq!(results, vec![key(1).to_string()]); + assert_eq!(results, vec![key(1)]); } #[test] @@ -224,6 +242,6 @@ mod tests { let results = doc_engine .index_lookup("items", "$.category", "tools", false) .unwrap(); - assert_eq!(results, vec![key(1).to_string()]); + assert_eq!(results, vec![key(1)]); } } diff --git a/nodedb/src/engine/document/store/engine/delete.rs b/nodedb/src/engine/document/store/engine/delete.rs index 611c68c07..c84c5b76f 100644 --- a/nodedb/src/engine/document/store/engine/delete.rs +++ b/nodedb/src/engine/document/store/engine/delete.rs @@ -60,7 +60,7 @@ impl<'a> DocumentEngine<'a> { self.database_id, self.tenant_id, collection, - &doc_id_str, + doc_id, )?; Ok(self .sparse diff --git a/nodedb/src/engine/document/store/engine/put.rs b/nodedb/src/engine/document/store/engine/put.rs index 3e4b7c1c8..38e02ccf6 100644 --- a/nodedb/src/engine/document/store/engine/put.rs +++ b/nodedb/src/engine/document/store/engine/put.rs @@ -93,7 +93,7 @@ impl<'a> DocumentEngine<'a> { collection, &index_path.path, &v, - &doc_id_str, + doc_id, )?; } } diff --git a/nodedb/src/engine/sparse/btree/keys.rs b/nodedb/src/engine/sparse/btree/keys.rs index d6585842b..216692f4e 100644 --- a/nodedb/src/engine/sparse/btree/keys.rs +++ b/nodedb/src/engine/sparse/btree/keys.rs @@ -52,13 +52,15 @@ pub(super) fn with_tenant_key( } /// Build a database/tenant-scoped index key `"{db}:{tenant}:{a}:{b}:{c}:{d}"`. +/// `d` is written via `Display`, so a [`nodedb_types::StorageKey`] lands as +/// its 8 hex characters with no intermediate `String`. pub(in crate::engine::sparse) fn with_tenant_key4( database_id: u64, tenant_id: u64, a: &str, b: &str, c: &str, - d: &str, + d: impl std::fmt::Display, f: impl FnOnce(&str) -> R, ) -> R { KEY_BUF.with(|buf| { @@ -75,7 +77,7 @@ pub(in crate::engine::sparse) fn with_tenant_key4( buf.push(':'); buf.push_str(c); buf.push(':'); - buf.push_str(d); + let _ = write!(buf, "{d}"); f(&buf) }) } diff --git a/nodedb/src/engine/sparse/btree/mod.rs b/nodedb/src/engine/sparse/btree/mod.rs index 49729b912..556c2e46f 100644 --- a/nodedb/src/engine/sparse/btree/mod.rs +++ b/nodedb/src/engine/sparse/btree/mod.rs @@ -12,5 +12,5 @@ pub mod tables; pub use engine::SparseEngine; pub(crate) use keys::coll_prefix; pub(in crate::engine::sparse) use keys::{tenant_prefix, with_tenant_key4}; -pub(crate) use tables::DOCUMENTS; -pub(in crate::engine::sparse) use tables::{INDEXES, invalid_storage_key_err, redb_err}; +pub(crate) use tables::{DOCUMENTS, invalid_storage_key_err}; +pub(in crate::engine::sparse) use tables::{INDEXES, redb_err}; diff --git a/nodedb/src/engine/sparse/btree/tables.rs b/nodedb/src/engine/sparse/btree/tables.rs index ed9c10240..98757f290 100644 --- a/nodedb/src/engine/sparse/btree/tables.rs +++ b/nodedb/src/engine/sparse/btree/tables.rs @@ -22,18 +22,16 @@ pub(in crate::engine::sparse) fn redb_err(ctx: &str, e: E) } } -/// Report a DOCUMENTS row whose key does not parse as a [`nodedb_types::StorageKey`]. +/// Report a row on `table` whose key does not parse as a [`nodedb_types::StorageKey`]. /// -/// A non-parsing key on this table is a violated storage invariant, not a -/// legacy row to skip: every DOCUMENTS key is minted by [`StorageKey::for_surrogate`]. -pub(in crate::engine::sparse) fn invalid_storage_key_err( - collection: &str, - key: &str, -) -> crate::Error { +/// A non-parsing key on any of DOCUMENTS, INDEXES, or INDEXES_VERSIONED is a +/// violated storage invariant, not a legacy row to skip: every stored key is +/// minted by [`StorageKey::for_surrogate`]. +pub(crate) fn invalid_storage_key_err(table: &str, collection: &str, key: &str) -> crate::Error { crate::Error::Storage { engine: "sparse".into(), detail: format!( - "collection '{collection}' has a DOCUMENTS row whose key is not a valid storage key: '{key}'" + "collection '{collection}' has a {table} row whose key is not a valid storage key: '{key}'" ), } } diff --git a/nodedb/src/engine/sparse/btree_index.rs b/nodedb/src/engine/sparse/btree_index.rs index ab2544749..8525f4d6c 100644 --- a/nodedb/src/engine/sparse/btree_index.rs +++ b/nodedb/src/engine/sparse/btree_index.rs @@ -5,10 +5,13 @@ //! Index key format: `"{database_id}:{tenant_id}:{collection}:{field}:{value}:{document_id}"`. //! Extracted from `btree.rs` — document CRUD stays there, index ops live here. +use nodedb_types::StorageKey; use redb::{ReadableDatabase, ReadableTable, WriteTransaction}; use tracing::debug; -use super::btree::{DOCUMENTS, INDEXES, SparseEngine, coll_prefix, redb_err}; +use super::btree::{ + DOCUMENTS, INDEXES, SparseEngine, coll_prefix, invalid_storage_key_err, redb_err, +}; /// Identifies a single secondary-index entry for an in-txn mutation. /// @@ -21,7 +24,7 @@ pub struct IndexEntryTxn<'a> { pub collection: &'a str, pub field: &'a str, pub value: &'a str, - pub document_id: &'a str, + pub document_id: &'a StorageKey, } /// Parameters for [`SparseEngine::range_scan`]. @@ -46,7 +49,7 @@ impl SparseEngine { database_id: u64, tenant_id: u64, collection: &str, - document_id: &str, + document_id: &StorageKey, ) -> crate::Result<()> { let write_txn = self .db @@ -79,7 +82,7 @@ impl SparseEngine { database_id: u64, tenant_id: u64, collection: &str, - document_id: &str, + document_id: &StorageKey, ) -> crate::Result<()> { let prefix = coll_prefix(database_id, tenant_id, collection); let end = format!("{prefix}\u{ffff}"); @@ -231,7 +234,7 @@ impl SparseEngine { collection: &str, field: &str, value: &str, - document_id: &str, + document_id: &StorageKey, ) -> crate::Result<()> { super::btree::with_tenant_key4( database_id, @@ -368,7 +371,7 @@ impl SparseEngine { collection: &str, field: &str, value: &str, - document_id: &str, + document_id: &StorageKey, ) -> crate::Result<()> { super::btree::with_tenant_key4( database_id, @@ -402,7 +405,7 @@ impl SparseEngine { collection: &str, field: &str, limit: usize, - ) -> crate::Result> { + ) -> crate::Result> { let prefix = format!( "{}{field}:", coll_prefix(database_id, tenant_id, collection) @@ -430,7 +433,9 @@ impl SparseEngine { { let value = &rest[..colon_pos]; let doc_id = &rest[colon_pos + 1..]; - results.push((doc_id.to_string(), value.to_string())); + let doc_id = StorageKey::parse(doc_id) + .ok_or_else(|| invalid_storage_key_err("INDEXES", collection, doc_id))?; + results.push((doc_id, value.to_string())); } } @@ -494,6 +499,8 @@ pub fn index_key_for(entry: IndexEntryTxn<'_>) -> String { #[cfg(test)] mod tests { + use nodedb_types::Surrogate; + use super::*; fn open_temp() -> (SparseEngine, tempfile::TempDir) { @@ -502,13 +509,25 @@ mod tests { (engine, dir) } + fn key(surrogate: u32) -> StorageKey { + StorageKey::for_surrogate(Surrogate::new(surrogate)) + } + #[test] fn range_scan_with_index() { let (engine, _dir) = open_temp(); - engine.index_put(0, 1, "users", "age", "025", "u1").unwrap(); - engine.index_put(0, 1, "users", "age", "030", "u2").unwrap(); - engine.index_put(0, 1, "users", "age", "035", "u3").unwrap(); - engine.index_put(0, 1, "users", "age", "040", "u4").unwrap(); + engine + .index_put(0, 1, "users", "age", "025", &key(1)) + .unwrap(); + engine + .index_put(0, 1, "users", "age", "030", &key(2)) + .unwrap(); + engine + .index_put(0, 1, "users", "age", "035", &key(3)) + .unwrap(); + engine + .index_put(0, 1, "users", "age", "040", &key(4)) + .unwrap(); let results = engine .range_scan(RangeScanParams { database_id: 0, @@ -527,13 +546,17 @@ mod tests { fn delete_index_entries_for_field() { let (engine, _dir) = open_temp(); engine - .index_put(0, 1, "users", "email", "alice@example.com", "u1") + .index_put(0, 1, "users", "email", "alice@example.com", &key(1)) + .unwrap(); + engine + .index_put(0, 1, "users", "email", "bob@example.com", &key(2)) + .unwrap(); + engine + .index_put(0, 1, "users", "age", "30", &key(1)) .unwrap(); engine - .index_put(0, 1, "users", "email", "bob@example.com", "u2") + .index_put(0, 1, "users", "age", "25", &key(2)) .unwrap(); - engine.index_put(0, 1, "users", "age", "30", "u1").unwrap(); - engine.index_put(0, 1, "users", "age", "25", "u2").unwrap(); let removed = engine .delete_index_entries_for_field(0, 1, "users", "email") .unwrap(); diff --git a/nodedb/src/engine/sparse/btree_scan.rs b/nodedb/src/engine/sparse/btree_scan.rs index 474549f2d..c79734ed3 100644 --- a/nodedb/src/engine/sparse/btree_scan.rs +++ b/nodedb/src/engine/sparse/btree_scan.rs @@ -44,7 +44,7 @@ impl SparseEngine { // Extract document_id from key format "{database_id}:{tenant}:{collection}:{doc_id}" let doc_id = key.strip_prefix(&prefix).unwrap_or(key); let storage_key = StorageKey::parse(doc_id) - .ok_or_else(|| invalid_storage_key_err(collection, doc_id))?; + .ok_or_else(|| invalid_storage_key_err("DOCUMENTS", collection, doc_id))?; let value = entry.1.value().to_vec(); results.push((storage_key, value)); } @@ -100,7 +100,7 @@ impl SparseEngine { // Extract document_id from key format "{database_id}:{tenant}:{collection}:{doc_id}" let doc_id = key.strip_prefix(&prefix).unwrap_or(key); let storage_key = StorageKey::parse(doc_id) - .ok_or_else(|| invalid_storage_key_err(collection, doc_id))?; + .ok_or_else(|| invalid_storage_key_err("DOCUMENTS", collection, doc_id))?; let value = entry.1.value(); f(&storage_key, value)?; count += 1; @@ -152,7 +152,7 @@ impl SparseEngine { let key = entry.0.value(); let doc_id = key.strip_prefix(&prefix).unwrap_or(key); let storage_key = StorageKey::parse(doc_id) - .ok_or_else(|| invalid_storage_key_err(collection, doc_id))?; + .ok_or_else(|| invalid_storage_key_err("DOCUMENTS", collection, doc_id))?; let value = entry.1.value().to_vec(); chunk.push((storage_key, value)); total += 1; @@ -331,7 +331,7 @@ impl SparseEngine { let key = entry.0.value(); let doc_id = key.strip_prefix(&prefix).unwrap_or(key); let storage_key = StorageKey::parse(doc_id) - .ok_or_else(|| invalid_storage_key_err(collection, doc_id))?; + .ok_or_else(|| invalid_storage_key_err("DOCUMENTS", collection, doc_id))?; // Evaluate predicate on raw bytes — skip allocation if no match. if !predicate(&storage_key, value_bytes) { From 203a48c92e1a14da4bddfbf0b88ede08903bda94 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 12 Sep 2026 08:27:36 +0800 Subject: [PATCH 09/17] refactor(sparse): use StorageKey for versioned document and index keys Replace raw string doc_id parameters with the typed StorageKey across the versioned btree key layout, document store engine, and every executor handler that reads or writes documents. This removes the NUL-byte validation that existed only to protect string-based keys and derives PartialOrd/Ord/Hash on StorageKey so it can be used directly as a key and in index structures. invalid_storage_key_err now takes a KeyedTable enum instead of a raw table name string, covering the versioned documents and indexes tables alongside the existing ones. --- nodedb-types/src/row_identity.rs | 2 +- .../enforcement/materialized_sum/apply.rs | 5 +- .../enforcement/materialized_sum/rmw.rs | 19 +- .../data/executor/handlers/bulk_dml/scan.rs | 2 +- .../control/calvin_overlay_stage_bulk.rs | 8 +- .../handlers/control/range_scan_versioned.rs | 36 +++- .../executor/handlers/document/index_fetch.rs | 24 ++- .../executor/handlers/document/read/fetch.rs | 139 ++++++------- .../executor/handlers/document/read/mod.rs | 1 - .../handlers/document/resolve/context.rs | 2 +- .../executor/handlers/point/apply_delete.rs | 8 +- .../executor/handlers/point/apply_put/core.rs | 4 +- .../data/executor/handlers/point/delete.rs | 4 +- .../src/data/executor/handlers/point/get.rs | 7 +- .../data/executor/handlers/point/insert.rs | 12 +- .../executor/handlers/point/update/exec.rs | 2 +- .../executor/handlers/point/update/persist.rs | 9 +- .../executor/handlers/point/update_reindex.rs | 8 +- .../executor/handlers/transaction/batch.rs | 4 +- .../transaction/index_write_values.rs | 6 +- .../transaction/stage_write/constraint.rs | 3 +- .../stage_write/stage_point_document.rs | 21 +- .../transaction/stage_write/stage_upsert.rs | 6 +- .../transaction/sub_plan_doc/delete.rs | 8 +- .../handlers/transaction/sub_plan_doc/put.rs | 8 +- .../handlers/transaction/undo/document.rs | 106 ++++------ .../handlers/transaction/undo/entry.rs | 18 +- .../handlers/transaction/undo/rollback.rs | 35 ++-- .../executor/handlers/upsert/exec/dispatch.rs | 4 +- nodedb/src/data/executor/scan_versioned.rs | 13 +- .../src/engine/document/store/engine/batch.rs | 6 +- .../engine/document/store/engine/delete.rs | 9 +- .../src/engine/document/store/engine/get.rs | 10 +- .../src/engine/document/store/engine/put.rs | 7 +- nodedb/src/engine/sparse/btree/mod.rs | 2 +- nodedb/src/engine/sparse/btree/tables.rs | 34 +++- nodedb/src/engine/sparse/btree_index.rs | 7 +- nodedb/src/engine/sparse/btree_scan.rs | 23 ++- .../src/engine/sparse/btree_versioned/doc.rs | 188 +++++++++--------- .../src/engine/sparse/btree_versioned/key.rs | 37 +--- .../src/engine/sparse/btree_versioned/mod.rs | 11 +- .../engine/sparse/btree_versioned/purge.rs | 72 ++++--- .../src/engine/sparse/btree_versioned/scan.rs | 43 ++-- .../engine/sparse/btree_versioned/value.rs | 4 +- .../inproc/cases/bitemporal_temporal_purge.rs | 11 +- .../inproc/cases/document_bitemporal_dml.rs | 6 +- .../inproc/cases/document_bitemporal_store.rs | 4 +- 47 files changed, 487 insertions(+), 511 deletions(-) diff --git a/nodedb-types/src/row_identity.rs b/nodedb-types/src/row_identity.rs index e12e69893..0cff13b1d 100644 --- a/nodedb-types/src/row_identity.rs +++ b/nodedb-types/src/row_identity.rs @@ -30,7 +30,7 @@ use crate::Surrogate; /// Fixed-width lowercase hex, so lexicographic order matches surrogate order /// and a range scan iterates rows in surrogate order with no extra index. /// Internal: a storage key never reaches a client. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)] pub struct StorageKey(Surrogate); impl StorageKey { diff --git a/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs b/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs index b2ca030b4..39f53e7f4 100644 --- a/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs +++ b/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs @@ -84,11 +84,8 @@ use crate::types::DatabaseId; pub(in crate::data::executor) struct TargetWrite { /// Target collection name. pub collection: String, - /// Storage key of the target row — the hex-encoded surrogate. - pub document_id: String, /// The target row's surrogate, so an undo entry addresses the same identity - /// the forward write used. The old code had no surrogate to record and - /// pushed `Surrogate::ZERO`. + /// the forward write used. pub surrogate: Surrogate, /// The MessagePack body this write handed to `apply_point_put` — NOT the /// bytes that reached storage. diff --git a/nodedb/src/data/executor/enforcement/materialized_sum/rmw.rs b/nodedb/src/data/executor/enforcement/materialized_sum/rmw.rs index cdb8f03ea..a357c97ad 100644 --- a/nodedb/src/data/executor/enforcement/materialized_sum/rmw.rs +++ b/nodedb/src/data/executor/enforcement/materialized_sum/rmw.rs @@ -35,7 +35,6 @@ use crate::data::executor::doc_format; use crate::data::executor::handlers::document::read::decode::decode_scanned_document; use crate::data::executor::handlers::point::apply_put::PointPutParams; use crate::data::executor::sparse_body_format::SparseBodyFormat; -use crate::engine::document::store::surrogate_to_doc_id; use crate::types::{DatabaseId, Lsn, TenantId}; /// Everything one balance move needs, independent of which transaction it lands @@ -90,7 +89,9 @@ impl CoreLoop { txn: &WriteTransaction, params: &BalanceRmw<'_>, ) -> crate::Result { - let document_id = surrogate_to_doc_id(params.surrogate); + let storage_key = nodedb_types::StorageKey::for_surrogate(params.surrogate); + // `PointPutParams` still carries the document id as text. + let document_id = storage_key.to_string(); // The TARGET collection's encoding is resolved from `doc_configs`, not // assumed: the target is a different collection from the source and may @@ -116,8 +117,7 @@ impl CoreLoop { }); } - let Some(old_bytes) = self.read_balance_row(txn, params, &document_id, params.surrogate)? - else { + let Some(old_bytes) = self.read_balance_row(txn, params, &storage_key)? else { return Err(params.target_not_found()); }; @@ -183,7 +183,7 @@ impl CoreLoop { params.database_id, params.tid, params.target_collection, - &nodedb_types::StorageKey::for_surrogate(params.surrogate), + &storage_key, ); return Err(e); } @@ -191,7 +191,6 @@ impl CoreLoop { Ok(TargetWrite { collection: params.target_collection.to_string(), - document_id, surrogate: params.surrogate, body, outcome, @@ -208,24 +207,22 @@ impl CoreLoop { &self, txn: &WriteTransaction, params: &BalanceRmw<'_>, - document_id: &str, - surrogate: Surrogate, + storage_key: &nodedb_types::StorageKey, ) -> crate::Result>> { if self.is_bitemporal(params.database_id, params.tid, params.target_collection) { self.sparse.versioned_get_current( params.database_id, params.tid, params.target_collection, - document_id, + storage_key, ) } else { - let key = nodedb_types::StorageKey::for_surrogate(surrogate); self.sparse.get_in_txn( txn, params.database_id, params.tid, params.target_collection, - &key, + storage_key, ) } } diff --git a/nodedb/src/data/executor/handlers/bulk_dml/scan.rs b/nodedb/src/data/executor/handlers/bulk_dml/scan.rs index 93cf53b65..c98eb7fed 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/scan.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/scan.rs @@ -49,7 +49,7 @@ impl CoreLoop { { let doc_id = StorageKey::parse(doc_id).ok_or_else(|| { crate::engine::sparse::btree::invalid_storage_key_err( - "DOCUMENTS", + crate::engine::sparse::btree::KeyedTable::Documents, collection, doc_id, ) diff --git a/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs b/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs index 6a384466c..466bf1226 100644 --- a/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs +++ b/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs @@ -42,7 +42,6 @@ use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::bulk_dml::scan::ollp_predicted_doc_ids; use crate::data::executor::handlers::transaction::overlay::Staged; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; use crate::types::{DatabaseId, TenantId, TxnId}; /// Loudly reject a Calvin bulk predicate plan that reached overlay staging @@ -176,7 +175,6 @@ impl CoreLoop { predicted_sorted.sort_unstable(); for surrogate in predicted_sorted { - let doc_id = surrogate_to_doc_id(Surrogate::new(surrogate)); let storage_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); // Current body: overlay wins over base (read-your-own-writes), @@ -195,7 +193,7 @@ impl CoreLoop { database_id.as_u64(), tid, collection, - &doc_id, + &storage_key, ) } else { self.sparse @@ -219,7 +217,7 @@ impl CoreLoop { )?; // Decide the staged post-image against the write policy: this is // the row the Calvin flush will install. - let identity = crate::engine::document::store::identity_of(&doc_id); + let identity = storage_key.to_identity(); self.stage_admit_write( rls_write_check, &new_body, @@ -228,6 +226,8 @@ impl CoreLoop { tid, collection, )?; + // The overlay's doc-id side map is keyed by text. + let doc_id = storage_key.to_string(); self.stage_bulk_put_capped(txn_id, &coll_key, surrogate, &doc_id, new_body)?; } Ok(()) diff --git a/nodedb/src/data/executor/handlers/control/range_scan_versioned.rs b/nodedb/src/data/executor/handlers/control/range_scan_versioned.rs index 753247c87..99c9b617c 100644 --- a/nodedb/src/data/executor/handlers/control/range_scan_versioned.rs +++ b/nodedb/src/data/executor/handlers/control/range_scan_versioned.rs @@ -106,15 +106,17 @@ impl CoreLoop { // Predicate: decode each current body, extract `field`, keep in-range // rows. `extract_index_values(_, field, false)` yields the scalar // string form for the path (0 or 1 value for a non-array field). - // The scan API's predicate is `Fn(&str, &[u8]) -> bool`, so an + // The scan API's predicate is `Fn(&StorageKey, &[u8]) -> bool`, so an // undecodable body is captured through this `Cell` side-channel and // checked once the scan finishes. Returning `false` and moving on // would drop the row from the answer with nothing anywhere saying a // row was dropped, which reads to the client as a smaller — but // correct-looking — result set. let decode_err: std::cell::Cell> = std::cell::Cell::new(None); - let predicate = |doc_id: &str, body: &[u8]| match decode_body(body, strict_schema.as_ref()) - { + let predicate = |doc_id: &nodedb_types::StorageKey, body: &[u8]| match decode_body( + body, + strict_schema.as_ref(), + ) { Err(e) => { decode_err.set(Some(e)); false @@ -140,7 +142,7 @@ impl CoreLoop { Some(filters) => match nodedb_types::json_msgpack::json_to_msgpack(&doc) { Ok(mp) => { let mp = if strict_schema.is_none() { - let identity = crate::engine::document::store::identity_of(doc_id); + let identity = doc_id.to_identity(); nodedb_query::msgpack_scan::inject_str_field( &mp, "id", @@ -168,7 +170,7 @@ impl CoreLoop { // version. A statement that goes over its deadline mid-scan stops here // and the check below turns the short result into an error. let deadline = crate::data::executor::deadline::DeadlineCheck::for_task(task); - let mut scanned = match self.sparse.versioned_scan_as_of( + let scanned = match self.sparse.versioned_scan_as_of( crate::engine::sparse::btree_versioned::VersionedScanParams { database_id: task.request.database_id.as_u64(), tenant: tid, @@ -192,6 +194,28 @@ impl CoreLoop { } }; + // `merge_overlay_into_scan` operates on hex-rendered keys — the same + // text shape the base (non-versioned) scan path carries — so the + // typed keys from the versioned scan render once here at the + // boundary, and the predicate re-parses back to a `StorageKey` per + // candidate row. + let mut scanned: Vec<(String, Vec)> = scanned + .into_iter() + .map(|(k, v)| (k.to_string(), v)) + .collect(); + let predicate_str = + |doc_id: &str, body: &[u8]| match nodedb_types::StorageKey::parse(doc_id) { + Some(key) => predicate(&key, body), + None => { + decode_err.set(Some(crate::engine::sparse::btree::invalid_storage_key_err( + crate::engine::sparse::btree::KeyedTable::DocumentsVersioned, + collection, + doc_id, + ))); + false + } + }; + // Read-your-own-writes: fold this transaction's staging overlay onto // the current-version base result, using the SAME range predicate on // the raw stored bodies (staged strict bodies are Binary Tuples, like @@ -204,7 +228,7 @@ impl CoreLoop { crate::types::TenantId::new(tid), collection.to_string(), ); - self.merge_overlay_into_scan(txn_id, &coll_key, &mut scanned, &predicate); + self.merge_overlay_into_scan(txn_id, &coll_key, &mut scanned, &predicate_str); } // Both passes above run the same predicate; check the side-channel once, diff --git a/nodedb/src/data/executor/handlers/document/index_fetch.rs b/nodedb/src/data/executor/handlers/document/index_fetch.rs index e99f87356..5827474f3 100644 --- a/nodedb/src/data/executor/handlers/document/index_fetch.rs +++ b/nodedb/src/data/executor/handlers/document/index_fetch.rs @@ -272,18 +272,20 @@ impl CoreLoop { // the staged `Put` bytes and only falls back to a base fetch // when the overlay has nothing staged for this surrogate. let fetched = self.overlay_or_base_body(task.request.txn_id, &coll_key, doc_id, || { - if bitemporal { - self.sparse - .versioned_get_current(database_id, tid, collection, doc_id) - } else { - // `doc_id` is an index-lookup result, not a scan of - // DOCUMENTS itself; a shape that fails to parse as a - // storage key names no row in that table, matching what - // a lookup on the unparsed key would already have found. - match nodedb_types::StorageKey::parse(doc_id) { - Some(key) => self.sparse.get(database_id, tid, collection, &key), - None => Ok(None), + // `doc_id` is an index-lookup result, not a scan of DOCUMENTS + // itself; a shape that fails to parse as a storage key names + // no row in that table, matching what a lookup on the + // unparsed key would already have found. + match nodedb_types::StorageKey::parse(doc_id) { + Some(key) => { + if bitemporal { + self.sparse + .versioned_get_current(database_id, tid, collection, &key) + } else { + self.sparse.get(database_id, tid, collection, &key) + } } + None => Ok(None), } }); match fetched { diff --git a/nodedb/src/data/executor/handlers/document/read/fetch.rs b/nodedb/src/data/executor/handlers/document/read/fetch.rs index 8c09192d9..39a06caa7 100644 --- a/nodedb/src/data/executor/handlers/document/read/fetch.rs +++ b/nodedb/src/data/executor/handlers/document/read/fetch.rs @@ -26,7 +26,7 @@ use nodedb_types::StorageKey; use super::audit_body::{inject_temporal_columns, strict_audit_body}; use super::fetch_types::FetchedRows; -use super::{DocFetchParams, DocScanMode, parse_fetched_key}; +use super::{DocFetchParams, DocScanMode}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::core_loop::filter_match::matches_with_resolved_schema; use crate::data::executor::scan_normalize::{sparse_body_to_msgpack, sparse_row_to_doc}; @@ -68,24 +68,24 @@ impl CoreLoop { // msgpack) operates uniformly, then hand it downstream with no // schema (bodies are already normalized). // `versioned_scan_as_of` takes an infallible - // `Fn(&str, &[u8]) -> bool` predicate (a storage-engine - // primitive out of scope for this fix), so a + // `Fn(&StorageKey, &[u8]) -> bool` predicate, so a // division/modulo-by-zero is captured via this `Cell` // side-channel and checked once the scan returns, rather // than silently folded away. let predicate_err: Cell> = Cell::new(None); - let predicate = |doc_id: &str, body: &[u8]| match matches_with_resolved_schema( - strict_schema, - filter_predicates, - doc_id, - body, - ) { - Ok(b) => b, - Err(e) => { - predicate_err.set(Some(e)); - false - } - }; + let predicate = + |doc_id: &StorageKey, body: &[u8]| match matches_with_resolved_schema( + strict_schema, + filter_predicates, + &doc_id.to_string(), + body, + ) { + Ok(b) => b, + Err(e) => { + predicate_err.set(Some(e)); + false + } + }; let raw = self.sparse.versioned_scan_as_of( crate::engine::sparse::btree_versioned::VersionedScanParams { database_id: task.request.database_id.as_u64(), @@ -114,7 +114,7 @@ impl CoreLoop { SparseBodyFormatRef::from_schema(strict_schema), ) .into_owned(); - (doc_id, mp) + (doc_id.to_string(), mp) }) .collect(); Ok(FetchedRows { @@ -130,18 +130,19 @@ impl CoreLoop { // so a user can `SELECT` / `ORDER BY` / project on them. // See the `AsOf` arm above for the `Cell` side-channel rationale. let predicate_err: Cell> = Cell::new(None); - let predicate = |doc_id: &str, body: &[u8]| match matches_with_resolved_schema( - strict_schema, - filter_predicates, - doc_id, - body, - ) { - Ok(b) => b, - Err(e) => { - predicate_err.set(Some(e)); - false - } - }; + let predicate = + |doc_id: &StorageKey, body: &[u8]| match matches_with_resolved_schema( + strict_schema, + filter_predicates, + &doc_id.to_string(), + body, + ) { + Ok(b) => b, + Err(e) => { + predicate_err.set(Some(e)); + false + } + }; let raw = self.sparse.versioned_scan_all( crate::engine::sparse::btree_versioned::VersionedScanParams { database_id: task.request.database_id.as_u64(), @@ -169,7 +170,7 @@ impl CoreLoop { row.valid_from_ms, row.valid_until_ms, )?; - rows.push((row.doc_id, with_ts)); + rows.push((row.doc_id.to_string(), with_ts)); } Ok(FetchedRows { rows, @@ -225,12 +226,10 @@ impl CoreLoop { SparseBodyFormat::VectorSidecar ); - // `scan_documents_filtered`/`versioned_scan_as_of` - // take an infallible `Fn(&str, &[u8]) -> bool` predicate (a - // storage-engine primitive out of scope for this fix), so a - // division/modulo-by-zero is captured via this `Cell` side-channel - // and checked once every branch below returns, rather than silently - // folded away. + // `scan_documents_filtered`/`versioned_scan_as_of` take an infallible + // predicate, so a division/modulo-by-zero is captured via this `Cell` + // side-channel and checked once every branch below returns, rather + // than silently folded away. let predicate_err: Cell> = Cell::new(None); let matches = |doc_id: &str, value: &[u8]| -> bool { if filter_predicates.is_empty() { @@ -255,37 +254,25 @@ impl CoreLoop { } } }; - // `scan_documents_filtered` hands the predicate a typed `StorageKey`; - // `matches` (and `versioned_scan_as_of`'s predicate) still take the - // row's storage key as text, so this renders it once per candidate row. + // `scan_documents_filtered` and `versioned_scan_as_of` hand the + // predicate a typed `StorageKey`; `matches` still takes the row's + // storage key as text, so this renders it once per candidate row. let matches_by_key = |key: &StorageKey, value: &[u8]| matches(&key.to_string(), value); - // `versioned_scan_as_of` hands back a rendered storage key, parsed - // back ONCE here rather than carried as text and re-parsed later. A - // shape that fails to parse names a fetch-pipeline bug, not a row to - // skip. - let parse_row_key = |id: String, body: Vec| -> crate::Result<(StorageKey, Vec)> { - Ok((parse_fetched_key(collection, &id)?, body)) - }; - let rows: Vec<(StorageKey, Vec)> = if filter_predicates.is_empty() { if bitemporal { - self.sparse - .versioned_scan_as_of( - crate::engine::sparse::btree_versioned::VersionedScanParams { - database_id, - tenant: tid, - coll: collection, - sys_cutoff_ms: None, - valid_at_ms: None, - limit: fetch_limit, - }, - &|_, _| true, - &stop, - )? - .into_iter() - .map(|(id, body)| parse_row_key(id, body)) - .collect::>>()? + self.sparse.versioned_scan_as_of( + crate::engine::sparse::btree_versioned::VersionedScanParams { + database_id, + tenant: tid, + coll: collection, + sys_cutoff_ms: None, + valid_at_ms: None, + limit: fetch_limit, + }, + &|_: &StorageKey, _: &[u8]| true, + &stop, + )? } else { // Routed through the filtered scan with an always-true // predicate rather than `scan_documents`: the unfiltered full @@ -302,22 +289,18 @@ impl CoreLoop { )? } } else if bitemporal { - self.sparse - .versioned_scan_as_of( - crate::engine::sparse::btree_versioned::VersionedScanParams { - database_id, - tenant: tid, - coll: collection, - sys_cutoff_ms: None, - valid_at_ms: None, - limit: fetch_limit, - }, - &matches, - &stop, - )? - .into_iter() - .map(|(id, body)| parse_row_key(id, body)) - .collect::>>()? + self.sparse.versioned_scan_as_of( + crate::engine::sparse::btree_versioned::VersionedScanParams { + database_id, + tenant: tid, + coll: collection, + sys_cutoff_ms: None, + valid_at_ms: None, + limit: fetch_limit, + }, + &matches_by_key, + &stop, + )? } else { self.sparse.scan_documents_filtered( database_id, diff --git a/nodedb/src/data/executor/handlers/document/read/mod.rs b/nodedb/src/data/executor/handlers/document/read/mod.rs index f2fd5054c..45fc1b974 100644 --- a/nodedb/src/data/executor/handlers/document/read/mod.rs +++ b/nodedb/src/data/executor/handlers/document/read/mod.rs @@ -11,5 +11,4 @@ pub mod materialize_scan; pub mod projection; pub mod scan; -use fetch_types::parse_fetched_key; pub(in crate::data::executor) use fetch_types::{DocFetchParams, DocScanMode}; diff --git a/nodedb/src/data/executor/handlers/document/resolve/context.rs b/nodedb/src/data/executor/handlers/document/resolve/context.rs index 255c677b7..496fe2907 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/context.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/context.rs @@ -83,7 +83,7 @@ impl CoreLoop { ) -> Result>, ErrorCode> { let read = if self.is_bitemporal(database_id, tid, collection) { self.sparse - .versioned_get_current(database_id, tid, collection, &row_key.to_string()) + .versioned_get_current(database_id, tid, collection, row_key) } else { self.sparse.get(database_id, tid, collection, row_key) }; diff --git a/nodedb/src/data/executor/handlers/point/apply_delete.rs b/nodedb/src/data/executor/handlers/point/apply_delete.rs index 31a59b640..a3c3a59df 100644 --- a/nodedb/src/data/executor/handlers/point/apply_delete.rs +++ b/nodedb/src/data/executor/handlers/point/apply_delete.rs @@ -163,9 +163,9 @@ impl CoreLoop { let mut bitemporal_sys_from_ms: Option = None; let mut bitemporal_index_tuples: Vec<(String, String)> = Vec::new(); let prior = if bitemporal { - let prior = self - .sparse - .versioned_get_current(database_id, tid, collection, row_key)?; + let prior = + self.sparse + .versioned_get_current(database_id, tid, collection, &storage_key)?; if let Some(ref body) = prior { if enforce && let Some(config) = self.doc_configs.get(&config_key) { run_delete_enforcement( @@ -185,7 +185,7 @@ impl CoreLoop { database_id, tid, collection, - row_key, + &storage_key, sys_from, )?; // Index tombstones: reflect every current value so diff --git a/nodedb/src/data/executor/handlers/point/apply_put/core.rs b/nodedb/src/data/executor/handlers/point/apply_put/core.rs index b05d26086..3afa0f576 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/core.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/core.rs @@ -101,7 +101,7 @@ impl CoreLoop { .is_some_and(|config| !config.index_paths.is_empty()); let old_value = if bitemporal { self.sparse - .versioned_get_current(database_id, tid, collection, document_id)? + .versioned_get_current(database_id, tid, collection, &storage_key)? } else if need_old { self.sparse .get(database_id, tid, collection, &storage_key)? @@ -147,7 +147,7 @@ impl CoreLoop { database_id, tenant: tid, coll: collection, - doc_id: document_id, + doc_id: &storage_key, sys_from_ms, valid_from_ms, valid_until_ms, diff --git a/nodedb/src/data/executor/handlers/point/delete.rs b/nodedb/src/data/executor/handlers/point/delete.rs index b9f8fd104..66458181d 100644 --- a/nodedb/src/data/executor/handlers/point/delete.rs +++ b/nodedb/src/data/executor/handlers/point/delete.rs @@ -279,11 +279,9 @@ impl CoreLoop { } let database_id = task.request.database_id.as_u64(); let storage_key = crate::engine::document::store::StorageKey::for_surrogate(surrogate); - let row_key = storage_key.to_string(); - let row_key = row_key.as_str(); let stored = if self.is_bitemporal(database_id, tid, collection) { self.sparse - .versioned_get_current(database_id, tid, collection, row_key)? + .versioned_get_current(database_id, tid, collection, &storage_key)? } else { self.sparse .get(database_id, tid, collection, &storage_key)? diff --git a/nodedb/src/data/executor/handlers/point/get.rs b/nodedb/src/data/executor/handlers/point/get.rs index 35e1f6594..5e787cdf5 100644 --- a/nodedb/src/data/executor/handlers/point/get.rs +++ b/nodedb/src/data/executor/handlers/point/get.rs @@ -8,7 +8,6 @@ use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::scan_normalize::{sparse_body_to_msgpack, sparse_row_to_doc}; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; use nodedb_types::Surrogate; pub(in crate::data::executor) struct PointGetParams<'a> { @@ -39,8 +38,6 @@ impl CoreLoop { system_as_of_ms, valid_at_ms, } = p; - let row_key = surrogate_to_doc_id(surrogate); - let row_key = row_key.as_str(); let storage_key = nodedb_types::StorageKey::for_surrogate(surrogate); debug!( core = self.core_id, @@ -75,7 +72,7 @@ impl CoreLoop { database_id, tid, collection, - row_key, + &storage_key, system_as_of_ms, valid_at_ms, ) { @@ -107,7 +104,7 @@ impl CoreLoop { } else { let res = if bitemporal { self.sparse - .versioned_get_current(database_id, tid, collection, row_key) + .versioned_get_current(database_id, tid, collection, &storage_key) } else { self.sparse.get(database_id, tid, collection, &storage_key) }; diff --git a/nodedb/src/data/executor/handlers/point/insert.rs b/nodedb/src/data/executor/handlers/point/insert.rs index 795017e14..f6c469e9a 100644 --- a/nodedb/src/data/executor/handlers/point/insert.rs +++ b/nodedb/src/data/executor/handlers/point/insert.rs @@ -67,7 +67,6 @@ impl CoreLoop { } = p; let storage_key = StorageKey::for_surrogate(surrogate); let row_key = storage_key.to_string(); - let row_key = row_key.as_str(); let document_identity = RowIdentity::from_user_key(document_id); debug!( core = self.core_id, @@ -111,8 +110,13 @@ impl CoreLoop { // and schemaless collections alike (see `dml::convert_insert`). let bitemporal = self.is_bitemporal(database_id, tid, collection); let exists_result = if bitemporal { - self.sparse - .versioned_exists_current_in_txn(&txn, database_id, tid, collection, row_key) + self.sparse.versioned_exists_current_in_txn( + &txn, + database_id, + tid, + collection, + &storage_key, + ) } else { self.sparse .exists_in_txn(&txn, database_id, tid, collection, &storage_key) @@ -168,7 +172,7 @@ impl CoreLoop { database_id: task.request.database_id.as_u64(), tid, collection, - document_id: row_key, + document_id: &row_key, surrogate, value: effective_value, index_text: true, diff --git a/nodedb/src/data/executor/handlers/point/update/exec.rs b/nodedb/src/data/executor/handlers/point/update/exec.rs index 8ec3dbc0a..3e7930045 100644 --- a/nodedb/src/data/executor/handlers/point/update/exec.rs +++ b/nodedb/src/data/executor/handlers/point/update/exec.rs @@ -138,7 +138,7 @@ impl CoreLoop { let database_id = task.request.database_id.as_u64(); let get_result = if bitemporal { self.sparse - .versioned_get_current(database_id, tid, collection, row_key) + .versioned_get_current(database_id, tid, collection, &storage_key) } else { self.sparse.get(database_id, tid, collection, &storage_key) }; diff --git a/nodedb/src/data/executor/handlers/point/update/persist.rs b/nodedb/src/data/executor/handlers/point/update/persist.rs index fbaec126d..7aabce0cb 100644 --- a/nodedb/src/data/executor/handlers/point/update/persist.rs +++ b/nodedb/src/data/executor/handlers/point/update/persist.rs @@ -34,10 +34,9 @@ pub(in crate::data::executor) struct PointUpdatePersist<'a> { pub(in crate::data::executor) database_id: u64, pub(in crate::data::executor) tid: u64, pub(in crate::data::executor) collection: &'a str, - /// Storage key (the surrogate hex). + /// Storage key (the surrogate hex), rendered — used in error messages. pub(in crate::data::executor) row_key: &'a str, - /// The same storage key, typed — passed alongside `row_key` because the - /// versioned-table methods below still take the rendered text. + /// The same storage key, typed — what every storage call below takes. pub(in crate::data::executor) storage_key: &'a crate::engine::document::store::StorageKey, /// The row as it was before this update — the old side of the index diff, /// and the pre-image every folded constraint subtracts. @@ -129,7 +128,7 @@ impl CoreLoop { database_id, tid, collection, - doc_id: row_key, + doc_id: storage_key, sys_from_ms, valid_from_ms: i64::MIN, valid_until_ms: i64::MAX, @@ -154,7 +153,7 @@ impl CoreLoop { database_id, tenant: tid, coll: collection, - doc_id: row_key, + doc_id: storage_key, sys_from_ms, valid_from_ms: i64::MIN, valid_until_ms: i64::MAX, diff --git a/nodedb/src/data/executor/handlers/point/update_reindex.rs b/nodedb/src/data/executor/handlers/point/update_reindex.rs index 677cb7999..1a27ee3e4 100644 --- a/nodedb/src/data/executor/handlers/point/update_reindex.rs +++ b/nodedb/src/data/executor/handlers/point/update_reindex.rs @@ -25,7 +25,7 @@ pub(in crate::data::executor) struct BitemporalUpdateReindex<'a> { pub database_id: u64, pub tid: u64, pub collection: &'a str, - pub doc_id: &'a str, + pub doc_id: &'a crate::engine::document::store::StorageKey, pub sys_from_ms: i64, pub valid_from_ms: i64, pub valid_until_ms: i64, @@ -117,6 +117,8 @@ impl CoreLoop { // to `note_index_write_values` after the caller's commit without a // borrow conflict. let mut touched_values: Vec<(String, String)> = Vec::new(); + // INDEXES_VERSIONED still keys on the storage key as text. + let doc_id_str = p.doc_id.to_string(); for path in p.index_paths { let new_values = Self::indexed_values_for_path(p.new_doc, path); @@ -136,7 +138,7 @@ impl CoreLoop { coll: p.collection, field: &path.path, value, - doc_id: p.doc_id, + doc_id: &doc_id_str, sys_from_ms: p.sys_from_ms, }, )?; @@ -152,7 +154,7 @@ impl CoreLoop { coll: p.collection, field: &path.path, value, - doc_id: p.doc_id, + doc_id: &doc_id_str, sys_from_ms: p.sys_from_ms, }, )?; diff --git a/nodedb/src/data/executor/handlers/transaction/batch.rs b/nodedb/src/data/executor/handlers/transaction/batch.rs index 34f58df62..8990d3a87 100644 --- a/nodedb/src/data/executor/handlers/transaction/batch.rs +++ b/nodedb/src/data/executor/handlers/transaction/batch.rs @@ -452,7 +452,7 @@ impl CoreLoop { } else { crate::event::WriteOp::Insert }, - identity: crate::engine::document::store::identity_of(&document_id), + identity: document_id.to_identity(), new_value: None, old_value, }), @@ -464,7 +464,7 @@ impl CoreLoop { } => Some(DeferredWrite { collection, op: crate::event::WriteOp::Delete, - identity: crate::engine::document::store::identity_of(&document_id), + identity: document_id.to_identity(), new_value: None, old_value: Some(old_value), }), diff --git a/nodedb/src/data/executor/handlers/transaction/index_write_values.rs b/nodedb/src/data/executor/handlers/transaction/index_write_values.rs index 233c7be4d..4f0ccdf66 100644 --- a/nodedb/src/data/executor/handlers/transaction/index_write_values.rs +++ b/nodedb/src/data/executor/handlers/transaction/index_write_values.rs @@ -207,8 +207,7 @@ mod tests { fn put_entry(collection: &str, field: &str, value: &str) -> UndoEntry { UndoEntry::PutDocument { collection: collection.to_string(), - document_id: "doc".to_string(), - surrogate: Surrogate::new(1), + document_id: nodedb_types::StorageKey::for_surrogate(Surrogate::new(1)), old_value: None, bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), @@ -221,8 +220,7 @@ mod tests { fn delete_entry(collection: &str, field: &str, value: &str) -> UndoEntry { UndoEntry::DeleteDocument { collection: collection.to_string(), - document_id: "doc".to_string(), - surrogate: Surrogate::new(2), + document_id: nodedb_types::StorageKey::for_surrogate(Surrogate::new(2)), old_value: Vec::new(), bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/constraint.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/constraint.rs index 7e61bbd35..d1c7c56e3 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/constraint.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/constraint.rs @@ -31,7 +31,6 @@ impl CoreLoop { pub(super) fn stage_pk_present( &self, ctx: &StageCtx<'_>, - row_key: &str, storage_key: &StorageKey, bitemporal: bool, overlay: OverlayPk, @@ -51,7 +50,7 @@ impl CoreLoop { ctx.database_id, ctx.tid, ctx.collection, - row_key, + storage_key, )? } else { self.sparse.exists_in_txn( diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_point_document.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_point_document.rs index 0ba14e5ac..7a0c062a0 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_point_document.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_point_document.rs @@ -36,17 +36,10 @@ impl CoreLoop { if_absent: bool, ) -> Response { let storage_key = StorageKey::for_surrogate(ctx.surrogate); - let row_key = storage_key.to_string(); let bitemporal = self.is_bitemporal(ctx.database_id, ctx.tid, ctx.collection); let overlay_pk = self.stage_overlay_pk(ctx); - let present = match self.stage_pk_present( - ctx, - row_key.as_str(), - &storage_key, - bitemporal, - overlay_pk, - ) { + let present = match self.stage_pk_present(ctx, &storage_key, bitemporal, overlay_pk) { Ok(p) => p, Err(e) => return self.response_error(ctx.task, e), }; @@ -98,16 +91,9 @@ impl CoreLoop { // and an earlier statement in this transaction may already have // tombstoned it. let storage_key = StorageKey::for_surrogate(ctx.surrogate); - let row_key = storage_key.to_string(); let bitemporal = self.is_bitemporal(ctx.database_id, ctx.tid, ctx.collection); let overlay_pk = self.stage_overlay_pk(ctx); - let present = match self.stage_pk_present( - ctx, - row_key.as_str(), - &storage_key, - bitemporal, - overlay_pk, - ) { + let present = match self.stage_pk_present(ctx, &storage_key, bitemporal, overlay_pk) { Ok(p) => p, Err(e) => return self.response_error(ctx.task, e), }; @@ -161,7 +147,6 @@ impl CoreLoop { ctx.collection.to_string(), ); let storage_key = StorageKey::for_surrogate(ctx.surrogate); - let row_key = storage_key.to_string(); // Reject direct updates to generated columns (matches the durable path). if let Some(config) = self.doc_configs.get(&config_key) @@ -188,7 +173,7 @@ impl CoreLoop { ctx.database_id, ctx.tid, ctx.collection, - row_key.as_str(), + &storage_key, ) } else { self.sparse diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs index e2ee76496..8fb1ec83a 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs @@ -22,7 +22,6 @@ use crate::data::executor::handlers::transaction::overlay::Staged; use crate::data::executor::handlers::upsert::{apply_on_conflict_updates, merge_values}; use crate::data::executor::response_codec; use crate::data::executor::strict_format; -use crate::engine::document::store::surrogate_to_doc_id; use crate::types::TenantId; impl CoreLoop { @@ -102,16 +101,15 @@ impl CoreLoop { Some(Staged::Tombstone) => Ok(None), None => { let bitemporal = self.is_bitemporal(ctx.database_id, ctx.tid, ctx.collection); + let storage_key = nodedb_types::StorageKey::for_surrogate(ctx.surrogate); if bitemporal { - let row_key = surrogate_to_doc_id(ctx.surrogate); self.sparse.versioned_get_current( ctx.database_id, ctx.tid, ctx.collection, - row_key.as_str(), + &storage_key, ) } else { - let storage_key = nodedb_types::StorageKey::for_surrogate(ctx.surrogate); self.sparse .get(ctx.database_id, ctx.tid, ctx.collection, &storage_key) } diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs index 79d4601d0..5c86558e5 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs @@ -39,8 +39,6 @@ impl CoreLoop { user_roles, resolved_sum_targets, } = p; - let row_key = crate::engine::document::store::surrogate_to_doc_id(surrogate); - let row_key = row_key.as_str(); let database_id = dummy_task.request.database_id.as_u64(); let hook_ctx = HookCtx { database_id, @@ -132,8 +130,7 @@ impl CoreLoop { for target in target_writes { undo_log.push(UndoEntry::PutDocument { collection: target.collection, - document_id: target.document_id, - surrogate: target.surrogate, + document_id: nodedb_types::StorageKey::for_surrogate(target.surrogate), old_value: target.outcome.prior_value, bitemporal_sys_from_ms: target.outcome.bitemporal_sys_from_ms, bitemporal_index_tuples: target.outcome.bitemporal_index_tuples, @@ -163,8 +160,7 @@ impl CoreLoop { if let Some(old) = outcome.prior_value { undo_log.push(UndoEntry::DeleteDocument { collection: collection.to_string(), - document_id: row_key.to_string(), - surrogate, + document_id: nodedb_types::StorageKey::for_surrogate(surrogate), old_value: old, bitemporal_sys_from_ms: outcome.bitemporal_sys_from_ms, bitemporal_index_tuples: outcome.bitemporal_index_tuples, diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs index 5339951ea..30ec6213e 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs @@ -135,7 +135,7 @@ impl CoreLoop { database_id, tid, collection, - row_key, + &storage_key, ) } else { self.sparse @@ -294,8 +294,7 @@ impl CoreLoop { for target in target_writes { undo_log.push(UndoEntry::PutDocument { collection: target.collection, - document_id: target.document_id, - surrogate: target.surrogate, + document_id: nodedb_types::StorageKey::for_surrogate(target.surrogate), old_value: target.outcome.prior_value, bitemporal_sys_from_ms: target.outcome.bitemporal_sys_from_ms, bitemporal_index_tuples: target.outcome.bitemporal_index_tuples, @@ -322,8 +321,7 @@ impl CoreLoop { undo_log.push(UndoEntry::PutDocument { collection: collection.to_string(), - document_id: row_key.to_string(), - surrogate, + document_id: storage_key, old_value: outcome.prior_value, bitemporal_sys_from_ms: outcome.bitemporal_sys_from_ms, bitemporal_index_tuples: outcome.bitemporal_index_tuples, diff --git a/nodedb/src/data/executor/handlers/transaction/undo/document.rs b/nodedb/src/data/executor/handlers/transaction/undo/document.rs index ae5ee632e..49f8b0052 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/document.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/document.rs @@ -6,6 +6,7 @@ //! return `Err((entry_index, detail))` on fatal failure so the caller can //! escalate to a typed `RollbackFailed` response. +use nodedb_types::StorageKey; use tracing::error; use crate::data::executor::core_loop::CoreLoop; @@ -19,7 +20,7 @@ pub(super) struct UndoDocumentContext<'a> { pub tid: u64, pub entry_index: usize, pub collection: &'a str, - pub document_id: &'a str, + pub document_id: &'a StorageKey, } impl CoreLoop { @@ -36,7 +37,6 @@ impl CoreLoop { UndoEntry::PutDocument { collection, document_id, - surrogate, old_value, bitemporal_sys_from_ms, bitemporal_index_tuples, @@ -51,11 +51,8 @@ impl CoreLoop { collection: &collection, document_id: &document_id, }; - // The entry carries the surrogate directly, so the storage - // key is minted from it rather than re-parsed out of - // `document_id`'s text. - let storage_key = - crate::engine::document::store::StorageKey::for_surrogate(surrogate); + let storage_key = document_id; + let surrogate = document_id.surrogate(); if let Some(sys_from_ms) = bitemporal_sys_from_ms { // Bitemporal op: never wrote the non-versioned table, so // physically remove the appended version row (+ its index @@ -94,12 +91,7 @@ impl CoreLoop { // restore the stale entries this put removed. Empty on the // bitemporal path (its index reversal happened in // `undo_bitemporal_write` above), so this is a no-op there. - self.undo_secondary_index( - ctx, - &storage_key, - &secondary_index_added, - &secondary_index_removed, - )?; + self.undo_secondary_index(ctx, &secondary_index_added, &secondary_index_removed)?; // Revert inverted index: remove the postings this rolled-back // put wrote. FATAL on failure — a rollback that leaves stale FTS // postings behind is the same silent-partial-success corruption @@ -136,7 +128,6 @@ impl CoreLoop { UndoEntry::DeleteDocument { collection, document_id, - surrogate, old_value, bitemporal_sys_from_ms, bitemporal_index_tuples, @@ -150,11 +141,8 @@ impl CoreLoop { collection: &collection, document_id: &document_id, }; - // The entry carries the surrogate directly, so the storage - // key is minted from it rather than re-parsed out of - // `document_id`'s text. - let storage_key = - crate::engine::document::store::StorageKey::for_surrogate(surrogate); + let storage_key = document_id; + let surrogate = document_id.surrogate(); if let Some(sys_from_ms) = bitemporal_sys_from_ms { self.undo_bitemporal_write(ctx, sys_from_ms, &bitemporal_index_tuples)?; } else { @@ -179,7 +167,7 @@ impl CoreLoop { // Restore the plain secondary-index entries the forward delete // cascade removed. Empty on the bitemporal path (no plain // INDEXES entries there), so this is a no-op for it. - self.undo_secondary_index(ctx, &storage_key, &[], &secondary_index_tuples)?; + self.undo_secondary_index(ctx, &[], &secondary_index_tuples)?; // Re-index the restored document into the full-text inverted // index. The forward delete cascade removed its postings // unconditionally, so a rollback that restored the row but not @@ -238,6 +226,8 @@ impl CoreLoop { self.sparse .versioned_remove_in_txn(&txn, database_id, tid, collection, document_id, sys_from_ms) .map_err(|e| map_err("version remove", e.to_string()))?; + // INDEXES_VERSIONED still keys on the storage key as text. + let doc_id_str = document_id.to_string(); for (field, value) in index_tuples { self.sparse .versioned_index_remove_in_txn( @@ -248,7 +238,7 @@ impl CoreLoop { coll: collection, field, value, - doc_id: document_id, + doc_id: &doc_id_str, sys_from_ms, }, ) @@ -269,7 +259,6 @@ impl CoreLoop { fn undo_secondary_index( &self, ctx: UndoDocumentContext<'_>, - storage_key: &crate::engine::document::store::StorageKey, to_remove: &[(String, String)], to_restore: &[(String, String)], ) -> Result<(), (usize, String)> { @@ -296,12 +285,12 @@ impl CoreLoop { }; for (field, value) in to_remove { self.sparse - .index_remove(database_id, tid, collection, field, value, storage_key) + .index_remove(database_id, tid, collection, field, value, document_id) .map_err(|e| map_err("remove", e.to_string()))?; } for (field, value) in to_restore { self.sparse - .index_put(database_id, tid, collection, field, value, storage_key) + .index_put(database_id, tid, collection, field, value, document_id) .map_err(|e| map_err("restore", e.to_string()))?; } Ok(()) @@ -370,9 +359,13 @@ mod tests { const DB: u64 = 0; const TID: u64 = 1; + fn storage_key(surrogate: u32) -> StorageKey { + StorageKey::for_surrogate(nodedb_types::Surrogate::new(surrogate)) + } + fn seed_version( core: &crate::data::executor::core_loop::CoreLoop, - doc: &str, + doc: u32, t: i64, body: &[u8], ) { @@ -381,7 +374,7 @@ mod tests { database_id: DB, tenant: TID, coll: "c", - doc_id: doc, + doc_id: &storage_key(doc), sys_from_ms: t, valid_from_ms: 0, valid_until_ms: i64::MAX, @@ -390,7 +383,7 @@ mod tests { .unwrap(); } - fn seed_index(core: &crate::data::executor::core_loop::CoreLoop, doc: &str, t: i64) { + fn seed_index(core: &crate::data::executor::core_loop::CoreLoop, doc: u32, t: i64) { core.sparse .versioned_index_put(VersionedIndexEntry { database_id: DB, @@ -398,7 +391,7 @@ mod tests { coll: "c", field: "status", value: "active", - doc_id: doc, + doc_id: &storage_key(doc).to_string(), sys_from_ms: t, }) .unwrap(); @@ -416,21 +409,21 @@ mod tests { let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); let t = 1_000; - seed_version(&core, "d1", t, b"v1"); - seed_index(&core, "d1", t); + let d1 = storage_key(1); + seed_version(&core, 1, t, b"v1"); + seed_index(&core, 1, t); assert!( core.sparse - .versioned_get_current(DB, TID, "c", "d1") + .versioned_get_current(DB, TID, "c", &d1) .unwrap() .is_some() ); - assert_eq!(index_lookup(&core), vec!["d1".to_string()]); + assert_eq!(index_lookup(&core), vec![d1.to_string()]); let entry = UndoEntry::PutDocument { collection: "c".into(), - document_id: "d1".into(), - surrogate: nodedb_types::Surrogate::ZERO, + document_id: d1, old_value: None, bitemporal_sys_from_ms: Some(t), bitemporal_index_tuples: vec![("status".into(), "active".into())], @@ -442,7 +435,7 @@ mod tests { assert!( core.sparse - .versioned_get_current(DB, TID, "c", "d1") + .versioned_get_current(DB, TID, "c", &d1) .unwrap() .is_none(), "version row must be physically gone" @@ -456,10 +449,11 @@ mod tests { let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); // Live version at T1, then a tombstone at T2 (plus an index tombstone). - seed_version(&core, "d1", 1_000, b"v1"); - seed_index(&core, "d1", 1_000); + let d1 = storage_key(1); + seed_version(&core, 1, 1_000, b"v1"); + seed_index(&core, 1, 1_000); core.sparse - .versioned_tombstone(DB, TID, "c", "d1", 2_000) + .versioned_tombstone(DB, TID, "c", &d1, 2_000) .unwrap(); core.sparse .versioned_index_tombstone(VersionedIndexEntry { @@ -468,7 +462,7 @@ mod tests { coll: "c", field: "status", value: "active", - doc_id: "d1", + doc_id: &d1.to_string(), sys_from_ms: 2_000, }) .unwrap(); @@ -476,15 +470,14 @@ mod tests { // Tombstone hides the row. assert!( core.sparse - .versioned_get_current(DB, TID, "c", "d1") + .versioned_get_current(DB, TID, "c", &d1) .unwrap() .is_none() ); let entry = UndoEntry::DeleteDocument { collection: "c".into(), - document_id: "d1".into(), - surrogate: nodedb_types::Surrogate::ZERO, + document_id: d1, old_value: b"v1".to_vec(), bitemporal_sys_from_ms: Some(2_000), bitemporal_index_tuples: vec![("status".into(), "active".into())], @@ -496,11 +489,11 @@ mod tests { // Removing the tombstone restores the prior live version as current. assert_eq!( core.sparse - .versioned_get_current(DB, TID, "c", "d1") + .versioned_get_current(DB, TID, "c", &d1) .unwrap(), Some(b"v1".to_vec()) ); - assert_eq!(index_lookup(&core), vec!["d1".to_string()]); + assert_eq!(index_lookup(&core), vec![d1.to_string()]); } #[test] @@ -519,8 +512,7 @@ mod tests { core.chain_hashes.insert(key(), "h1".into()); let restore = UndoEntry::PutDocument { collection: "c".into(), - document_id: "nonexistent".into(), - surrogate: nodedb_types::Surrogate::ZERO, + document_id: storage_key(0), old_value: None, bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), @@ -537,8 +529,7 @@ mod tests { // Genesis case: undo removes the key entirely. let genesis = UndoEntry::PutDocument { collection: "c".into(), - document_id: "nonexistent".into(), - surrogate: nodedb_types::Surrogate::ZERO, + document_id: storage_key(0), old_value: None, bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), @@ -550,26 +541,19 @@ mod tests { assert!(!core.chain_hashes.contains_key(&key())); } - fn storage_key(surrogate: u32) -> crate::engine::document::store::StorageKey { - crate::engine::document::store::StorageKey::for_surrogate(nodedb_types::Surrogate::new( - surrogate, - )) - } - #[test] fn plain_put_undo_restores_or_removes_the_surrogate_row() { let dir = tempfile::tempdir().unwrap(); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); // Overwrite case: current holds "new", undo restores "old". `undo` - // reads the row through the entry's `surrogate`, not `document_id`'s - // text, so the seeded row and the entry share one surrogate. + // reads the row through the entry's storage key, so the seeded row + // and the entry share one surrogate. let key1 = storage_key(1); core.sparse.put(DB, TID, "c", &key1, b"new").unwrap(); let overwrite = UndoEntry::PutDocument { collection: "c".into(), - document_id: key1.to_string(), - surrogate: key1.surrogate(), + document_id: key1, old_value: Some(b"old".to_vec()), bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), @@ -588,8 +572,7 @@ mod tests { core.sparse.put(DB, TID, "c", &key2, b"inserted").unwrap(); let insert = UndoEntry::PutDocument { collection: "c".into(), - document_id: key2.to_string(), - surrogate: key2.surrogate(), + document_id: key2, old_value: None, bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), @@ -610,8 +593,7 @@ mod tests { let key1 = storage_key(1); let entry = UndoEntry::DeleteDocument { collection: "c".into(), - document_id: key1.to_string(), - surrogate: key1.surrogate(), + document_id: key1, old_value: b"prior".to_vec(), bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), diff --git a/nodedb/src/data/executor/handlers/transaction/undo/entry.rs b/nodedb/src/data/executor/handlers/transaction/undo/entry.rs index 34572968d..64d30ae38 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/entry.rs @@ -34,10 +34,9 @@ pub(in crate::data::executor) enum UndoEntry { /// Undo a PointPut by deleting the document (or restoring the old value). PutDocument { collection: String, - /// Hex-encoded surrogate (the redb storage key). - document_id: String, - /// Numeric surrogate for FTS index rollback. - surrogate: nodedb_types::Surrogate, + /// The redb storage key. `.surrogate()` recovers the numeric surrogate + /// FTS index rollback needs. + document_id: nodedb_types::StorageKey, /// `None` if the document didn't exist before (inserted); `Some(bytes)` /// if it was overwritten (updated). old_value: Option>, @@ -64,12 +63,11 @@ pub(in crate::data::executor) enum UndoEntry { /// Undo a PointDelete by re-inserting the document. DeleteDocument { collection: String, - /// Hex-encoded surrogate (the redb storage key). - document_id: String, - /// Numeric surrogate for FTS inverted-index rollback re-indexing. The - /// forward delete cascade removed this document's postings; a rolled-back - /// delete recomputes and re-inserts them under this surrogate. - surrogate: nodedb_types::Surrogate, + /// The redb storage key. `.surrogate()` recovers the numeric surrogate + /// the FTS inverted-index rollback re-indexes under: the forward + /// delete cascade removed this document's postings, and a + /// rolled-back delete recomputes and re-inserts them under it. + document_id: nodedb_types::StorageKey, old_value: Vec, /// System-time key of the versioned tombstone row this op appended on a /// bitemporal collection. `None` = plain op → re-insert via the diff --git a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs index 43a3d4591..82ac9345e 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs @@ -118,13 +118,17 @@ mod tests { const DB: u64 = 0; const TID: u64 = 1; - fn seed_version(core: &CoreLoop, doc: &str, t: i64, body: &[u8]) { + fn key(surrogate: u32) -> nodedb_types::StorageKey { + nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)) + } + + fn seed_version(core: &CoreLoop, doc: u32, t: i64, body: &[u8]) { core.sparse .versioned_put(VersionedPut { database_id: DB, tenant: TID, coll: "c", - doc_id: doc, + doc_id: &key(doc), sys_from_ms: t, valid_from_ms: 0, valid_until_ms: i64::MAX, @@ -133,7 +137,7 @@ mod tests { .unwrap(); } - fn seed_index(core: &CoreLoop, doc: &str, t: i64) { + fn seed_index(core: &CoreLoop, doc: u32, t: i64) { core.sparse .versioned_index_put(VersionedIndexEntry { database_id: DB, @@ -141,7 +145,7 @@ mod tests { coll: "c", field: "status", value: "active", - doc_id: doc, + doc_id: &key(doc).to_string(), sys_from_ms: t, }) .unwrap(); @@ -163,20 +167,21 @@ mod tests { fn rollback_undo_log_restores_pre_txn_state_for_bitemporal_put_then_delete() { let dir = tempfile::tempdir().unwrap(); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let d1 = key(1); - // Pre-txn state: nothing exists for "d1". + // Pre-txn state: nothing exists for `d1`. assert!( core.sparse - .versioned_get_current(DB, TID, "c", "d1") + .versioned_get_current(DB, TID, "c", &d1) .unwrap() .is_none() ); // Forward tx: PUT at t=1000, then DELETE (tombstone) at t=2000. - seed_version(&core, "d1", 1_000, b"v1"); - seed_index(&core, "d1", 1_000); + seed_version(&core, 1, 1_000, b"v1"); + seed_index(&core, 1, 1_000); core.sparse - .versioned_tombstone(DB, TID, "c", "d1", 2_000) + .versioned_tombstone(DB, TID, "c", &d1, 2_000) .unwrap(); core.sparse .versioned_index_tombstone(VersionedIndexEntry { @@ -185,7 +190,7 @@ mod tests { coll: "c", field: "status", value: "active", - doc_id: "d1", + doc_id: &d1.to_string(), sys_from_ms: 2_000, }) .unwrap(); @@ -193,7 +198,7 @@ mod tests { // Sanity: the forward tx did delete the row (as observed mid-tx). assert!( core.sparse - .versioned_get_current(DB, TID, "c", "d1") + .versioned_get_current(DB, TID, "c", &d1) .unwrap() .is_none() ); @@ -201,8 +206,7 @@ mod tests { let undo_log = vec![ UndoEntry::PutDocument { collection: "c".into(), - document_id: "d1".into(), - surrogate: nodedb_types::Surrogate::ZERO, + document_id: d1, old_value: None, bitemporal_sys_from_ms: Some(1_000), bitemporal_index_tuples: vec![("status".into(), "active".into())], @@ -212,8 +216,7 @@ mod tests { }, UndoEntry::DeleteDocument { collection: "c".into(), - document_id: "d1".into(), - surrogate: nodedb_types::Surrogate::ZERO, + document_id: d1, old_value: b"v1".to_vec(), bitemporal_sys_from_ms: Some(2_000), bitemporal_index_tuples: vec![("status".into(), "active".into())], @@ -230,7 +233,7 @@ mod tests { // Pre-txn state restored: no current version, no index entry. assert!( core.sparse - .versioned_get_current(DB, TID, "c", "d1") + .versioned_get_current(DB, TID, "c", &d1) .unwrap() .is_none(), "aborted bitemporal put+delete must leave no current version behind" diff --git a/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs b/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs index 00a51a679..16ff2f867 100644 --- a/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs +++ b/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs @@ -114,11 +114,11 @@ impl CoreLoop { // per branch. Gates the live HNSW re-index + the post-apply redo // write-set below; a non-vector collection pays neither. let has_vectors = self.collection_has_vectors(database_id, tid, collection); + let key = nodedb_types::StorageKey::for_surrogate(surrogate); let existing = if bitemporal { self.sparse - .versioned_get_current(database_id, tid, collection, row_key) + .versioned_get_current(database_id, tid, collection, &key) } else { - let key = nodedb_types::StorageKey::for_surrogate(surrogate); self.sparse.get(database_id, tid, collection, &key) }; diff --git a/nodedb/src/data/executor/scan_versioned.rs b/nodedb/src/data/executor/scan_versioned.rs index ab0a3be85..7579e9635 100644 --- a/nodedb/src/data/executor/scan_versioned.rs +++ b/nodedb/src/data/executor/scan_versioned.rs @@ -34,7 +34,7 @@ impl CoreLoop { valid_at_ms: None, limit, }, - &|_, _| true, + &|_: &nodedb_types::StorageKey, _: &[u8]| true, // No task in scope: this helper serves callers that supply their // own bound (an explicit `limit`), so no deadline cuts it short. &crate::engine::sparse::scan_stop::never_stop, @@ -46,16 +46,7 @@ impl CoreLoop { ); let mut normalized = Vec::with_capacity(docs.len()); - for (id, raw) in docs { - // The versioned table keys every row by the same surrogate hex - // as the plain table; a shape that fails to parse is a violated - // storage invariant, not a legacy row to skip. - let key = nodedb_types::StorageKey::parse(&id).ok_or_else(|| crate::Error::Storage { - engine: "sparse".into(), - detail: format!( - "collection '{collection}' has a versioned row whose key is not a valid storage key: '{id}'" - ), - })?; + for (key, raw) in docs { normalized.push(sparse_row_to_doc(&key, &raw, format.as_format_ref())); } Ok(normalized) diff --git a/nodedb/src/engine/document/store/engine/batch.rs b/nodedb/src/engine/document/store/engine/batch.rs index 050ea7624..c9cc053a4 100644 --- a/nodedb/src/engine/document/store/engine/batch.rs +++ b/nodedb/src/engine/document/store/engine/batch.rs @@ -90,7 +90,7 @@ impl<'a> DocumentEngine<'a> { .map(|id| { StorageKey::parse(&id).ok_or_else(|| { crate::engine::sparse::btree::invalid_storage_key_err( - "INDEXES_VERSIONED", + crate::engine::sparse::btree::KeyedTable::IndexesVersioned, collection, &id, ) @@ -121,7 +121,9 @@ impl<'a> DocumentEngine<'a> { if key.starts_with(&expected_prefix) { let doc_id = StorageKey::parse(doc_id).ok_or_else(|| { crate::engine::sparse::btree::invalid_storage_key_err( - "INDEXES", collection, doc_id, + crate::engine::sparse::btree::KeyedTable::Indexes, + collection, + doc_id, ) })?; doc_ids.push(doc_id); diff --git a/nodedb/src/engine/document/store/engine/delete.rs b/nodedb/src/engine/document/store/engine/delete.rs index c84c5b76f..b987e372e 100644 --- a/nodedb/src/engine/document/store/engine/delete.rs +++ b/nodedb/src/engine/document/store/engine/delete.rs @@ -12,15 +12,12 @@ use crate::engine::document::store::extract::extract_index_values_rmpv; impl<'a> DocumentEngine<'a> { pub fn delete(&self, collection: &str, doc_id: &StorageKey) -> crate::Result { - // The versioned table and the INDEXES table both take the storage - // key as text; rendered once here at the boundary. - let doc_id_str = doc_id.to_string(); if self.is_bitemporal(collection) { let prior_body = self.sparse.versioned_get_current( self.database_id, self.tenant_id, collection, - &doc_id_str, + doc_id, )?; let Some(body) = prior_body else { return Ok(false); @@ -30,12 +27,14 @@ impl<'a> DocumentEngine<'a> { self.database_id, self.tenant_id, collection, - &doc_id_str, + doc_id, sys_from, )?; if let Some(config) = self.configs.get(collection) && let Ok(rmpv_val) = crate::util::bounded_msgpack::read_value(&body) { + // INDEXES_VERSIONED still keys on the storage key as text. + let doc_id_str = doc_id.to_string(); for index_path in &config.index_paths { for v in extract_index_values_rmpv(&rmpv_val, &index_path.path, index_path.is_array) diff --git a/nodedb/src/engine/document/store/engine/get.rs b/nodedb/src/engine/document/store/engine/get.rs index 0ad882227..1f7c641ad 100644 --- a/nodedb/src/engine/document/store/engine/get.rs +++ b/nodedb/src/engine/document/store/engine/get.rs @@ -18,7 +18,7 @@ impl<'a> DocumentEngine<'a> { self.database_id, self.tenant_id, collection, - &doc_id.to_string(), + doc_id, )? } else { self.sparse @@ -41,12 +41,8 @@ impl<'a> DocumentEngine<'a> { /// Get raw MessagePack bytes (zero-copy path for DataFusion UDFs). pub fn get_raw(&self, collection: &str, doc_id: &StorageKey) -> crate::Result>> { if self.is_bitemporal(collection) { - self.sparse.versioned_get_current( - self.database_id, - self.tenant_id, - collection, - &doc_id.to_string(), - ) + self.sparse + .versioned_get_current(self.database_id, self.tenant_id, collection, doc_id) } else { self.sparse .get(self.database_id, self.tenant_id, collection, doc_id) diff --git a/nodedb/src/engine/document/store/engine/put.rs b/nodedb/src/engine/document/store/engine/put.rs index 38e02ccf6..987f14c9d 100644 --- a/nodedb/src/engine/document/store/engine/put.rs +++ b/nodedb/src/engine/document/store/engine/put.rs @@ -39,9 +39,6 @@ impl<'a> DocumentEngine<'a> { msgpack_bytes: &[u8], ) -> crate::Result<()> { let bitemporal = self.is_bitemporal(collection); - // The versioned table and the INDEXES table both take the storage - // key as text; rendered once here at the boundary. - let doc_id_str = doc_id.to_string(); if bitemporal { let sys_from = wall_now_ms(); @@ -50,7 +47,7 @@ impl<'a> DocumentEngine<'a> { database_id: self.database_id, tenant: self.tenant_id, coll: collection, - doc_id: &doc_id_str, + doc_id, sys_from_ms: sys_from, valid_from_ms: i64::MIN, valid_until_ms: i64::MAX, @@ -69,6 +66,8 @@ impl<'a> DocumentEngine<'a> { if let Some(config) = self.configs.get(collection) && let Ok(value) = crate::util::bounded_msgpack::read_value(msgpack_bytes) { + // INDEXES_VERSIONED still keys on the storage key as text. + let doc_id_str = doc_id.to_string(); for index_path in &config.index_paths { let values = extract_index_values_rmpv(&value, &index_path.path, index_path.is_array); diff --git a/nodedb/src/engine/sparse/btree/mod.rs b/nodedb/src/engine/sparse/btree/mod.rs index 556c2e46f..75c0e3aae 100644 --- a/nodedb/src/engine/sparse/btree/mod.rs +++ b/nodedb/src/engine/sparse/btree/mod.rs @@ -12,5 +12,5 @@ pub mod tables; pub use engine::SparseEngine; pub(crate) use keys::coll_prefix; pub(in crate::engine::sparse) use keys::{tenant_prefix, with_tenant_key4}; -pub(crate) use tables::{DOCUMENTS, invalid_storage_key_err}; +pub(crate) use tables::{DOCUMENTS, KeyedTable, invalid_storage_key_err}; pub(in crate::engine::sparse) use tables::{INDEXES, redb_err}; diff --git a/nodedb/src/engine/sparse/btree/tables.rs b/nodedb/src/engine/sparse/btree/tables.rs index 98757f290..d777b31e2 100644 --- a/nodedb/src/engine/sparse/btree/tables.rs +++ b/nodedb/src/engine/sparse/btree/tables.rs @@ -22,16 +22,40 @@ pub(in crate::engine::sparse) fn redb_err(ctx: &str, e: E) } } +/// The redb tables whose keys embed a [`nodedb_types::StorageKey`]. +#[derive(Debug, Clone, Copy)] +pub(crate) enum KeyedTable { + Documents, + DocumentsVersioned, + Indexes, + IndexesVersioned, +} + +impl KeyedTable { + fn name(self) -> &'static str { + match self { + Self::Documents => "DOCUMENTS", + Self::DocumentsVersioned => "DOCUMENTS_VERSIONED", + Self::Indexes => "INDEXES", + Self::IndexesVersioned => "INDEXES_VERSIONED", + } + } +} + /// Report a row on `table` whose key does not parse as a [`nodedb_types::StorageKey`]. /// -/// A non-parsing key on any of DOCUMENTS, INDEXES, or INDEXES_VERSIONED is a -/// violated storage invariant, not a legacy row to skip: every stored key is -/// minted by [`StorageKey::for_surrogate`]. -pub(crate) fn invalid_storage_key_err(table: &str, collection: &str, key: &str) -> crate::Error { +/// A non-parsing key is a violated storage invariant, not a row to skip: +/// every stored key is minted by [`StorageKey::for_surrogate`]. +pub(crate) fn invalid_storage_key_err( + table: KeyedTable, + collection: &str, + key: &str, +) -> crate::Error { crate::Error::Storage { engine: "sparse".into(), detail: format!( - "collection '{collection}' has a {table} row whose key is not a valid storage key: '{key}'" + "collection '{collection}' has a {} row whose key is not a valid storage key: '{key}'", + table.name() ), } } diff --git a/nodedb/src/engine/sparse/btree_index.rs b/nodedb/src/engine/sparse/btree_index.rs index 8525f4d6c..e4dce70a9 100644 --- a/nodedb/src/engine/sparse/btree_index.rs +++ b/nodedb/src/engine/sparse/btree_index.rs @@ -10,7 +10,7 @@ use redb::{ReadableDatabase, ReadableTable, WriteTransaction}; use tracing::debug; use super::btree::{ - DOCUMENTS, INDEXES, SparseEngine, coll_prefix, invalid_storage_key_err, redb_err, + DOCUMENTS, INDEXES, KeyedTable, SparseEngine, coll_prefix, invalid_storage_key_err, redb_err, }; /// Identifies a single secondary-index entry for an in-txn mutation. @@ -433,8 +433,9 @@ impl SparseEngine { { let value = &rest[..colon_pos]; let doc_id = &rest[colon_pos + 1..]; - let doc_id = StorageKey::parse(doc_id) - .ok_or_else(|| invalid_storage_key_err("INDEXES", collection, doc_id))?; + let doc_id = StorageKey::parse(doc_id).ok_or_else(|| { + invalid_storage_key_err(KeyedTable::Indexes, collection, doc_id) + })?; results.push((doc_id, value.to_string())); } } diff --git a/nodedb/src/engine/sparse/btree_scan.rs b/nodedb/src/engine/sparse/btree_scan.rs index c79734ed3..2f8ecf287 100644 --- a/nodedb/src/engine/sparse/btree_scan.rs +++ b/nodedb/src/engine/sparse/btree_scan.rs @@ -7,7 +7,8 @@ use redb::{ReadableDatabase, ReadableTable}; use tracing::debug; use super::btree::{ - DOCUMENTS, INDEXES, SparseEngine, coll_prefix, invalid_storage_key_err, redb_err, tenant_prefix, + DOCUMENTS, INDEXES, KeyedTable, SparseEngine, coll_prefix, invalid_storage_key_err, redb_err, + tenant_prefix, }; impl SparseEngine { @@ -43,8 +44,9 @@ impl SparseEngine { let key = entry.0.value(); // Extract document_id from key format "{database_id}:{tenant}:{collection}:{doc_id}" let doc_id = key.strip_prefix(&prefix).unwrap_or(key); - let storage_key = StorageKey::parse(doc_id) - .ok_or_else(|| invalid_storage_key_err("DOCUMENTS", collection, doc_id))?; + let storage_key = StorageKey::parse(doc_id).ok_or_else(|| { + invalid_storage_key_err(KeyedTable::Documents, collection, doc_id) + })?; let value = entry.1.value().to_vec(); results.push((storage_key, value)); } @@ -99,8 +101,9 @@ impl SparseEngine { let key = entry.0.value(); // Extract document_id from key format "{database_id}:{tenant}:{collection}:{doc_id}" let doc_id = key.strip_prefix(&prefix).unwrap_or(key); - let storage_key = StorageKey::parse(doc_id) - .ok_or_else(|| invalid_storage_key_err("DOCUMENTS", collection, doc_id))?; + let storage_key = StorageKey::parse(doc_id).ok_or_else(|| { + invalid_storage_key_err(KeyedTable::Documents, collection, doc_id) + })?; let value = entry.1.value(); f(&storage_key, value)?; count += 1; @@ -151,8 +154,9 @@ impl SparseEngine { let entry = entry.map_err(|e| redb_err("doc entry", e))?; let key = entry.0.value(); let doc_id = key.strip_prefix(&prefix).unwrap_or(key); - let storage_key = StorageKey::parse(doc_id) - .ok_or_else(|| invalid_storage_key_err("DOCUMENTS", collection, doc_id))?; + let storage_key = StorageKey::parse(doc_id).ok_or_else(|| { + invalid_storage_key_err(KeyedTable::Documents, collection, doc_id) + })?; let value = entry.1.value().to_vec(); chunk.push((storage_key, value)); total += 1; @@ -330,8 +334,9 @@ impl SparseEngine { let value_bytes = entry.1.value(); let key = entry.0.value(); let doc_id = key.strip_prefix(&prefix).unwrap_or(key); - let storage_key = StorageKey::parse(doc_id) - .ok_or_else(|| invalid_storage_key_err("DOCUMENTS", collection, doc_id))?; + let storage_key = StorageKey::parse(doc_id).ok_or_else(|| { + invalid_storage_key_err(KeyedTable::Documents, collection, doc_id) + })?; // Evaluate predicate on raw bytes — skip allocation if no match. if !predicate(&storage_key, value_bytes) { diff --git a/nodedb/src/engine/sparse/btree_versioned/doc.rs b/nodedb/src/engine/sparse/btree_versioned/doc.rs index 38e679df3..fc39de290 100644 --- a/nodedb/src/engine/sparse/btree_versioned/doc.rs +++ b/nodedb/src/engine/sparse/btree_versioned/doc.rs @@ -2,6 +2,7 @@ //! Document-level operations on the versioned document table. +use nodedb_types::StorageKey; use redb::{ReadableDatabase, ReadableTable, TableDefinition}; use super::key::{doc_prefix, doc_prefix_end, versioned_doc_key}; @@ -16,7 +17,7 @@ use crate::engine::sparse::btree::{SparseEngine, redb_err}; /// raw envelope sentinels (`i64::MIN` / `i64::MAX` when unbounded) so the /// handler can surface them uniformly across engines. pub struct VersionedRow { - pub doc_id: String, + pub doc_id: StorageKey, pub system_from_ms: i64, pub valid_from_ms: i64, pub valid_until_ms: i64, @@ -49,7 +50,7 @@ impl SparseEngine { /// Append one version. Always creates a new key; never overwrites an /// earlier version at the same `sys_from_ms`. pub fn versioned_put(&self, p: VersionedPut<'_>) -> crate::Result<()> { - let key = versioned_doc_key(p.database_id, p.tenant, p.coll, p.doc_id, p.sys_from_ms)?; + let key = versioned_doc_key(p.database_id, p.tenant, p.coll, p.doc_id, p.sys_from_ms); let val = encode_value(TAG_LIVE, p.valid_from_ms, p.valid_until_ms, p.body); let txn = self .db @@ -76,7 +77,7 @@ impl SparseEngine { txn: &redb::WriteTransaction, p: VersionedPut<'_>, ) -> crate::Result<()> { - let key = versioned_doc_key(p.database_id, p.tenant, p.coll, p.doc_id, p.sys_from_ms)?; + let key = versioned_doc_key(p.database_id, p.tenant, p.coll, p.doc_id, p.sys_from_ms); let val = encode_value(TAG_LIVE, p.valid_from_ms, p.valid_until_ms, p.body); let mut t = txn .open_table(DOCUMENTS_VERSIONED) @@ -92,10 +93,10 @@ impl SparseEngine { database_id: u64, tenant: u64, coll: &str, - doc_id: &str, + doc_id: &StorageKey, sys_from_ms: i64, ) -> crate::Result<()> { - let key = versioned_doc_key(database_id, tenant, coll, doc_id, sys_from_ms)?; + let key = versioned_doc_key(database_id, tenant, coll, doc_id, sys_from_ms); let val = encode_value(TAG_TOMBSTONE, 0, 0, &[]); let txn = self .db @@ -119,10 +120,10 @@ impl SparseEngine { database_id: u64, tenant: u64, coll: &str, - doc_id: &str, + doc_id: &StorageKey, sys_from_ms: i64, ) -> crate::Result<()> { - let key = versioned_doc_key(database_id, tenant, coll, doc_id, sys_from_ms)?; + let key = versioned_doc_key(database_id, tenant, coll, doc_id, sys_from_ms); let val = encode_value(TAG_TOMBSTONE, 0, 0, &[]); let mut t = txn .open_table(DOCUMENTS_VERSIONED) @@ -144,10 +145,10 @@ impl SparseEngine { database_id: u64, tenant: u64, coll: &str, - doc_id: &str, + doc_id: &StorageKey, sys_from_ms: i64, ) -> crate::Result<()> { - let key = versioned_doc_key(database_id, tenant, coll, doc_id, sys_from_ms)?; + let key = versioned_doc_key(database_id, tenant, coll, doc_id, sys_from_ms); let mut t = txn .open_table(DOCUMENTS_VERSIONED) .map_err(|e| redb_err("open table", e))?; @@ -166,7 +167,7 @@ impl SparseEngine { database_id: u64, tenant: u64, coll: &str, - doc_id: &str, + doc_id: &StorageKey, ) -> crate::Result { let lo = doc_prefix(database_id, tenant, coll, doc_id); let hi = doc_prefix_end(database_id, tenant, coll, doc_id); @@ -196,7 +197,7 @@ impl SparseEngine { database_id: u64, tenant: u64, coll: &str, - doc_id: &str, + doc_id: &StorageKey, ) -> crate::Result { let lo = doc_prefix(database_id, tenant, coll, doc_id); let hi = doc_prefix_end(database_id, tenant, coll, doc_id); @@ -233,13 +234,13 @@ impl SparseEngine { database_id: u64, tenant: u64, coll: &str, - doc_id: &str, + doc_id: &StorageKey, sys_cutoff_ms: Option, valid_at_ms: Option, ) -> crate::Result>> { let lo = doc_prefix(database_id, tenant, coll, doc_id); let hi = match sys_cutoff_ms { - Some(ms) => versioned_doc_key(database_id, tenant, coll, doc_id, ms)?, + Some(ms) => versioned_doc_key(database_id, tenant, coll, doc_id, ms), None => doc_prefix_end(database_id, tenant, coll, doc_id), }; let txn = self.db.begin_read().map_err(|e| redb_err("read txn", e))?; @@ -285,7 +286,7 @@ impl SparseEngine { database_id: u64, tenant: u64, coll: &str, - doc_id: &str, + doc_id: &StorageKey, ) -> crate::Result>> { self.versioned_get_as_of(database_id, tenant, coll, doc_id, None, None) } @@ -302,12 +303,16 @@ mod tests { (engine, dir) } - fn put(e: &SparseEngine, coll: &str, id: &str, sys_from: i64, body: &[u8]) { + fn key(surrogate: u32) -> StorageKey { + StorageKey::for_surrogate(nodedb_types::Surrogate::new(surrogate)) + } + + fn put(e: &SparseEngine, coll: &str, id: u32, sys_from: i64, body: &[u8]) { e.versioned_put(VersionedPut { database_id: 1, tenant: 1, coll, - doc_id: id, + doc_id: &key(id), sys_from_ms: sys_from, valid_from_ms: 0, valid_until_ms: i64::MAX, @@ -319,7 +324,7 @@ mod tests { fn put_valid( e: &SparseEngine, coll: &str, - id: &str, + id: u32, sys_from: i64, valid_from: i64, valid_until: i64, @@ -329,7 +334,7 @@ mod tests { database_id: 1, tenant: 1, coll, - doc_id: id, + doc_id: &key(id), sys_from_ms: sys_from, valid_from_ms: valid_from, valid_until_ms: valid_until, @@ -341,31 +346,31 @@ mod tests { #[test] fn put_and_read_current() { let (e, _d) = open_temp(); - put(&e, "users", "u1", 100, b"v1"); - let got = e.versioned_get_current(1, 1, "users", "u1").unwrap(); + put(&e, "users", 1, 100, b"v1"); + let got = e.versioned_get_current(1, 1, "users", &key(1)).unwrap(); assert_eq!(got.as_deref(), Some(b"v1" as &[u8])); } #[test] fn ceiling_picks_newest_le_cutoff() { let (e, _d) = open_temp(); - put(&e, "c", "k", 100, b"a"); - put(&e, "c", "k", 200, b"b"); - put(&e, "c", "k", 300, b"c"); + put(&e, "c", 1, 100, b"a"); + put(&e, "c", 1, 200, b"b"); + put(&e, "c", 1, 300, b"c"); assert_eq!( - e.versioned_get_as_of(1, 1, "c", "k", Some(150), None) + e.versioned_get_as_of(1, 1, "c", &key(1), Some(150), None) .unwrap() .as_deref(), Some(b"a" as &[u8]) ); assert_eq!( - e.versioned_get_as_of(1, 1, "c", "k", Some(250), None) + e.versioned_get_as_of(1, 1, "c", &key(1), Some(250), None) .unwrap() .as_deref(), Some(b"b" as &[u8]) ); assert_eq!( - e.versioned_get_as_of(1, 1, "c", "k", Some(400), None) + e.versioned_get_as_of(1, 1, "c", &key(1), Some(400), None) .unwrap() .as_deref(), Some(b"c" as &[u8]) @@ -375,9 +380,9 @@ mod tests { #[test] fn ceiling_before_first_version_is_none() { let (e, _d) = open_temp(); - put(&e, "c", "k", 200, b"x"); + put(&e, "c", 1, 200, b"x"); assert!( - e.versioned_get_as_of(1, 1, "c", "k", Some(100), None) + e.versioned_get_as_of(1, 1, "c", &key(1), Some(100), None) .unwrap() .is_none() ); @@ -386,16 +391,16 @@ mod tests { #[test] fn tombstone_hides_row_at_and_after_cutoff() { let (e, _d) = open_temp(); - put(&e, "c", "k", 100, b"x"); - e.versioned_tombstone(1, 1, "c", "k", 200).unwrap(); + put(&e, "c", 1, 100, b"x"); + e.versioned_tombstone(1, 1, "c", &key(1), 200).unwrap(); assert_eq!( - e.versioned_get_as_of(1, 1, "c", "k", Some(150), None) + e.versioned_get_as_of(1, 1, "c", &key(1), Some(150), None) .unwrap() .as_deref(), Some(b"x" as &[u8]) ); assert!( - e.versioned_get_as_of(1, 1, "c", "k", Some(250), None) + e.versioned_get_as_of(1, 1, "c", &key(1), Some(250), None) .unwrap() .is_none() ); @@ -404,22 +409,22 @@ mod tests { #[test] fn valid_time_predicate_skips_out_of_window_versions() { let (e, _d) = open_temp(); - put_valid(&e, "c", "k", 10, 0, 100, b"v1"); - put_valid(&e, "c", "k", 20, 200, 300, b"v2"); + put_valid(&e, "c", 1, 10, 0, 100, b"v1"); + put_valid(&e, "c", 1, 20, 200, 300, b"v2"); // valid-time hole at 150: neither version applies. assert!( - e.versioned_get_as_of(1, 1, "c", "k", Some(10_000), Some(150)) + e.versioned_get_as_of(1, 1, "c", &key(1), Some(10_000), Some(150)) .unwrap() .is_none() ); assert_eq!( - e.versioned_get_as_of(1, 1, "c", "k", Some(10_000), Some(50)) + e.versioned_get_as_of(1, 1, "c", &key(1), Some(10_000), Some(50)) .unwrap() .as_deref(), Some(b"v1" as &[u8]) ); assert_eq!( - e.versioned_get_as_of(1, 1, "c", "k", Some(10_000), Some(250)) + e.versioned_get_as_of(1, 1, "c", &key(1), Some(10_000), Some(250)) .unwrap() .as_deref(), Some(b"v2" as &[u8]) @@ -429,12 +434,12 @@ mod tests { #[test] fn gdpr_erase_preserves_history_structure_but_hides_body() { let (e, _d) = open_temp(); - put(&e, "c", "k", 100, b"pii"); - put(&e, "c", "k", 200, b"more-pii"); - let n = e.versioned_gdpr_erase(1, 1, "c", "k").unwrap(); + put(&e, "c", 1, 100, b"pii"); + put(&e, "c", 1, 200, b"more-pii"); + let n = e.versioned_gdpr_erase(1, 1, "c", &key(1)).unwrap(); assert_eq!(n, 2); assert!( - e.versioned_get_as_of(1, 1, "c", "k", Some(150), None) + e.versioned_get_as_of(1, 1, "c", &key(1), Some(150), None) .unwrap() .is_none() ); @@ -443,9 +448,9 @@ mod tests { #[test] fn scan_returns_latest_per_doc_id() { let (e, _d) = open_temp(); - put(&e, "c", "a", 100, b"a1"); - put(&e, "c", "a", 200, b"a2"); - put(&e, "c", "b", 150, b"b1"); + put(&e, "c", 1, 100, b"a1"); + put(&e, "c", 1, 200, b"a2"); + put(&e, "c", 2, 150, b"b1"); let all = e .versioned_scan_as_of( VersionedScanParams { @@ -456,24 +461,28 @@ mod tests { valid_at_ms: None, limit: 100, }, - &|_: &str, _: &[u8]| true, + &|_: &StorageKey, _: &[u8]| true, &crate::engine::sparse::scan_stop::never_stop, ) .unwrap(); - let map: std::collections::HashMap<_, _> = all.into_iter().collect(); - assert_eq!(map.get("a").map(|v| v.as_slice()), Some(b"a2" as &[u8])); - assert_eq!(map.get("b").map(|v| v.as_slice()), Some(b"b1" as &[u8])); + let find = |id: StorageKey| { + all.iter() + .find(|(k, _)| *k == id) + .map(|(_, v)| v.as_slice()) + }; + assert_eq!(find(key(1)), Some(b"a2" as &[u8])); + assert_eq!(find(key(2)), Some(b"b1" as &[u8])); } #[test] fn scan_all_returns_every_version_in_system_time_order() { let (e, _d) = open_temp(); // One document updated three times under different system times. - put(&e, "c", "a", 100, b"a1"); - put(&e, "c", "a", 200, b"a2"); - put(&e, "c", "a", 300, b"a3"); + put(&e, "c", 1, 100, b"a1"); + put(&e, "c", 1, 200, b"a2"); + put(&e, "c", 1, 300, b"a3"); // A second document interleaved by system time. - put(&e, "c", "b", 150, b"b1"); + put(&e, "c", 2, 150, b"b1"); let all = e .versioned_scan_all( @@ -485,7 +494,7 @@ mod tests { valid_at_ms: None, limit: 100, }, - &|_: &str, _: &[u8]| true, + &|_: &StorageKey, _: &[u8]| true, &crate::engine::sparse::scan_stop::never_stop, ) .unwrap(); @@ -495,19 +504,19 @@ mod tests { let times: Vec = all.iter().map(|r| r.system_from_ms).collect(); assert_eq!(times, vec![100, 150, 200, 300]); // System-time and body line up per version. - let row_of = |r: &VersionedRow| (r.doc_id.clone(), r.system_from_ms, r.body.clone()); - assert_eq!(row_of(&all[0]), ("a".to_string(), 100, b"a1".to_vec())); - assert_eq!(row_of(&all[1]), ("b".to_string(), 150, b"b1".to_vec())); - assert_eq!(row_of(&all[2]), ("a".to_string(), 200, b"a2".to_vec())); - assert_eq!(row_of(&all[3]), ("a".to_string(), 300, b"a3".to_vec())); + let row_of = |r: &VersionedRow| (r.doc_id, r.system_from_ms, r.body.clone()); + assert_eq!(row_of(&all[0]), (key(1), 100, b"a1".to_vec())); + assert_eq!(row_of(&all[1]), (key(2), 150, b"b1".to_vec())); + assert_eq!(row_of(&all[2]), (key(1), 200, b"a2".to_vec())); + assert_eq!(row_of(&all[3]), (key(1), 300, b"a3".to_vec())); } #[test] fn scan_all_skips_tombstoned_versions() { let (e, _d) = open_temp(); - put(&e, "c", "a", 100, b"a1"); - e.versioned_tombstone(1, 1, "c", "a", 200).unwrap(); - put(&e, "c", "a", 300, b"a3"); + put(&e, "c", 1, 100, b"a1"); + e.versioned_tombstone(1, 1, "c", &key(1), 200).unwrap(); + put(&e, "c", 1, 300, b"a3"); let all = e .versioned_scan_all( VersionedScanParams { @@ -518,7 +527,7 @@ mod tests { valid_at_ms: None, limit: 100, }, - &|_: &str, _: &[u8]| true, + &|_: &StorageKey, _: &[u8]| true, &crate::engine::sparse::scan_stop::never_stop, ) .unwrap(); @@ -535,10 +544,11 @@ mod tests { // counts MATCHING versions — not raw scanned rows. let (e, _d) = open_temp(); for i in 0..10i64 { - put(&e, "c", "a", 100 + i, format!("v{i}").as_bytes()); + put(&e, "c", 1, 100 + i, format!("v{i}").as_bytes()); } // Match only odd-suffixed bodies: v1, v3, v5, v7, v9. - let odd = |_: &str, body: &[u8]| body.last().map(|b| (b - b'0') % 2 == 1).unwrap_or(false); + let odd = + |_: &StorageKey, body: &[u8]| body.last().map(|b| (b - b'0') % 2 == 1).unwrap_or(false); let rows = e .versioned_scan_all( @@ -570,11 +580,11 @@ mod tests { // early-stop must count matching documents, so a selective filter cannot // make the scan return fewer rows than exist. let (e, _d) = open_temp(); - for (i, id) in ["a", "b", "c", "d", "e", "f"].iter().enumerate() { - put(&e, "c", id, 100 + i as i64, format!("x{i}").as_bytes()); + for (i, id) in [1u32, 2, 3, 4, 5, 6].iter().enumerate() { + put(&e, "c", *id, 100 + i as i64, format!("x{i}").as_bytes()); } - // Match only even-suffixed bodies: x0 (a), x2 (c), x4 (e). - let even = |_: &str, body: &[u8]| { + // Match only even-suffixed bodies: x0 (id 1), x2 (id 3), x4 (id 5). + let even = |_: &StorageKey, body: &[u8]| { body.last() .map(|b| (b - b'0').is_multiple_of(2)) .unwrap_or(false) @@ -611,8 +621,8 @@ mod tests { #[test] fn scan_as_of_hides_tombstoned_rows() { let (e, _d) = open_temp(); - put(&e, "c", "a", 100, b"a1"); - e.versioned_tombstone(1, 1, "c", "a", 200).unwrap(); + put(&e, "c", 1, 100, b"a1"); + e.versioned_tombstone(1, 1, "c", &key(1), 200).unwrap(); let at_150 = e .versioned_scan_as_of( VersionedScanParams { @@ -623,7 +633,7 @@ mod tests { valid_at_ms: None, limit: 100, }, - &|_: &str, _: &[u8]| true, + &|_: &StorageKey, _: &[u8]| true, &crate::engine::sparse::scan_stop::never_stop, ) .unwrap(); @@ -638,7 +648,7 @@ mod tests { valid_at_ms: None, limit: 100, }, - &|_: &str, _: &[u8]| true, + &|_: &StorageKey, _: &[u8]| true, &crate::engine::sparse::scan_stop::never_stop, ) .unwrap(); @@ -648,20 +658,26 @@ mod tests { #[test] fn versioned_remove_in_txn_deletes_the_version() { let (e, _d) = open_temp(); - put(&e, "c", "k", 100, b"v1"); + put(&e, "c", 1, 100, b"v1"); assert_eq!( - e.versioned_get_current(1, 1, "c", "k").unwrap().as_deref(), + e.versioned_get_current(1, 1, "c", &key(1)) + .unwrap() + .as_deref(), Some(b"v1" as &[u8]) ); let txn = e.db.begin_write().unwrap(); - e.versioned_remove_in_txn(&txn, 1, 1, "c", "k", 100) + e.versioned_remove_in_txn(&txn, 1, 1, "c", &key(1), 100) .unwrap(); txn.commit().unwrap(); - assert!(e.versioned_get_current(1, 1, "c", "k").unwrap().is_none()); assert!( - e.versioned_get_as_of(1, 1, "c", "k", Some(100), None) + e.versioned_get_current(1, 1, "c", &key(1)) + .unwrap() + .is_none() + ); + assert!( + e.versioned_get_as_of(1, 1, "c", &key(1), Some(100), None) .unwrap() .is_none() ); @@ -671,24 +687,8 @@ mod tests { fn versioned_remove_in_txn_on_missing_key_is_ok() { let (e, _d) = open_temp(); let txn = e.db.begin_write().unwrap(); - let r = e.versioned_remove_in_txn(&txn, 1, 1, "c", "does-not-exist", 999); + let r = e.versioned_remove_in_txn(&txn, 1, 1, "c", &key(999), 999); assert!(r.is_ok()); txn.commit().unwrap(); } - - #[test] - fn nul_in_doc_id_is_rejected() { - let (e, _d) = open_temp(); - let r = e.versioned_put(VersionedPut { - database_id: 1, - tenant: 1, - coll: "c", - doc_id: "a\x00b", - sys_from_ms: 100, - valid_from_ms: 0, - valid_until_ms: i64::MAX, - body: b"x", - }); - assert!(r.is_err()); - } } diff --git a/nodedb/src/engine/sparse/btree_versioned/key.rs b/nodedb/src/engine/sparse/btree_versioned/key.rs index 3cfea2fa8..8b1cdee67 100644 --- a/nodedb/src/engine/sparse/btree_versioned/key.rs +++ b/nodedb/src/engine/sparse/btree_versioned/key.rs @@ -2,40 +2,36 @@ //! Key layout for the versioned document and index tables. +use nodedb_types::StorageKey; + /// 20-digit zero-pad for i64 lexicographic ordering under reverse-scan. pub fn format_sys_from(sys_from_ms: i64) -> String { format!("{sys_from_ms:020}") } -/// Build a versioned document key. Returns an error if `doc_id` contains -/// a NUL byte — NUL is reserved as the version separator. +/// Build a versioned document key. pub fn versioned_doc_key( database_id: u64, tenant: u64, coll: &str, - doc_id: &str, + doc_id: &StorageKey, sys_from_ms: i64, -) -> crate::Result { - if doc_id.as_bytes().contains(&0) { - return Err(crate::Error::BadRequest { - detail: "document id may not contain NUL byte".into(), - }); - } - Ok(format!( +) -> String { + format!( "{database_id}:{tenant}:{coll}:{doc_id}\x00{}", format_sys_from(sys_from_ms) - )) + ) } /// Prefix matching every version of a single doc_id — used by reverse-scan. -pub fn doc_prefix(database_id: u64, tenant: u64, coll: &str, doc_id: &str) -> String { +pub fn doc_prefix(database_id: u64, tenant: u64, coll: &str, doc_id: &StorageKey) -> String { format!("{database_id}:{tenant}:{coll}:{doc_id}\x00") } /// Upper-bound exclusive companion of [`doc_prefix`]: because `\x00` is /// the minimum byte, `\x01` is the next-greater separator and bounds all /// suffixes for this doc_id cleanly. -pub fn doc_prefix_end(database_id: u64, tenant: u64, coll: &str, doc_id: &str) -> String { +pub fn doc_prefix_end(database_id: u64, tenant: u64, coll: &str, doc_id: &StorageKey) -> String { format!("{database_id}:{tenant}:{coll}:{doc_id}\x01") } @@ -71,18 +67,3 @@ pub fn parse_sys_from(key: &str) -> Option { let (_, suffix) = key.rsplit_once('\x00')?; suffix.parse().ok() } - -/// Extract `doc_id` slice from a versioned key (between the -/// `{database_id}:{tenant}:{coll}:` and `\x00` boundaries). Returns the whole -/// remainder when parsing fails. -pub fn parse_doc_id<'a>( - key: &'a str, - database_id: u64, - tenant: u64, - coll: &str, -) -> Option<&'a str> { - let prefix = format!("{database_id}:{tenant}:{coll}:"); - let rest = key.strip_prefix(&prefix)?; - let (id, _) = rest.rsplit_once('\x00')?; - Some(id) -} diff --git a/nodedb/src/engine/sparse/btree_versioned/mod.rs b/nodedb/src/engine/sparse/btree_versioned/mod.rs index 98a03f481..31e98c74a 100644 --- a/nodedb/src/engine/sparse/btree_versioned/mod.rs +++ b/nodedb/src/engine/sparse/btree_versioned/mod.rs @@ -2,12 +2,11 @@ //! Versioned document storage — bitemporal key layout backed by redb. //! -//! Key: `"{tenant}:{coll}:{doc_id}\x00{system_from_ms:020}"`. +//! Key: `"{tenant}:{coll}:{doc_id}\x00{system_from_ms:020}"`, where `doc_id` +//! is a [`nodedb_types::StorageKey`]'s 8-hex-character rendering — a hex +//! render cannot contain the `\x00` version separator. //! Value: `[tag:u8][valid_from_ms:i64 LE][valid_until_ms:i64 LE][body...]` //! where `tag = 0x00` (live), `0xFF` (tombstone), `0xFE` (GDPR erased). -//! -//! `\x00` is the reserved version separator; callers must reject doc_ids -//! that contain a NUL byte. pub mod doc; pub mod index; @@ -18,8 +17,8 @@ pub mod value; pub use doc::VersionedRow; pub use key::{ - coll_prefix, coll_prefix_end, doc_prefix, doc_prefix_end, format_sys_from, parse_doc_id, - parse_sys_from, tenant_prefix, tenant_prefix_end, versioned_doc_key, + coll_prefix, coll_prefix_end, doc_prefix, doc_prefix_end, format_sys_from, parse_sys_from, + tenant_prefix, tenant_prefix_end, versioned_doc_key, }; pub use value::{ DecodedValue, TAG_GDPR_ERASED, TAG_LIVE, TAG_TOMBSTONE, VersionedIndexEntry, VersionedPut, diff --git a/nodedb/src/engine/sparse/btree_versioned/purge.rs b/nodedb/src/engine/sparse/btree_versioned/purge.rs index b48922149..c97191169 100644 --- a/nodedb/src/engine/sparse/btree_versioned/purge.rs +++ b/nodedb/src/engine/sparse/btree_versioned/purge.rs @@ -259,9 +259,15 @@ impl SparseEngine { #[cfg(test)] mod tests { + use nodedb_types::StorageKey; + use super::super::value::VersionedPut; use super::*; + fn key(surrogate: u32) -> StorageKey { + StorageKey::for_surrogate(nodedb_types::Surrogate::new(surrogate)) + } + fn fresh_engine() -> (tempfile::TempDir, SparseEngine) { let dir = tempfile::tempdir().unwrap(); let path = dir.path().join("sparse.redb"); @@ -269,12 +275,12 @@ mod tests { (dir, eng) } - fn put_version(eng: &SparseEngine, doc_id: &str, sys_from: i64) { + fn put_version(eng: &SparseEngine, doc_id: u32, sys_from: i64) { eng.versioned_put(VersionedPut { database_id: 1, tenant: 1, coll: "c", - doc_id, + doc_id: &key(doc_id), body: b"payload", sys_from_ms: sys_from, valid_from_ms: 0, @@ -283,9 +289,9 @@ mod tests { .unwrap(); } - fn count_doc_versions(eng: &SparseEngine, doc_id: &str) -> usize { - let lo = super::super::key::doc_prefix(1, 1, "c", doc_id); - let hi = super::super::key::doc_prefix_end(1, 1, "c", doc_id); + fn count_doc_versions(eng: &SparseEngine, doc_id: u32) -> usize { + let lo = super::super::key::doc_prefix(1, 1, "c", &key(doc_id)); + let hi = super::super::key::doc_prefix_end(1, 1, "c", &key(doc_id)); let txn = eng.db.begin_read().unwrap(); let t = txn.open_table(DOCUMENTS_VERSIONED).unwrap(); t.range(lo.as_str()..hi.as_str()).unwrap().count() @@ -294,35 +300,35 @@ mod tests { #[test] fn purge_drops_superseded_doc_versions_below_cutoff() { let (_d, eng) = fresh_engine(); - put_version(&eng, "d1", 100); - put_version(&eng, "d1", 200); - put_version(&eng, "d1", 300); - assert_eq!(count_doc_versions(&eng, "d1"), 3); + put_version(&eng, 1, 100); + put_version(&eng, 1, 200); + put_version(&eng, 1, 300); + assert_eq!(count_doc_versions(&eng, 1), 3); let (docs, _idx) = eng .purge_superseded_document_versions(1, 1, "c", 150) .unwrap(); assert_eq!(docs, 1, "only v@100 is below cutoff AND superseded"); - assert_eq!(count_doc_versions(&eng, "d1"), 2); + assert_eq!(count_doc_versions(&eng, 1), 2); } #[test] fn purge_preserves_latest_version_even_below_cutoff() { let (_d, eng) = fresh_engine(); - put_version(&eng, "d1", 100); + put_version(&eng, 1, 100); // Only one version: it's the latest, never deleted. let (docs, _idx) = eng .purge_superseded_document_versions(1, 1, "c", 10_000) .unwrap(); assert_eq!(docs, 0); - assert_eq!(count_doc_versions(&eng, "d1"), 1); + assert_eq!(count_doc_versions(&eng, 1), 1); } #[test] fn purge_groups_by_doc_id() { let (_d, eng) = fresh_engine(); // Two docs, each with 3 versions at 100/200/300. - for id in ["d1", "d2"] { + for id in [1u32, 2] { put_version(&eng, id, 100); put_version(&eng, id, 200); put_version(&eng, id, 300); @@ -331,16 +337,16 @@ mod tests { .purge_superseded_document_versions(1, 1, "c", 150) .unwrap(); assert_eq!(docs, 2, "one victim per doc"); - assert_eq!(count_doc_versions(&eng, "d1"), 2); - assert_eq!(count_doc_versions(&eng, "d2"), 2); + assert_eq!(count_doc_versions(&eng, 1), 2); + assert_eq!(count_doc_versions(&eng, 2), 2); } #[test] fn purge_is_idempotent() { let (_d, eng) = fresh_engine(); - put_version(&eng, "d1", 100); - put_version(&eng, "d1", 200); - put_version(&eng, "d1", 300); + put_version(&eng, 1, 100); + put_version(&eng, 1, 200); + put_version(&eng, 1, 300); let (a, _) = eng .purge_superseded_document_versions(1, 1, "c", 150) .unwrap(); @@ -363,7 +369,7 @@ mod tests { fn delete_all_versioned_for_collection_drops_every_version() { let (_d, eng) = fresh_engine(); // Two docs, three system-time versions each in coll "c". - for id in ["d1", "d2"] { + for id in [1u32, 2] { put_version(&eng, id, 100); put_version(&eng, id, 200); put_version(&eng, id, 300); @@ -373,7 +379,7 @@ mod tests { database_id: 1, tenant: 1, coll: "other", - doc_id: "d1", + doc_id: &key(1), body: b"x", sys_from_ms: 100, valid_from_ms: 0, @@ -384,7 +390,7 @@ mod tests { database_id: 1, tenant: 2, coll: "c", - doc_id: "d1", + doc_id: &key(1), body: b"x", sys_from_ms: 100, valid_from_ms: 0, @@ -417,7 +423,7 @@ mod tests { database_id: 1, tenant: 1, coll, - doc_id: "d1", + doc_id: &key(1), body: b"x", sys_from_ms: 100, valid_from_ms: 0, @@ -430,7 +436,7 @@ mod tests { database_id: 1, tenant: 2, coll: "c", - doc_id: "d1", + doc_id: &key(1), body: b"x", sys_from_ms: 100, valid_from_ms: 0, @@ -453,13 +459,13 @@ mod tests { fn purge_scoped_to_tenant_collection() { let (_d, eng) = fresh_engine(); // same doc_id in another collection / tenant; must not be touched. - put_version(&eng, "d1", 100); - put_version(&eng, "d1", 200); + put_version(&eng, 1, 100); + put_version(&eng, 1, 200); eng.versioned_put(VersionedPut { database_id: 1, tenant: 1, coll: "other", - doc_id: "d1", + doc_id: &key(1), body: b"x", sys_from_ms: 100, valid_from_ms: 0, @@ -470,7 +476,7 @@ mod tests { database_id: 1, tenant: 1, coll: "other", - doc_id: "d1", + doc_id: &key(1), body: b"x", sys_from_ms: 200, valid_from_ms: 0, @@ -481,7 +487,7 @@ mod tests { database_id: 1, tenant: 2, coll: "c", - doc_id: "d1", + doc_id: &key(1), body: b"x", sys_from_ms: 100, valid_from_ms: 0, @@ -492,7 +498,7 @@ mod tests { database_id: 1, tenant: 2, coll: "c", - doc_id: "d1", + doc_id: &key(1), body: b"x", sys_from_ms: 200, valid_from_ms: 0, @@ -505,8 +511,8 @@ mod tests { .unwrap(); assert_eq!(docs, 1); // The other collection and tenant still have all their versions. - let lo_other = super::super::key::doc_prefix(1, 1, "other", "d1"); - let hi_other = super::super::key::doc_prefix_end(1, 1, "other", "d1"); + let lo_other = super::super::key::doc_prefix(1, 1, "other", &key(1)); + let hi_other = super::super::key::doc_prefix_end(1, 1, "other", &key(1)); let txn = eng.db.begin_read().unwrap(); let t = txn.open_table(DOCUMENTS_VERSIONED).unwrap(); assert_eq!( @@ -515,8 +521,8 @@ mod tests { .count(), 2 ); - let lo_t2 = super::super::key::doc_prefix(1, 2, "c", "d1"); - let hi_t2 = super::super::key::doc_prefix_end(1, 2, "c", "d1"); + let lo_t2 = super::super::key::doc_prefix(1, 2, "c", &key(1)); + let hi_t2 = super::super::key::doc_prefix_end(1, 2, "c", &key(1)); assert_eq!(t.range(lo_t2.as_str()..hi_t2.as_str()).unwrap().count(), 2); } } diff --git a/nodedb/src/engine/sparse/btree_versioned/scan.rs b/nodedb/src/engine/sparse/btree_versioned/scan.rs index ed28dfa0e..0f8fda1de 100644 --- a/nodedb/src/engine/sparse/btree_versioned/scan.rs +++ b/nodedb/src/engine/sparse/btree_versioned/scan.rs @@ -7,12 +7,13 @@ //! the caller's `stop` signal once per scanned version so a long scan can be //! ended from outside. +use nodedb_types::StorageKey; use redb::{ReadableDatabase, ReadableTable}; use super::doc::{DOCUMENTS_VERSIONED, VersionedRow}; use super::key::{coll_prefix, coll_prefix_end, format_sys_from}; use super::value::{VersionedScanParams, decode_value}; -use crate::engine::sparse::btree::{SparseEngine, redb_err}; +use crate::engine::sparse::btree::{KeyedTable, SparseEngine, invalid_storage_key_err, redb_err}; impl SparseEngine { /// Scan every doc_id in a collection at the requested cutoff. @@ -32,9 +33,9 @@ impl SparseEngine { pub fn versioned_scan_as_of( &self, params: VersionedScanParams<'_>, - predicate: &dyn Fn(&str, &[u8]) -> bool, + predicate: &dyn Fn(&StorageKey, &[u8]) -> bool, stop: &dyn Fn() -> bool, - ) -> crate::Result)>> { + ) -> crate::Result)>> { let VersionedScanParams { database_id, tenant, @@ -58,7 +59,7 @@ impl SparseEngine { // The table is sorted by (doc_id, sys_from) ascending so we can // stream and flush whenever doc_id changes. let mut out = Vec::new(); - let mut current_id: Option = None; + let mut current_id: Option = None; let mut best_for_current: Option<(i64, Vec)> = None; for r in range { @@ -70,9 +71,12 @@ impl SparseEngine { let Some((id_part, suffix)) = key_str.rsplit_once('\x00') else { continue; }; - let Some(doc_id) = id_part.strip_prefix(lo.as_str()) else { + let Some(seg) = id_part.strip_prefix(lo.as_str()) else { continue; }; + let doc_id = StorageKey::parse(seg).ok_or_else(|| { + invalid_storage_key_err(KeyedTable::DocumentsVersioned, coll, seg) + })?; if let Some(ref c) = cutoff_key && suffix > c.as_str() { @@ -81,14 +85,14 @@ impl SparseEngine { let Ok(sf) = suffix.parse::() else { continue; }; - if current_id.as_deref() != Some(doc_id) { - if let Some(prev_id) = current_id.as_ref() { + if current_id != Some(doc_id) { + if let Some(prev_id) = current_id { flush_scan(prev_id, &best_for_current, valid_at_ms, predicate, &mut out)?; if out.len() >= limit { return Ok(out); } } - current_id = Some(doc_id.to_string()); + current_id = Some(doc_id); best_for_current = None; } let val = v.value().to_vec(); @@ -98,7 +102,7 @@ impl SparseEngine { _ => (sf, val), }); } - if let Some(prev_id) = current_id.as_ref() { + if let Some(prev_id) = current_id { flush_scan(prev_id, &best_for_current, valid_at_ms, predicate, &mut out)?; } Ok(out) @@ -132,7 +136,7 @@ impl SparseEngine { pub fn versioned_scan_all( &self, params: VersionedScanParams<'_>, - predicate: &dyn Fn(&str, &[u8]) -> bool, + predicate: &dyn Fn(&StorageKey, &[u8]) -> bool, stop: &dyn Fn() -> bool, ) -> crate::Result> { let VersionedScanParams { @@ -163,9 +167,12 @@ impl SparseEngine { let Some((id_part, suffix)) = key_str.rsplit_once('\x00') else { continue; }; - let Some(doc_id) = id_part.strip_prefix(lo.as_str()) else { + let Some(seg) = id_part.strip_prefix(lo.as_str()) else { continue; }; + let doc_id = StorageKey::parse(seg).ok_or_else(|| { + invalid_storage_key_err(KeyedTable::DocumentsVersioned, coll, seg) + })?; let Ok(sf) = suffix.parse::() else { continue; }; @@ -181,11 +188,11 @@ impl SparseEngine { } // Push the caller's scan filters down here so the `limit` truncation // below counts only matching versions, never raw scanned rows. - if !predicate(doc_id, decoded.body) { + if !predicate(&doc_id, decoded.body) { continue; } all.push(VersionedRow { - doc_id: doc_id.to_string(), + doc_id, system_from_ms: sf, valid_from_ms: decoded.valid_from_ms, valid_until_ms: decoded.valid_until_ms, @@ -206,11 +213,11 @@ impl SparseEngine { /// Emit the newest-per-doc-id entry into `out` if it's live and passes /// the valid-time predicate. fn flush_scan( - id: &str, + id: StorageKey, pick: &Option<(i64, Vec)>, valid_at_ms: Option, - predicate: &dyn Fn(&str, &[u8]) -> bool, - out: &mut Vec<(String, Vec)>, + predicate: &dyn Fn(&StorageKey, &[u8]) -> bool, + out: &mut Vec<(StorageKey, Vec)>, ) -> crate::Result<()> { let Some((_sf, v)) = pick else { return Ok(()) }; let decoded = decode_value(v)?; @@ -224,9 +231,9 @@ fn flush_scan( } // Caller's scan filters are pushed down here so they are applied before the // row counts toward the scan's `limit`. - if !predicate(id, decoded.body) { + if !predicate(&id, decoded.body) { return Ok(()); } - out.push((id.to_string(), decoded.body.to_vec())); + out.push((id, decoded.body.to_vec())); Ok(()) } diff --git a/nodedb/src/engine/sparse/btree_versioned/value.rs b/nodedb/src/engine/sparse/btree_versioned/value.rs index cf617f066..fa1a69ead 100644 --- a/nodedb/src/engine/sparse/btree_versioned/value.rs +++ b/nodedb/src/engine/sparse/btree_versioned/value.rs @@ -2,6 +2,8 @@ //! Versioned payload format: `[tag:u8][valid_from_ms:i64 LE][valid_until_ms:i64 LE][body...]`. +use nodedb_types::StorageKey; + use super::key::format_sys_from; pub const TAG_LIVE: u8 = 0x00; @@ -66,7 +68,7 @@ pub struct VersionedPut<'a> { pub database_id: u64, pub tenant: u64, pub coll: &'a str, - pub doc_id: &'a str, + pub doc_id: &'a StorageKey, pub sys_from_ms: i64, pub valid_from_ms: i64, pub valid_until_ms: i64, diff --git a/nodedb/tests/inproc/cases/bitemporal_temporal_purge.rs b/nodedb/tests/inproc/cases/bitemporal_temporal_purge.rs index 606e0b540..73c101875 100644 --- a/nodedb/tests/inproc/cases/bitemporal_temporal_purge.rs +++ b/nodedb/tests/inproc/cases/bitemporal_temporal_purge.rs @@ -25,9 +25,9 @@ //! we use the registry's snapshot API directly and verify each entry //! maps to the correct `MetaOp::TemporalPurge*` variant by tag. -use nodedb_types::TenantId; use nodedb_types::config::BitemporalRetention; use nodedb_types::temporal::ms_to_ordinal_upper; +use nodedb_types::{StorageKey, Surrogate, TenantId}; use nodedb_wal::{TemporalPurgeEngine, TemporalPurgePayload}; use nodedb::engine::bitemporal::{ @@ -38,6 +38,11 @@ use nodedb::engine::graph::edge_store::temporal::EdgeRef; use nodedb::engine::sparse::btree::SparseEngine; use nodedb::engine::sparse::btree_versioned::VersionedPut; +/// Storage key for a small test surrogate. +fn key(surrogate: u32) -> StorageKey { + StorageKey::for_surrogate(Surrogate::new(surrogate)) +} + // ---------- EdgeStore ---------- #[test] @@ -88,7 +93,7 @@ fn document_strict_end_to_end_purge() { database_id: 0, tenant: 1, coll: "users", - doc_id: "u1", + doc_id: &key(1), body: b"payload", sys_from_ms: sys, valid_from_ms: 0, @@ -108,7 +113,7 @@ fn document_strict_end_to_end_purge() { database_id: 0, tenant: 1, coll: "orphan", - doc_id: "o1", + doc_id: &key(2), body: b"payload", sys_from_ms: 400, valid_from_ms: 0, diff --git a/nodedb/tests/inproc/cases/document_bitemporal_dml.rs b/nodedb/tests/inproc/cases/document_bitemporal_dml.rs index c1899b9a4..2d7367bf3 100644 --- a/nodedb/tests/inproc/cases/document_bitemporal_dml.rs +++ b/nodedb/tests/inproc/cases/document_bitemporal_dml.rs @@ -57,7 +57,7 @@ fn update_via_put_creates_new_version_and_preserves_old() { // History at t_mid still sees v=1. let body = sparse - .versioned_get_as_of(0, 1, "c", &key(1).to_string(), Some(t_mid), None) + .versioned_get_as_of(0, 1, "c", &key(1), Some(t_mid), None) .unwrap() .expect("historical version"); let rmpv_val = rmpv::decode::read_value(&mut body.as_slice()).unwrap(); @@ -83,7 +83,7 @@ fn delete_appends_tombstone_but_prior_version_still_visible_as_of() { // Historical read at t_before_delete: still Alice. let body = sparse - .versioned_get_as_of(0, 1, "c", &key(1).to_string(), Some(t_before_delete), None) + .versioned_get_as_of(0, 1, "c", &key(1), Some(t_before_delete), None) .unwrap() .expect("pre-delete version still reachable"); let rmpv_val = rmpv::decode::read_value(&mut body.as_slice()).unwrap(); @@ -105,7 +105,7 @@ fn ten_sequential_updates_produce_ten_reachable_versions() { } for (cutoff, expected) in &cutoffs { let body = sparse - .versioned_get_as_of(0, 1, "c", &key(1).to_string(), Some(*cutoff), None) + .versioned_get_as_of(0, 1, "c", &key(1), Some(*cutoff), None) .unwrap() .unwrap_or_else(|| panic!("missing version at cutoff {cutoff}")); let rmpv_val = rmpv::decode::read_value(&mut body.as_slice()).unwrap(); diff --git a/nodedb/tests/inproc/cases/document_bitemporal_store.rs b/nodedb/tests/inproc/cases/document_bitemporal_store.rs index 9a5cadc21..55f6cc4c5 100644 --- a/nodedb/tests/inproc/cases/document_bitemporal_store.rs +++ b/nodedb/tests/inproc/cases/document_bitemporal_store.rs @@ -77,7 +77,7 @@ fn bitemporal_multiple_puts_retain_history_via_versioned_get_as_of() { // A cutoff before the second write should surface v=1. let body = sparse - .versioned_get_as_of(0, 1, "c", &key(1).to_string(), Some(t_mid), None) + .versioned_get_as_of(0, 1, "c", &key(1), Some(t_mid), None) .unwrap() .expect("version at cutoff"); let val: serde_json::Value = { @@ -112,7 +112,7 @@ fn non_bitemporal_collection_uses_legacy_storage() { valid_at_ms: None, limit: 100, }, - &|_: &str, _: &[u8]| true, + &|_: &nodedb_types::StorageKey, _: &[u8]| true, &nodedb::engine::sparse::scan_stop::never_stop, ) .unwrap(); From 7952fa0e7f5a26c2fdd28520de8c2f9547a7dba1 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 12 Sep 2026 09:18:16 +0800 Subject: [PATCH 10/17] refactor(sparse): key versioned index entries by StorageKey VersionedIndexEntry::doc_id and versioned_index_lookup_as_of now carry StorageKey directly instead of round-tripping through its string form, removing the parse-at-the-boundary calls in callers and tests. --- .../executor/handlers/point/apply_delete.rs | 2 +- .../executor/handlers/point/apply_put/core.rs | 2 +- .../executor/handlers/point/update_reindex.rs | 6 +- .../handlers/transaction/undo/document.rs | 14 ++--- .../handlers/transaction/undo/rollback.rs | 6 +- .../src/engine/document/store/engine/batch.rs | 17 +---- .../engine/document/store/engine/delete.rs | 4 +- .../src/engine/document/store/engine/put.rs | 4 +- .../engine/sparse/btree_versioned/index.rs | 62 +++++++++++-------- .../engine/sparse/btree_versioned/value.rs | 2 +- .../inproc/cases/document_bitemporal_dml.rs | 6 +- 11 files changed, 56 insertions(+), 69 deletions(-) diff --git a/nodedb/src/data/executor/handlers/point/apply_delete.rs b/nodedb/src/data/executor/handlers/point/apply_delete.rs index a3c3a59df..1eef38f7f 100644 --- a/nodedb/src/data/executor/handlers/point/apply_delete.rs +++ b/nodedb/src/data/executor/handlers/point/apply_delete.rs @@ -220,7 +220,7 @@ impl CoreLoop { coll: collection, field: &path.path, value: &value, - doc_id: row_key, + doc_id: &storage_key, sys_from_ms: sys_from, }, )?; diff --git a/nodedb/src/data/executor/handlers/point/apply_put/core.rs b/nodedb/src/data/executor/handlers/point/apply_put/core.rs index 3afa0f576..bbdc93d7c 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/core.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/core.rs @@ -269,7 +269,7 @@ impl CoreLoop { coll: collection, field: &path.path, value: &value, - doc_id: document_id, + doc_id: &storage_key, sys_from_ms, }, )?; diff --git a/nodedb/src/data/executor/handlers/point/update_reindex.rs b/nodedb/src/data/executor/handlers/point/update_reindex.rs index 1a27ee3e4..3e968ba94 100644 --- a/nodedb/src/data/executor/handlers/point/update_reindex.rs +++ b/nodedb/src/data/executor/handlers/point/update_reindex.rs @@ -117,8 +117,6 @@ impl CoreLoop { // to `note_index_write_values` after the caller's commit without a // borrow conflict. let mut touched_values: Vec<(String, String)> = Vec::new(); - // INDEXES_VERSIONED still keys on the storage key as text. - let doc_id_str = p.doc_id.to_string(); for path in p.index_paths { let new_values = Self::indexed_values_for_path(p.new_doc, path); @@ -138,7 +136,7 @@ impl CoreLoop { coll: p.collection, field: &path.path, value, - doc_id: &doc_id_str, + doc_id: p.doc_id, sys_from_ms: p.sys_from_ms, }, )?; @@ -154,7 +152,7 @@ impl CoreLoop { coll: p.collection, field: &path.path, value, - doc_id: &doc_id_str, + doc_id: p.doc_id, sys_from_ms: p.sys_from_ms, }, )?; diff --git a/nodedb/src/data/executor/handlers/transaction/undo/document.rs b/nodedb/src/data/executor/handlers/transaction/undo/document.rs index 49f8b0052..bcc16db7e 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/document.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/document.rs @@ -226,8 +226,6 @@ impl CoreLoop { self.sparse .versioned_remove_in_txn(&txn, database_id, tid, collection, document_id, sys_from_ms) .map_err(|e| map_err("version remove", e.to_string()))?; - // INDEXES_VERSIONED still keys on the storage key as text. - let doc_id_str = document_id.to_string(); for (field, value) in index_tuples { self.sparse .versioned_index_remove_in_txn( @@ -238,7 +236,7 @@ impl CoreLoop { coll: collection, field, value, - doc_id: &doc_id_str, + doc_id: document_id, sys_from_ms, }, ) @@ -391,13 +389,13 @@ mod tests { coll: "c", field: "status", value: "active", - doc_id: &storage_key(doc).to_string(), + doc_id: &storage_key(doc), sys_from_ms: t, }) .unwrap(); } - fn index_lookup(core: &crate::data::executor::core_loop::CoreLoop) -> Vec { + fn index_lookup(core: &crate::data::executor::core_loop::CoreLoop) -> Vec { core.sparse .versioned_index_lookup_as_of(DB, TID, "c", "status", "active", None) .unwrap() @@ -419,7 +417,7 @@ mod tests { .unwrap() .is_some() ); - assert_eq!(index_lookup(&core), vec![d1.to_string()]); + assert_eq!(index_lookup(&core), vec![d1]); let entry = UndoEntry::PutDocument { collection: "c".into(), @@ -462,7 +460,7 @@ mod tests { coll: "c", field: "status", value: "active", - doc_id: &d1.to_string(), + doc_id: &d1, sys_from_ms: 2_000, }) .unwrap(); @@ -493,7 +491,7 @@ mod tests { .unwrap(), Some(b"v1".to_vec()) ); - assert_eq!(index_lookup(&core), vec![d1.to_string()]); + assert_eq!(index_lookup(&core), vec![d1]); } #[test] diff --git a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs index 82ac9345e..7e2203724 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs @@ -145,13 +145,13 @@ mod tests { coll: "c", field: "status", value: "active", - doc_id: &key(doc).to_string(), + doc_id: &key(doc), sys_from_ms: t, }) .unwrap(); } - fn index_lookup(core: &CoreLoop) -> Vec { + fn index_lookup(core: &CoreLoop) -> Vec { core.sparse .versioned_index_lookup_as_of(DB, TID, "c", "status", "active", None) .unwrap() @@ -190,7 +190,7 @@ mod tests { coll: "c", field: "status", value: "active", - doc_id: &d1.to_string(), + doc_id: &d1, sys_from_ms: 2_000, }) .unwrap(); diff --git a/nodedb/src/engine/document/store/engine/batch.rs b/nodedb/src/engine/document/store/engine/batch.rs index c9cc053a4..0c8965613 100644 --- a/nodedb/src/engine/document/store/engine/batch.rs +++ b/nodedb/src/engine/document/store/engine/batch.rs @@ -76,27 +76,14 @@ impl<'a> DocumentEngine<'a> { bitemporal: bool, ) -> crate::Result> { if bitemporal { - // The versioned index still yields text, so this parses at the boundary. - let ids = self.sparse.versioned_index_lookup_as_of( + return self.sparse.versioned_index_lookup_as_of( self.database_id, self.tenant_id, collection, path, value, None, - )?; - return ids - .into_iter() - .map(|id| { - StorageKey::parse(&id).ok_or_else(|| { - crate::engine::sparse::btree::invalid_storage_key_err( - crate::engine::sparse::btree::KeyedTable::IndexesVersioned, - collection, - &id, - ) - }) - }) - .collect(); + ); } let prefix_with_value = format!("{value}:"); let results = diff --git a/nodedb/src/engine/document/store/engine/delete.rs b/nodedb/src/engine/document/store/engine/delete.rs index b987e372e..926a453b9 100644 --- a/nodedb/src/engine/document/store/engine/delete.rs +++ b/nodedb/src/engine/document/store/engine/delete.rs @@ -33,8 +33,6 @@ impl<'a> DocumentEngine<'a> { if let Some(config) = self.configs.get(collection) && let Ok(rmpv_val) = crate::util::bounded_msgpack::read_value(&body) { - // INDEXES_VERSIONED still keys on the storage key as text. - let doc_id_str = doc_id.to_string(); for index_path in &config.index_paths { for v in extract_index_values_rmpv(&rmpv_val, &index_path.path, index_path.is_array) @@ -46,7 +44,7 @@ impl<'a> DocumentEngine<'a> { coll: collection, field: &index_path.path, value: &v, - doc_id: &doc_id_str, + doc_id, sys_from_ms: sys_from, }, )?; diff --git a/nodedb/src/engine/document/store/engine/put.rs b/nodedb/src/engine/document/store/engine/put.rs index 987f14c9d..54654a6c6 100644 --- a/nodedb/src/engine/document/store/engine/put.rs +++ b/nodedb/src/engine/document/store/engine/put.rs @@ -66,8 +66,6 @@ impl<'a> DocumentEngine<'a> { if let Some(config) = self.configs.get(collection) && let Ok(value) = crate::util::bounded_msgpack::read_value(msgpack_bytes) { - // INDEXES_VERSIONED still keys on the storage key as text. - let doc_id_str = doc_id.to_string(); for index_path in &config.index_paths { let values = extract_index_values_rmpv(&value, &index_path.path, index_path.is_array); @@ -81,7 +79,7 @@ impl<'a> DocumentEngine<'a> { coll: collection, field: &index_path.path, value: &v, - doc_id: &doc_id_str, + doc_id, sys_from_ms: sys_from, }, )?; diff --git a/nodedb/src/engine/sparse/btree_versioned/index.rs b/nodedb/src/engine/sparse/btree_versioned/index.rs index 0cffdfc91..c286c8f10 100644 --- a/nodedb/src/engine/sparse/btree_versioned/index.rs +++ b/nodedb/src/engine/sparse/btree_versioned/index.rs @@ -5,11 +5,12 @@ //! Index key: `"{database_id}:{tenant}:{coll}:{field}:{value}:{doc_id}\x00{sys_from:020}"`. //! Value: single byte (`0x00` live, `0xFF` tombstone). +use nodedb_types::StorageKey; use redb::{ReadableDatabase, ReadableTable, TableDefinition}; use super::key::format_sys_from; use super::value::{TAG_LIVE, TAG_TOMBSTONE, VersionedIndexEntry}; -use crate::engine::sparse::btree::{SparseEngine, redb_err}; +use crate::engine::sparse::btree::{KeyedTable, SparseEngine, invalid_storage_key_err, redb_err}; /// Keys carry the leading `{database_id}:` component. pub(crate) const INDEXES_VERSIONED: TableDefinition<&str, &[u8]> = @@ -37,11 +38,6 @@ impl SparseEngine { txn: &redb::WriteTransaction, e: VersionedIndexEntry<'_>, ) -> crate::Result<()> { - if e.doc_id.as_bytes().contains(&0) { - return Err(crate::Error::BadRequest { - detail: "document id may not contain NUL byte".into(), - }); - } let key = e.redb_key(); let mut t = txn .open_table(INDEXES_VERSIONED) @@ -119,7 +115,7 @@ impl SparseEngine { field: &str, value: &str, sys_cutoff_ms: Option, - ) -> crate::Result> { + ) -> crate::Result> { let lo = format!("{database_id}:{tenant}:{coll}:{field}:{value}:"); // `:` = 0x3A, next byte `;` = 0x3B gives a clean exclusive bound. let hi = format!("{database_id}:{tenant}:{coll}:{field}:{value};"); @@ -133,9 +129,9 @@ impl SparseEngine { .range(lo.as_str()..hi.as_str()) .map_err(|e| redb_err("range", e))?; - // Group by doc_id; keep newest-in-window tag per doc. - let mut out = Vec::new(); - let mut current_id: Option = None; + // Group by doc_id; keep newest-in-window tag per group. + let mut out: Vec = Vec::new(); + let mut current_id: Option = None; let mut best: Option<(i64, u8)> = None; for r in range { @@ -144,9 +140,11 @@ impl SparseEngine { let Some(rest) = key_str.strip_prefix(lo.as_str()) else { continue; }; - let Some((doc_id, suffix)) = rest.rsplit_once('\x00') else { + let Some((seg, suffix)) = rest.rsplit_once('\x00') else { continue; }; + let doc_id = StorageKey::parse(seg) + .ok_or_else(|| invalid_storage_key_err(KeyedTable::IndexesVersioned, coll, seg))?; if let Some(ref c) = cutoff_key && suffix > c.as_str() { @@ -157,14 +155,14 @@ impl SparseEngine { }; let tag = v.value().first().copied().unwrap_or(TAG_TOMBSTONE); - if current_id.as_deref() != Some(doc_id) { - if let Some(prev) = current_id.as_ref() + if current_id != Some(doc_id) { + if let Some(prev_id) = current_id && let Some((_, t)) = best && t == TAG_LIVE { - out.push(prev.clone()); + out.push(prev_id); } - current_id = Some(doc_id.to_string()); + current_id = Some(doc_id); best = None; } best = Some(match best.take() { @@ -172,11 +170,11 @@ impl SparseEngine { _ => (sf, tag), }); } - if let Some(prev) = current_id + if let Some(doc_id) = current_id && let Some((_, t)) = best && t == TAG_LIVE { - out.push(prev); + out.push(doc_id); } Ok(out) } @@ -184,6 +182,8 @@ impl SparseEngine { #[cfg(test)] mod tests { + use nodedb_types::Surrogate; + use super::*; fn open_temp() -> (SparseEngine, tempfile::TempDir) { @@ -192,11 +192,15 @@ mod tests { (engine, dir) } + fn key(surrogate: u32) -> StorageKey { + StorageKey::for_surrogate(Surrogate::new(surrogate)) + } + fn idx_entry<'a>( coll: &'a str, field: &'a str, value: &'a str, - doc_id: &'a str, + doc_id: &'a StorageKey, sys_from_ms: i64, ) -> VersionedIndexEntry<'a> { VersionedIndexEntry { @@ -213,17 +217,19 @@ mod tests { #[test] fn index_lookup_honors_cutoff_and_tombstone() { let (e, _d) = open_temp(); - e.versioned_index_put(idx_entry("c", "email", "a@x", "u1", 100)) + let u1 = key(1); + let u2 = key(2); + e.versioned_index_put(idx_entry("c", "email", "a@x", &u1, 100)) .unwrap(); - e.versioned_index_put(idx_entry("c", "email", "a@x", "u2", 150)) + e.versioned_index_put(idx_entry("c", "email", "a@x", &u2, 150)) .unwrap(); - e.versioned_index_tombstone(idx_entry("c", "email", "a@x", "u1", 200)) + e.versioned_index_tombstone(idx_entry("c", "email", "a@x", &u1, 200)) .unwrap(); let at_120 = e .versioned_index_lookup_as_of(1, 1, "c", "email", "a@x", Some(120)) .unwrap(); - assert_eq!(at_120, vec!["u1"]); + assert_eq!(at_120, vec![u1]); let at_175 = e .versioned_index_lookup_as_of(1, 1, "c", "email", "a@x", Some(175)) @@ -233,21 +239,22 @@ mod tests { let at_250 = e .versioned_index_lookup_as_of(1, 1, "c", "email", "a@x", Some(250)) .unwrap(); - assert_eq!(at_250, vec!["u2"]); + assert_eq!(at_250, vec![u2]); } #[test] fn versioned_index_remove_in_txn_removes_entry() { let (e, _d) = open_temp(); - e.versioned_index_put(idx_entry("c", "email", "a@x", "u1", 100)) + let u1 = key(1); + e.versioned_index_put(idx_entry("c", "email", "a@x", &u1, 100)) .unwrap(); let before = e .versioned_index_lookup_as_of(1, 1, "c", "email", "a@x", Some(150)) .unwrap(); - assert_eq!(before, vec!["u1"]); + assert_eq!(before, vec![u1]); let txn = e.db.begin_write().unwrap(); - e.versioned_index_remove_in_txn(&txn, idx_entry("c", "email", "a@x", "u1", 100)) + e.versioned_index_remove_in_txn(&txn, idx_entry("c", "email", "a@x", &u1, 100)) .unwrap(); txn.commit().unwrap(); @@ -260,9 +267,10 @@ mod tests { #[test] fn versioned_index_remove_in_txn_on_missing_key_is_ok() { let (e, _d) = open_temp(); + let u9 = key(9); let txn = e.db.begin_write().unwrap(); let r = - e.versioned_index_remove_in_txn(&txn, idx_entry("c", "email", "nobody@x", "u9", 999)); + e.versioned_index_remove_in_txn(&txn, idx_entry("c", "email", "nobody@x", &u9, 999)); assert!(r.is_ok()); txn.commit().unwrap(); } diff --git a/nodedb/src/engine/sparse/btree_versioned/value.rs b/nodedb/src/engine/sparse/btree_versioned/value.rs index fa1a69ead..8bc08c82e 100644 --- a/nodedb/src/engine/sparse/btree_versioned/value.rs +++ b/nodedb/src/engine/sparse/btree_versioned/value.rs @@ -85,7 +85,7 @@ pub struct VersionedIndexEntry<'a> { pub coll: &'a str, pub field: &'a str, pub value: &'a str, - pub doc_id: &'a str, + pub doc_id: &'a StorageKey, pub sys_from_ms: i64, } diff --git a/nodedb/tests/inproc/cases/document_bitemporal_dml.rs b/nodedb/tests/inproc/cases/document_bitemporal_dml.rs index 2d7367bf3..9020b0841 100644 --- a/nodedb/tests/inproc/cases/document_bitemporal_dml.rs +++ b/nodedb/tests/inproc/cases/document_bitemporal_dml.rs @@ -138,13 +138,13 @@ fn secondary_index_reflects_each_version_independently() { let ids_a_mid = sparse .versioned_index_lookup_as_of(0, 1, "c", "$.email", "a@x.com", Some(t_mid)) .unwrap(); - assert_eq!(ids_a_mid, vec![key(1).to_string()]); + assert_eq!(ids_a_mid, vec![key(1)]); // Current: "b@x.com" → u1. let ids_b_now = sparse .versioned_index_lookup_as_of(0, 1, "c", "$.email", "b@x.com", None) .unwrap(); - assert_eq!(ids_b_now, vec![key(1).to_string()]); + assert_eq!(ids_b_now, vec![key(1)]); // After delete → no current entry for b@x.com either. engine.delete("c", &key(1)).unwrap(); @@ -156,7 +156,7 @@ fn secondary_index_reflects_each_version_independently() { let ids_a_still = sparse .versioned_index_lookup_as_of(0, 1, "c", "$.email", "a@x.com", Some(t_mid)) .unwrap(); - assert_eq!(ids_a_still, vec![key(1).to_string()]); + assert_eq!(ids_a_still, vec![key(1)]); } #[test] From 2a4c45908ac4cb0aa25d275880dab709ce780ea5 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 12 Sep 2026 11:03:49 +0800 Subject: [PATCH 11/17] refactor(executor): thread StorageKey through scan and filter matching Replace ad-hoc &str/String doc-id handling in the scan, merge, and facet paths with the typed StorageKey, matching the storage layer's key representation end to end. Row matching, overlay merging, bulk DML staging, and secondary-index counting now compare typed keys instead of parsing or re-stringifying hex identities per row. --- .../data/executor/core_loop/filter_match.rs | 37 +++--- nodedb/src/data/executor/dispatch/meta.rs | 3 +- .../data/executor/handlers/bulk_dml/scan.rs | 25 ++-- .../data/executor/handlers/bulk_dml/update.rs | 49 ++++---- .../handlers/control/range_scan_versioned.rs | 28 +---- .../executor/handlers/document/index_fetch.rs | 30 ++--- .../executor/handlers/document/read/fetch.rs | 39 +++---- .../handlers/document/read/fetch_types.rs | 16 +-- .../document/read/materialize_scan.rs | 70 +++++------- .../executor/handlers/document/read/scan.rs | 89 ++++++--------- nodedb/src/data/executor/handlers/facet.rs | 5 +- .../executor/handlers/merge/target_docs.rs | 24 ++-- .../src/data/executor/handlers/text_search.rs | 2 +- .../executor/handlers/text_search_scan.rs | 15 +-- .../handlers/transaction/overlay/fts_merge.rs | 19 ++- .../handlers/transaction/overlay/merge.rs | 108 +++++++----------- .../stage_write/stage_bulk_delete.rs | 21 ++-- .../stage_write/stage_bulk_update.rs | 29 ++--- .../handlers/update_from_join_collect.rs | 53 +++++---- nodedb/src/engine/sparse/btree_scan.rs | 7 +- 20 files changed, 293 insertions(+), 376 deletions(-) diff --git a/nodedb/src/data/executor/core_loop/filter_match.rs b/nodedb/src/data/executor/core_loop/filter_match.rs index 0325eb510..18764b1f6 100644 --- a/nodedb/src/data/executor/core_loop/filter_match.rs +++ b/nodedb/src/data/executor/core_loop/filter_match.rs @@ -14,6 +14,7 @@ //! evaluate strict predicates identically. use nodedb_query::EvalError; +use nodedb_types::StorageKey; use nodedb_types::columnar::StrictSchema; use crate::bridge::scan_filter::ScanFilter; @@ -38,18 +39,17 @@ use super::CoreLoop; /// the behavior-flip rule applies: the query fails instead of the row being /// silently excluded. /// -/// `doc_id` is the row's storage key. A schemaless collection with no +/// `row_key` is the row's storage key. A schemaless collection with no /// declared `id` field carries its identity only in that key, never in the /// body, so the body is matched with `id` injected — the same injection /// [`super::super::row_shape::sparse_row_to_doc`] applies to a materialized /// row, so `WHERE id ...` sees the identity a reader of the same row sees. A -/// minted key injects the client-visible decimal identity, not the hex -/// storage key. A strict row already surfaces `id` as a real tuple column, so -/// no injection runs on that arm. +/// strict row already surfaces `id` as a real tuple column, so no injection +/// runs on that arm. pub(in crate::data::executor) fn matches_with_resolved_schema( strict_schema: Option<&StrictSchema>, filters: &[ScanFilter], - doc_id: &str, + row_key: &StorageKey, body: &[u8], ) -> Result { match strict_schema { @@ -58,10 +58,7 @@ pub(in crate::data::executor) fn matches_with_resolved_schema( None => Ok(false), }, None => { - // `doc_id` comes straight off a store iterator, so only a `&str` - // is available here, not a `StorageKey`. A value that fails to - // parse as a minted key is a legacy or user key, taken verbatim. - let identity = crate::engine::document::store::identity_of(doc_id); + let identity = row_key.to_identity(); let with_id = nodedb_query::msgpack_scan::inject_str_field(body, "id", identity.as_str()); ScanFilter::all_match_binary(filters, &with_id) @@ -96,18 +93,18 @@ impl CoreLoop { }) } - /// Build a reusable `Fn(&str, &[u8]) -> Result` closure - /// evaluating `filters` against a stored row's `(doc_id, body)`, + /// Build a reusable `Fn(&StorageKey, &[u8]) -> Result` + /// closure evaluating `filters` against a stored row's `(row_key, body)`, /// resolving `collection`'s strict schema ONCE up front (not per row) and /// capturing it in the closure. Suitable for a hot per-row scan loop /// directly, or as the fallible half of a Cell-wrapping call pattern — /// [`CoreLoop::merge_overlay_into_scan`] actually expects an - /// *infallible* `&dyn Fn(&str, &[u8]) -> bool`, not this function's own - /// `Result`-returning output, so callers that feed it into that merge - /// wrap the closure this function returns in a second, infallible one - /// that stashes any `Err` into a local `Cell>` and - /// checks it once the merge call returns — see - /// `stage_bulk_delete.rs`/`stage_bulk_update.rs`'s `raw_matches` / + /// *infallible* `&dyn Fn(&StorageKey, &[u8]) -> bool`, not this + /// function's own `Result`-returning output, so callers that feed it + /// into that merge wrap the closure this function returns in a second, + /// infallible one that stashes any `Err` into a local + /// `Cell>` and checks it once the merge call returns — + /// see `stage_bulk_delete.rs`/`stage_bulk_update.rs`'s `raw_matches` / /// `matches` pair for the exact pattern. pub(in crate::data::executor) fn strict_aware_matcher<'a>( &self, @@ -115,10 +112,10 @@ impl CoreLoop { tid: u64, collection: &str, filters: &'a [ScanFilter], - ) -> impl Fn(&str, &[u8]) -> Result + 'a { + ) -> impl Fn(&StorageKey, &[u8]) -> Result + 'a { let strict_schema = self.resolve_strict_schema(database_id, tid, collection); - move |doc_id: &str, body: &[u8]| { - matches_with_resolved_schema(strict_schema.as_ref(), filters, doc_id, body) + move |row_key: &StorageKey, body: &[u8]| { + matches_with_resolved_schema(strict_schema.as_ref(), filters, row_key, body) } } } diff --git a/nodedb/src/data/executor/dispatch/meta.rs b/nodedb/src/data/executor/dispatch/meta.rs index df6497593..dcbd92982 100644 --- a/nodedb/src/data/executor/dispatch/meta.rs +++ b/nodedb/src/data/executor/dispatch/meta.rs @@ -386,6 +386,7 @@ mod txn_created_columnar_engine_tests { use nodedb_bridge::buffer::RingBuffer; use nodedb_physical::physical_plan::MetaOp; + use nodedb_types::StorageKey; use nodedb_types::Surrogate; use nodedb_types::columnar::{ColumnDef, ColumnType, ColumnarSchema}; use nodedb_types::value::Value; @@ -768,7 +769,7 @@ mod txn_created_columnar_engine_tests { TenantId::new(TID), "refresh_a".to_string(), ); - let mut rows: Vec<(String, Vec)> = Vec::new(); + let mut rows: Vec<(StorageKey, Vec)> = Vec::new(); core.merge_overlay_into_scan(txn_a, &coll_key, &mut rows, &|_, _| true); core.reap_expired_overlays(); diff --git a/nodedb/src/data/executor/handlers/bulk_dml/scan.rs b/nodedb/src/data/executor/handlers/bulk_dml/scan.rs index c98eb7fed..ad8b34278 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/scan.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/scan.rs @@ -42,19 +42,20 @@ impl CoreLoop { let mut ids = Vec::new(); if let Ok(range) = table.range(prefix.as_str()..end.as_str()) { for entry in range.flatten() { - let key = entry.0.value(); + let full_key = entry.0.value(); let value_bytes = entry.1.value(); - if let Some(doc_id) = key.strip_prefix(&prefix) - && matches(doc_id, value_bytes)? - { - let doc_id = StorageKey::parse(doc_id).ok_or_else(|| { - crate::engine::sparse::btree::invalid_storage_key_err( - crate::engine::sparse::btree::KeyedTable::Documents, - collection, - doc_id, - ) - })?; - ids.push(doc_id); + let Some(rest) = full_key.strip_prefix(&prefix) else { + continue; + }; + let key = StorageKey::parse(rest).ok_or_else(|| { + crate::engine::sparse::btree::invalid_storage_key_err( + crate::engine::sparse::btree::KeyedTable::Documents, + collection, + rest, + ) + })?; + if matches(&key, value_bytes)? { + ids.push(key); } } } diff --git a/nodedb/src/data/executor/handlers/bulk_dml/update.rs b/nodedb/src/data/executor/handlers/bulk_dml/update.rs index 3a0f08a19..372de8e25 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/update.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/update.rs @@ -337,30 +337,26 @@ impl CoreLoop { &updated_bytes, ); // Record the committed row's write version against its - // surrogate + collection. Parsed once and reused below - // for the write-set entry (the row's doc_id is the - // hex-encoded surrogate storage key either way). - let row_surrogate = crate::engine::document::store::doc_id_to_surrogate(doc_id); - if let Some(surrogate) = row_surrogate { - self.note_surrogate_write_lsn(task, tid, collection, surrogate.as_u32()); - // Re-index the row's vectors from the new body - // (soft-delete the old HNSW node + insert the new - // one, keyed by the stable surrogate). No-op unless - // the collection has a vector field (gated above). - if has_vectors - && let Err(e) = self.update_reindex_vector_indexes(UpdateVectorReindex { - database_id, - tid, - collection, - row_key: doc_id, - surrogate, - new_body: &updated_bytes, - is_strict: strict_schema.is_some(), - has_vectors, - }) - { - return self.response_error(task, e); - } + // surrogate + collection. + let surrogate = storage_key.surrogate(); + self.note_surrogate_write_lsn(task, tid, collection, surrogate.as_u32()); + // Re-index the row's vectors from the new body (soft-delete the + // old HNSW node + insert the new one, keyed by the stable + // surrogate). No-op unless the collection has a vector field + // (gated above). + if has_vectors + && let Err(e) = self.update_reindex_vector_indexes(UpdateVectorReindex { + database_id, + tid, + collection, + row_key: doc_id, + surrogate, + new_body: &updated_bytes, + is_strict: strict_schema.is_some(), + has_vectors, + }) + { + return self.response_error(task, e); } // Emit an update event per affected row to the Event Plane, // so AFTER-UPDATE triggers and CDC/change-stream consumers @@ -398,9 +394,8 @@ impl CoreLoop { // Carry the surrogate + post-image back for a post-apply // `Put` redo. `updated_bytes` is moved as its last use; // gated on `has_vectors` so a non-vector collection pays - // nothing. Keyed by the row's surrogate parsed from its - // doc_id (the hex-encoded surrogate storage key). - if has_vectors && let Some(surrogate) = row_surrogate { + // nothing. + if has_vectors { write_set.push(WriteSetEntry { surrogate: surrogate.as_u32(), is_delete: false, diff --git a/nodedb/src/data/executor/handlers/control/range_scan_versioned.rs b/nodedb/src/data/executor/handlers/control/range_scan_versioned.rs index 99c9b617c..0e994b0d5 100644 --- a/nodedb/src/data/executor/handlers/control/range_scan_versioned.rs +++ b/nodedb/src/data/executor/handlers/control/range_scan_versioned.rs @@ -170,7 +170,7 @@ impl CoreLoop { // version. A statement that goes over its deadline mid-scan stops here // and the check below turns the short result into an error. let deadline = crate::data::executor::deadline::DeadlineCheck::for_task(task); - let scanned = match self.sparse.versioned_scan_as_of( + let mut scanned = match self.sparse.versioned_scan_as_of( crate::engine::sparse::btree_versioned::VersionedScanParams { database_id: task.request.database_id.as_u64(), tenant: tid, @@ -194,28 +194,6 @@ impl CoreLoop { } }; - // `merge_overlay_into_scan` operates on hex-rendered keys — the same - // text shape the base (non-versioned) scan path carries — so the - // typed keys from the versioned scan render once here at the - // boundary, and the predicate re-parses back to a `StorageKey` per - // candidate row. - let mut scanned: Vec<(String, Vec)> = scanned - .into_iter() - .map(|(k, v)| (k.to_string(), v)) - .collect(); - let predicate_str = - |doc_id: &str, body: &[u8]| match nodedb_types::StorageKey::parse(doc_id) { - Some(key) => predicate(&key, body), - None => { - decode_err.set(Some(crate::engine::sparse::btree::invalid_storage_key_err( - crate::engine::sparse::btree::KeyedTable::DocumentsVersioned, - collection, - doc_id, - ))); - false - } - }; - // Read-your-own-writes: fold this transaction's staging overlay onto // the current-version base result, using the SAME range predicate on // the raw stored bodies (staged strict bodies are Binary Tuples, like @@ -228,7 +206,7 @@ impl CoreLoop { crate::types::TenantId::new(tid), collection.to_string(), ); - self.merge_overlay_into_scan(txn_id, &coll_key, &mut scanned, &predicate_str); + self.merge_overlay_into_scan(txn_id, &coll_key, &mut scanned, &predicate); } // Both passes above run the same predicate; check the side-channel once, @@ -258,7 +236,7 @@ impl CoreLoop { }, None => body, }; - rows.push((id, mp)); + rows.push((id.to_string(), mp)); } // Sort ascending by `field` and cap at `limit` — the same ordering the diff --git a/nodedb/src/data/executor/handlers/document/index_fetch.rs b/nodedb/src/data/executor/handlers/document/index_fetch.rs index 5827474f3..b8a370439 100644 --- a/nodedb/src/data/executor/handlers/document/index_fetch.rs +++ b/nodedb/src/data/executor/handlers/document/index_fetch.rs @@ -13,6 +13,7 @@ //! Non-bitemporal collections keep the byte-identical plain //! `range_scan` + `sparse.get` path. +use nodedb_types::StorageKey; use tracing::{debug, warn}; use crate::bridge::envelope::{ErrorCode, Response}; @@ -72,8 +73,7 @@ impl CoreLoop { tid, ); match doc_engine.index_lookup(collection, path, value, bitemporal) { - Ok(doc_ids) => { - let mut doc_ids: Vec = doc_ids.into_iter().map(|k| k.to_string()).collect(); + Ok(mut doc_ids) => { if let Some(txn_id) = task.request.txn_id { let config_key = ( task.request.database_id, @@ -110,6 +110,7 @@ impl CoreLoop { return self.response_error(task, e); } } + let doc_ids: Vec = doc_ids.iter().map(|k| k.to_string()).collect(); let payload = serde_json::json!(doc_ids); match sonic_rs::to_vec(&payload) { Ok(bytes) => self.response_with_payload(task, bytes), @@ -176,9 +177,9 @@ impl CoreLoop { let bitemporal = self.is_bitemporal(database_id, tid, collection); let doc_engine = crate::engine::document::store::DocumentEngine::new(&self.sparse, database_id, tid); - let mut doc_ids: Vec = + let mut doc_ids: Vec = match doc_engine.index_lookup(collection, path, value, bitemporal) { - Ok(ids) => ids.into_iter().map(|k| k.to_string()).collect(), + Ok(ids) => ids, Err(e) => { return self.response_error( task, @@ -272,20 +273,11 @@ impl CoreLoop { // the staged `Put` bytes and only falls back to a base fetch // when the overlay has nothing staged for this surrogate. let fetched = self.overlay_or_base_body(task.request.txn_id, &coll_key, doc_id, || { - // `doc_id` is an index-lookup result, not a scan of DOCUMENTS - // itself; a shape that fails to parse as a storage key names - // no row in that table, matching what a lookup on the - // unparsed key would already have found. - match nodedb_types::StorageKey::parse(doc_id) { - Some(key) => { - if bitemporal { - self.sparse - .versioned_get_current(database_id, tid, collection, &key) - } else { - self.sparse.get(database_id, tid, collection, &key) - } - } - None => Ok(None), + if bitemporal { + self.sparse + .versioned_get_current(database_id, tid, collection, doc_id) + } else { + self.sparse.get(database_id, tid, collection, doc_id) } }); match fetched { @@ -321,7 +313,7 @@ impl CoreLoop { } else { bytes }; - rows.push((doc_id.clone(), payload)); + rows.push((doc_id.to_string(), payload)); } Ok(None) => { // Index entry pointed at a deleted doc — skip, don't diff --git a/nodedb/src/data/executor/handlers/document/read/fetch.rs b/nodedb/src/data/executor/handlers/document/read/fetch.rs index 39a06caa7..0e1d2106c 100644 --- a/nodedb/src/data/executor/handlers/document/read/fetch.rs +++ b/nodedb/src/data/executor/handlers/document/read/fetch.rs @@ -29,7 +29,7 @@ use super::fetch_types::FetchedRows; use super::{DocFetchParams, DocScanMode}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::core_loop::filter_match::matches_with_resolved_schema; -use crate::data::executor::scan_normalize::{sparse_body_to_msgpack, sparse_row_to_doc}; +use crate::data::executor::scan_normalize::sparse_body_to_msgpack; use crate::data::executor::sparse_body_format::{SparseBodyFormat, SparseBodyFormatRef}; use crate::data::executor::task::ExecutionTask; @@ -77,7 +77,7 @@ impl CoreLoop { |doc_id: &StorageKey, body: &[u8]| match matches_with_resolved_schema( strict_schema, filter_predicates, - &doc_id.to_string(), + doc_id, body, ) { Ok(b) => b, @@ -114,7 +114,7 @@ impl CoreLoop { SparseBodyFormatRef::from_schema(strict_schema), ) .into_owned(); - (doc_id.to_string(), mp) + (doc_id, mp) }) .collect(); Ok(FetchedRows { @@ -134,7 +134,7 @@ impl CoreLoop { |doc_id: &StorageKey, body: &[u8]| match matches_with_resolved_schema( strict_schema, filter_predicates, - &doc_id.to_string(), + doc_id, body, ) { Ok(b) => b, @@ -158,7 +158,7 @@ impl CoreLoop { if let Some(e) = predicate_err.take() { return Err(crate::Error::from(e)); } - let mut rows: Vec<(String, Vec)> = Vec::with_capacity(raw.len()); + let mut rows: Vec<(StorageKey, Vec)> = Vec::with_capacity(raw.len()); for row in raw { let msgpack_body = match strict_schema { Some(schema) => strict_audit_body(&row.body, schema)?, @@ -170,7 +170,7 @@ impl CoreLoop { row.valid_from_ms, row.valid_until_ms, )?; - rows.push((row.doc_id.to_string(), with_ts)); + rows.push((row.doc_id, with_ts)); } Ok(FetchedRows { rows, @@ -231,7 +231,7 @@ impl CoreLoop { // side-channel and checked once every branch below returns, rather // than silently folded away. let predicate_err: Cell> = Cell::new(None); - let matches = |doc_id: &str, value: &[u8]| -> bool { + let matches = |key: &StorageKey, value: &[u8]| -> bool { if filter_predicates.is_empty() { return true; } @@ -246,7 +246,7 @@ impl CoreLoop { } else { value }; - match matches_with_resolved_schema(strict_schema, filter_predicates, doc_id, value) { + match matches_with_resolved_schema(strict_schema, filter_predicates, key, value) { Ok(b) => b, Err(e) => { predicate_err.set(Some(e)); @@ -254,10 +254,6 @@ impl CoreLoop { } } }; - // `scan_documents_filtered` and `versioned_scan_as_of` hand the - // predicate a typed `StorageKey`; `matches` still takes the row's - // storage key as text, so this renders it once per candidate row. - let matches_by_key = |key: &StorageKey, value: &[u8]| matches(&key.to_string(), value); let rows: Vec<(StorageKey, Vec)> = if filter_predicates.is_empty() { if bitemporal { @@ -298,7 +294,7 @@ impl CoreLoop { valid_at_ms: None, limit: fetch_limit, }, - &matches_by_key, + &matches, &stop, )? } else { @@ -307,7 +303,7 @@ impl CoreLoop { tid, collection, fetch_limit, - &matches_by_key, + &matches, &stop, )? }; @@ -323,17 +319,20 @@ impl CoreLoop { // columns, projection, DISTINCT — sees the same standard-msgpack shape // it sees for every other collection. Without it the tagged values pass // through untouched and reach the client as `[4,"alice"]`. The key - // stays typed from the scan above, so this needs no re-parse. - let rows: Vec<(String, Vec)> = if is_vector_sidecar { + // stays typed; the row's envelope id is rendered once downstream. + let rows: Vec<(StorageKey, Vec)> = if is_vector_sidecar { rows.into_iter() .map(|(key, body)| { - sparse_row_to_doc(&key, &body, SparseBodyFormatRef::VectorSidecar) + let (_, mp) = crate::data::executor::scan_normalize::sparse_row_to_doc( + &key, + &body, + SparseBodyFormatRef::VectorSidecar, + ); + (key, mp) }) .collect() } else { - rows.into_iter() - .map(|(key, body)| (key.to_string(), body)) - .collect() + rows }; Ok(FetchedRows { diff --git a/nodedb/src/data/executor/handlers/document/read/fetch_types.rs b/nodedb/src/data/executor/handlers/document/read/fetch_types.rs index 0f9d4946c..e3a43cdbd 100644 --- a/nodedb/src/data/executor/handlers/document/read/fetch_types.rs +++ b/nodedb/src/data/executor/handlers/document/read/fetch_types.rs @@ -48,23 +48,9 @@ pub(in crate::data::executor) struct DocFetchParams<'a> { pub full_fetch: bool, } -/// Parse a fetched row's id back into the storage key it was minted as. -/// -/// Every row a document fetch produces is keyed by a rendered surrogate, so -/// a shape that fails to parse names a fetch-pipeline bug, never a row to -/// skip. -pub(super) fn parse_fetched_key(collection: &str, id: &str) -> crate::Result { - StorageKey::parse(id).ok_or_else(|| crate::Error::Storage { - engine: "sparse".into(), - detail: format!( - "collection '{collection}' fetched a row whose id is not a valid storage key: '{id}'" - ), - }) -} - /// Raw rows plus the schema the downstream should decode them with. pub(in crate::data::executor) struct FetchedRows { - pub rows: Vec<(String, Vec)>, + pub rows: Vec<(StorageKey, Vec)>, pub effective_schema: Option, /// The statement's deadline passed while the storage scan was running, so /// `rows` holds an arbitrary prefix of the answer. The caller MUST fail the diff --git a/nodedb/src/data/executor/handlers/document/read/materialize_scan.rs b/nodedb/src/data/executor/handlers/document/read/materialize_scan.rs index a904f0caa..9a6efc1f4 100644 --- a/nodedb/src/data/executor/handlers/document/read/materialize_scan.rs +++ b/nodedb/src/data/executor/handlers/document/read/materialize_scan.rs @@ -9,14 +9,15 @@ //! filter naming `id` sees the same identity the read paths produce. //! Payload: `[next_cursor: bin, entries: [[doc_id, surrogate, value], ...]]`. +use nodedb_types::StorageKey; +use redb::{ReadableDatabase, ReadableTable}; + use crate::bridge::envelope::Response; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::scan_normalize::sparse_row_to_doc; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::doc_id_to_surrogate; -use crate::engine::sparse::btree::DOCUMENTS; +use crate::engine::sparse::btree::{DOCUMENTS, KeyedTable, invalid_storage_key_err}; use crate::types::{DatabaseId, TenantId}; -use redb::{ReadableDatabase, ReadableTable}; impl CoreLoop { /// Execute a cursor-paginated raw document scan for the clone materializer. @@ -94,8 +95,8 @@ impl CoreLoop { // callers (`txn_id == None`) keep cursor-paginated base-only behavior. let txn_id = task.request.txn_id; - let mut entries: Vec<(String, u32, Vec)> = Vec::with_capacity(count.min(256)); - let mut last_doc_id = String::new(); + let mut entries: Vec<(StorageKey, Vec)> = Vec::with_capacity(count.min(256)); + let mut last_key: Option = None; for row in range { if txn_id.is_none() && entries.len() >= count { @@ -112,23 +113,21 @@ impl CoreLoop { ); } }; - let full_key = row.0.value().to_string(); - let doc_id = full_key - .strip_prefix(&prefix) - .unwrap_or(&full_key) - .to_string(); - let value = row.1.value().to_vec(); - - let surrogate = match doc_id_to_surrogate(&doc_id) { - Some(s) => s.as_u32(), + let full_key = row.0.value(); + let rest = full_key.strip_prefix(&prefix).unwrap_or(full_key); + let key = match StorageKey::parse(rest) { + Some(key) => key, None => { - // Skip non-surrogate keys (legacy or corrupted rows). - continue; + return self.response_error( + task, + invalid_storage_key_err(KeyedTable::Documents, collection, rest), + ); } }; + let value = row.1.value().to_vec(); - last_doc_id.clone_from(&doc_id); - entries.push((doc_id, surrogate, value)); + last_key = Some(key); + entries.push((key, value)); } // Fold the staging overlay into the base set: a staged tombstone @@ -140,24 +139,16 @@ impl CoreLoop { TenantId::new(tid), collection.to_string(), ); - let mut rows: Vec<(String, Vec)> = entries - .into_iter() - .map(|(doc_id, _surrogate, value)| (doc_id, value)) - .collect(); - self.merge_overlay_into_scan(txn_id, &coll_key, &mut rows, &|_, _| true); - entries = rows - .into_iter() - .filter_map(|(doc_id, value)| { - doc_id_to_surrogate(&doc_id).map(|s| (doc_id, s.as_u32(), value)) - }) - .collect(); + self.merge_overlay_into_scan(txn_id, &coll_key, &mut entries, &|_, _| true); // The whole set is returned in one response; the scan is complete. Vec::new() } else if entries.len() < count { // Next-cursor is the last doc_id_hex seen; empty = scan complete. Vec::new() } else { - last_doc_id.into_bytes() + last_key + .map(|k| k.to_string().into_bytes()) + .unwrap_or_default() }; // Normalize every body to standard msgpack and inject its `id` here — @@ -167,21 +158,16 @@ impl CoreLoop { let body_format = self.sparse_body_format(task.request.database_id, TenantId::new(tid), collection); let format_ref = body_format.as_format_ref(); - for entry in &mut entries { - // `entry.1` is this row's surrogate, already parsed out of - // `entry.0` when the row was collected above — minted directly - // from it rather than re-parsed. - let key = - nodedb_types::StorageKey::for_surrogate(nodedb_types::Surrogate::new(entry.1)); - let (_, normalized) = sparse_row_to_doc(&key, &entry.2, format_ref); - entry.2 = normalized; + for (key, value) in &mut entries { + let (_, normalized) = sparse_row_to_doc(key, value, format_ref); + *value = normalized; } // Encode response: [next_cursor: bin, entries: [[str, u32, bin], ...]] let mut payload = Vec::with_capacity( entries .iter() - .map(|(d, _, v)| d.len() + 4 + v.len() + 12) + .map(|(_, v)| 8 + 4 + v.len() + 12) .sum::() + next_cursor.len() + 16, @@ -189,10 +175,10 @@ impl CoreLoop { nodedb_query::msgpack_scan::write_array_header(&mut payload, 2); write_bin(&mut payload, &next_cursor); nodedb_query::msgpack_scan::write_array_header(&mut payload, entries.len()); - for (doc_id, surrogate, value) in &entries { + for (key, value) in &entries { nodedb_query::msgpack_scan::write_array_header(&mut payload, 3); - write_str(&mut payload, doc_id.as_bytes()); - write_u32(&mut payload, *surrogate); + write_str(&mut payload, key.to_string().as_bytes()); + write_u32(&mut payload, key.surrogate().as_u32()); write_bin(&mut payload, value); } diff --git a/nodedb/src/data/executor/handlers/document/read/scan.rs b/nodedb/src/data/executor/handlers/document/read/scan.rs index e8a25c1e8..6439f1e6d 100644 --- a/nodedb/src/data/executor/handlers/document/read/scan.rs +++ b/nodedb/src/data/executor/handlers/document/read/scan.rs @@ -2,9 +2,9 @@ //! Document collection scan handler. +use nodedb_types::StorageKey; use tracing::{debug, warn}; -use super::fetch_types::parse_fetched_key; use super::projection::{apply_projection, apply_projection_msgpack}; use super::{DocFetchParams, DocScanMode}; use crate::bridge::envelope::{ErrorCode, Response}; @@ -16,21 +16,6 @@ use crate::data::executor::scan_normalize::sparse_row_to_doc; use crate::data::executor::sparse_body_format::SparseBodyFormatRef; use crate::data::executor::task::ExecutionTask; -/// Shape one fetched row into `(identity, standard msgpack body)`. -/// -/// The id reaches this handler as rendered storage-key text several calls -/// removed from the scan that produced it: `document_scan_fetch` unifies every -/// source — current, `AS OF`, bitemporal — to `(String, Vec)`. -fn fetched_row_to_doc( - collection: &str, - id: &str, - body: &[u8], - body_format: SparseBodyFormatRef<'_>, -) -> crate::Result<(String, Vec)> { - let key = parse_fetched_key(collection, id)?; - Ok(sparse_row_to_doc(&key, body, body_format)) -} - /// Parameters for [`CoreLoop::execute_document_scan`]. pub(in crate::data::executor) struct DocumentScanParams<'a> { pub tid: u64, @@ -194,19 +179,19 @@ impl CoreLoop { collection.to_string(), ); // `merge_overlay_into_scan` takes an infallible - // `Fn(&str, &[u8]) -> bool` predicate, so a + // `Fn(&StorageKey, &[u8]) -> bool` predicate, so a // division/modulo-by-zero is captured via this `Cell` // side-channel and checked once the merge returns. let predicate_err: std::cell::Cell> = std::cell::Cell::new(None); - let matches = |doc_id: &str, value: &[u8]| -> bool { + let matches = |row_key: &StorageKey, value: &[u8]| -> bool { if filter_predicates.is_empty() { return true; } match crate::data::executor::core_loop::filter_match::matches_with_resolved_schema( effective_schema.as_ref(), &filter_predicates, - doc_id, + row_key, value, ) { Ok(b) => b, @@ -229,39 +214,50 @@ impl CoreLoop { // row-bounded; one whose rows are reordered or deduplicated // downstream had to gather the whole collection and is bounded // here instead. - if (limit == usize::MAX || !sort_keys.is_empty() || distinct) - && crate::data::executor::handlers::scan_budget::scan_bytes_exceeded( - &filtered, + if limit == usize::MAX || !sort_keys.is_empty() || distinct { + // Rendering to text happens downstream (once, per + // `sparse_row_to_doc`); a `StorageKey` always renders to + // exactly 8 hex characters, so the id contribution to the + // budget is that fixed width rather than a render. + let total_bytes = filtered.iter().fold(0usize, |acc, (_, value)| { + acc.saturating_add(value.len()).saturating_add(8) + }); + if crate::data::executor::handlers::scan_budget::budget_exceeded( + total_bytes, scan_budget_bytes, - ) - { - return self.response_error(task, ErrorCode::ResourcesExhausted); + ) { + return self.response_error(task, ErrorCode::ResourcesExhausted); + } } if let Some(pf) = prefilter { - filtered.retain(|(doc_id, _)| { - match crate::engine::document::store::doc_id_to_surrogate(doc_id) { - Some(surrogate) => pf.contains(surrogate), - None => false, - } - }); + filtered.retain(|(key, _)| pf.contains(key.surrogate())); } // Strict collections may store binary tuples. Sort and projection // operate on msgpack, so normalize binary tuples here — through // the shared converter, which leaves an already-msgpack body // borrowed and so costs nothing on the schemaless path. - let filtered = if !sort_keys.is_empty() || !projection.is_empty() { - match filtered + // Every stage past this point that reads fields (sort, window, + // projection, DISTINCT on a strict body) needs the normalized + // shape, so it is produced once here. A scan that reaches the + // client untouched keeps the stored bytes and renders only the + // envelope id. + let normalizes = !sort_keys.is_empty() + || !projection.is_empty() + || !computed_cols.is_empty() + || !window_specs.is_empty() + || effective_schema.is_some(); + let filtered: Vec<(String, Vec)> = if normalizes { + filtered .into_iter() - .map(|(id, bytes)| fetched_row_to_doc(collection, &id, &bytes, body_format)) - .collect::>>() - { - Ok(rows) => rows, - Err(e) => return self.response_error(task, e), - } + .map(|(key, bytes)| sparse_row_to_doc(&key, &bytes, body_format)) + .collect() } else { filtered + .into_iter() + .map(|(key, body)| (key.to_string(), body)) + .collect() }; let sorted = if sort_keys.is_empty() { @@ -305,9 +301,7 @@ impl CoreLoop { // `SELECT DISTINCT category`. Project first, then dedupe. let projected_rows: Vec<_> = match sorted .into_iter() - .map(|(doc_id, val)| { - let (doc_id, mp) = - fetched_row_to_doc(collection, &doc_id, &val, body_format)?; + .map(|(doc_id, mp)| { let projected = apply_projection_msgpack(&mp, &computed_cols, projection)?; Ok((doc_id, projected)) @@ -331,14 +325,9 @@ impl CoreLoop { } if !window_specs.is_empty() { - // Route through `sparse_row_to_doc`, like every sibling - // branch, so a schemaless row with no `id` field carries - // its storage-key identity into the window computation. let mut decoded_rows: Vec<(String, serde_json::Value)> = match sorted .into_iter() - .map(|(id, val)| { - let (doc_id, mp) = - fetched_row_to_doc(collection, &id, &val, body_format)?; + .map(|(doc_id, mp)| { crate::data::executor::doc_format::decode_document(&mp) .map(|doc| (doc_id, doc)) }) @@ -391,9 +380,7 @@ impl CoreLoop { // row, not the raw document. let projected_rows: Vec<_> = match sorted .into_iter() - .map(|(doc_id, value)| { - let (doc_id, mp) = - fetched_row_to_doc(collection, &doc_id, &value, body_format)?; + .map(|(doc_id, mp)| { let projected = apply_projection_msgpack(&mp, &computed_cols, projection)?; Ok((doc_id, projected)) diff --git a/nodedb/src/data/executor/handlers/facet.rs b/nodedb/src/data/executor/handlers/facet.rs index f19130888..7bf5f57f6 100644 --- a/nodedb/src/data/executor/handlers/facet.rs +++ b/nodedb/src/data/executor/handlers/facet.rs @@ -67,7 +67,8 @@ impl CoreLoop { } }; - let matching_set: HashSet = matching_ids.iter().map(|k| k.to_string()).collect(); + let matching_set: HashSet = + matching_ids.iter().copied().collect(); // Step 2: For each facet field, count values. let mut facet_result = serde_json::Map::new(); @@ -132,7 +133,7 @@ impl CoreLoop { tid: u64, collection: &str, field: &str, - matching_set: &HashSet, + matching_set: &HashSet, matching_ids: &[nodedb_types::StorageKey], ) -> Vec<(String, usize)> { // Fast path: index-backed counting with filtered doc set. diff --git a/nodedb/src/data/executor/handlers/merge/target_docs.rs b/nodedb/src/data/executor/handlers/merge/target_docs.rs index a186edb1d..8131565f4 100644 --- a/nodedb/src/data/executor/handlers/merge/target_docs.rs +++ b/nodedb/src/data/executor/handlers/merge/target_docs.rs @@ -9,9 +9,11 @@ //! identical target set or the classification they agree on is meaningless, so //! there is exactly one place that decides what "the target" is. +use nodedb_types::StorageKey; use redb::{ReadableDatabase, ReadableTable}; use crate::data::executor::core_loop::CoreLoop; +use crate::engine::sparse::btree::tables::{KeyedTable, invalid_storage_key_err}; impl CoreLoop { /// Collect every target row as `(doc_id, stored_bytes)` from a consistent @@ -24,10 +26,10 @@ impl CoreLoop { /// row, a staged put replaces the base body, and a staged put absent from /// base is appended — so an in-transaction MERGE resolved at COMMIT sees rows /// staged by earlier statements in the same transaction. The `doc_id` this - /// produces is the hex surrogate, matching the overlay's surrogate keying, so - /// staged and base bodies (same canonical stored form — Binary Tuple for a - /// strict target, MessagePack for a schemaless one) are merged like-for-like - /// and decoded identically downstream by `decode_target`. + /// produces is the storage key's text, matching the overlay's surrogate + /// keying, so staged and base bodies (same canonical stored form — Binary + /// Tuple for a strict target, MessagePack for a schemaless one) are merged + /// like-for-like and decoded identically downstream by `decode_target`. pub(in crate::data::executor) fn collect_target_docs( &self, database_id: u64, @@ -53,13 +55,16 @@ impl CoreLoop { detail: format!("open table: {e}"), })?; - let mut docs = Vec::new(); + let mut docs: Vec<(StorageKey, Vec)> = Vec::new(); if let Ok(range) = table.range(prefix.as_str()..end.as_str()) { for entry in range.flatten() { let key = entry.0.value(); let bytes = entry.1.value().to_vec(); - if let Some(doc_id) = key.strip_prefix(&prefix) { - docs.push((doc_id.to_string(), bytes)); + if let Some(rest) = key.strip_prefix(&prefix) { + let storage_key = StorageKey::parse(rest).ok_or_else(|| { + invalid_storage_key_err(KeyedTable::Documents, collection, rest) + })?; + docs.push((storage_key, bytes)); } } } @@ -76,6 +81,9 @@ impl CoreLoop { ); self.merge_overlay_into_scan(txn_id, &coll_key, &mut docs, &|_, _| true); } - Ok(docs) + Ok(docs + .into_iter() + .map(|(key, body)| (key.to_string(), body)) + .collect()) } } diff --git a/nodedb/src/data/executor/handlers/text_search.rs b/nodedb/src/data/executor/handlers/text_search.rs index 710afa17e..74f93dc8b 100644 --- a/nodedb/src/data/executor/handlers/text_search.rs +++ b/nodedb/src/data/executor/handlers/text_search.rs @@ -217,7 +217,7 @@ impl CoreLoop { } let hex_key = crate::engine::document::store::surrogate_to_doc_id(surrogate); let storage_key = nodedb_types::StorageKey::for_surrogate(surrogate); - let bytes_opt = match self.overlay_or_base_body(txn_id, &coll_key, &hex_key, || { + let bytes_opt = match self.overlay_or_base_body(txn_id, &coll_key, &storage_key, || { self.sparse.get(database_id, tid, collection, &storage_key) }) { Ok(b) => b, diff --git a/nodedb/src/data/executor/handlers/text_search_scan.rs b/nodedb/src/data/executor/handlers/text_search_scan.rs index d5bfbf7be..4932c7ef1 100644 --- a/nodedb/src/data/executor/handlers/text_search_scan.rs +++ b/nodedb/src/data/executor/handlers/text_search_scan.rs @@ -8,6 +8,7 @@ use tracing::debug; use nodedb_fts::FtsSearchParams; use nodedb_fts::posting::QueryMode; +use nodedb_types::StorageKey; use crate::bridge::envelope::{ErrorCode, Response}; @@ -204,11 +205,8 @@ impl CoreLoop { collection, BM25_SCAN_MAX_HITS, ); - // Rendered to text here: `merge_fts_rows_from_score_map` below and the - // per-row surrogate lookup both operate on the hex storage key as a - // string, out of this unit's typed scope. - let mut docs: Vec<(String, Vec)> = match scan_result { - Ok(d) => d.into_iter().map(|(k, v)| (k.to_string(), v)).collect(), + let mut docs: Vec<(StorageKey, Vec)> = match scan_result { + Ok(d) => d, Err(e) => { return self.response_error( task, @@ -243,15 +241,14 @@ impl CoreLoop { } let mut rows: Vec = Vec::with_capacity(docs.len()); - for (hex_key, bytes) in &docs { + for (key, bytes) in &docs { let mut value = match decode_scanned_document(bytes, format.as_format_ref()) { Ok(v) => v, Err(e) => return self.response_error(task, e), }; // Inject score into the document object. if let serde_json::Value::Object(ref mut map) = value { - let score = crate::engine::document::store::doc_id_to_surrogate(hex_key) - .and_then(|s| score_map.get(&s).copied()); + let score = score_map.get(&key.surrogate()).copied(); match score { Some(s) => { map.insert( @@ -268,7 +265,7 @@ impl CoreLoop { } } rows.push(DocumentRow { - id: hex_key.clone(), + id: key.to_string(), data: value, }); } diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/fts_merge.rs b/nodedb/src/data/executor/handlers/transaction/overlay/fts_merge.rs index 66ed14093..08272394c 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/fts_merge.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/fts_merge.rs @@ -34,7 +34,7 @@ use nodedb_types::Surrogate; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::transaction::overlay::{Staged, TxnOverlay}; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::StorageKey; use crate::types::{DatabaseId, TenantId, TxnId}; /// Scope + tuning for one FTS overlay merge: the transaction, the @@ -318,7 +318,7 @@ impl CoreLoop { pub(in crate::data::executor) fn merge_fts_rows_from_score_map( &self, params: FtsMergeParams<'_>, - rows: &mut Vec<(String, Vec)>, + rows: &mut Vec<(StorageKey, Vec)>, score_map: &HashMap, ) { let FtsMergeParams { @@ -335,15 +335,11 @@ impl CoreLoop { return; }; - let mut seen: std::collections::HashSet = rows - .iter() - .filter_map(|(k, _)| u32::from_str_radix(k, 16).ok()) - .collect(); + let mut seen: std::collections::HashSet = + rows.iter().map(|(k, _)| k.surrogate().as_u32()).collect(); rows.retain_mut(|(row_key, body)| { - let Ok(surrogate) = u32::from_str_radix(row_key, 16) else { - return true; - }; + let surrogate = row_key.surrogate().as_u32(); match overlay.get(&coll_key, surrogate) { Some(Staged::Tombstone) => false, Some(Staged::Put(staged_body)) => { @@ -361,7 +357,10 @@ impl CoreLoop { if let Staged::Put(body) = staged && score_map.contains_key(&Surrogate::new(surrogate)) { - rows.push((surrogate_to_doc_id(Surrogate::new(surrogate)), body.clone())); + rows.push(( + StorageKey::for_surrogate(Surrogate::new(surrogate)), + body.clone(), + )); seen.insert(surrogate); } } diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/merge.rs b/nodedb/src/data/executor/handlers/transaction/overlay/merge.rs index 32b454393..4bce930c6 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/merge.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/merge.rs @@ -31,14 +31,14 @@ use std::collections::HashSet; -use nodedb_types::Surrogate; use nodedb_types::columnar::StrictSchema; +use nodedb_types::{StorageKey, Surrogate}; use crate::bridge::scan_filter::ScanFilter; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::core_loop::filter_match::matches_with_resolved_schema; use crate::data::executor::handlers::transaction::overlay::{Staged, StagedTtl}; -use crate::engine::document::store::{extract_index_values, surrogate_to_doc_id}; +use crate::engine::document::store::extract_index_values; use crate::engine::kv::current_ms; use crate::types::{DatabaseId, TenantId, TxnId}; @@ -74,17 +74,17 @@ pub(in crate::data::executor) struct IndexOverlayMergeParams<'a> { } impl CoreLoop { - /// Merge the overlay for `txn_id` into `rows` (base scan `(hex_row_key, + /// Merge the overlay for `txn_id` into `rows` (base scan `(StorageKey, /// body)` pairs). `matches` is the SAME predicate the base scan applied, - /// evaluated on a stored row's `(doc_id, body)` (Binary Tuple for strict, + /// evaluated on a stored row's `(row_key, body)` (Binary Tuple for strict, /// MessagePack for schemaless). No-op when the transaction has no overlay /// entries. pub(in crate::data::executor) fn merge_overlay_into_scan( &self, txn_id: TxnId, coll_key: &(DatabaseId, TenantId, String), - rows: &mut Vec<(String, Vec)>, - matches: &dyn Fn(&str, &[u8]) -> bool, + rows: &mut Vec<(StorageKey, Vec)>, + matches: &dyn Fn(&StorageKey, &[u8]) -> bool, ) { // Read-your-own-writes refreshes the lease (see the reaper). self.touch_overlay(txn_id); @@ -95,22 +95,13 @@ impl CoreLoop { // Surrogates already represented in the base result. Additions consult // this to avoid re-adding a row that base already carries (or that the // retain pass has just superseded in place). - let mut seen: HashSet = rows - .iter() - .filter_map(|(k, _)| { - crate::engine::document::store::doc_id_to_surrogate(k).map(|s| s.as_u32()) - }) - .collect(); + let mut seen: HashSet = rows.iter().map(|(k, _)| k.surrogate().as_u32()).collect(); // Base-minus-superseded: a single in-place pass. Drop tombstoned rows, // replace put-superseded bodies and re-check the predicate, keep the // rest untouched. rows.retain_mut(|(row_key, body)| { - let Some(surrogate) = - crate::engine::document::store::doc_id_to_surrogate(row_key).map(|s| s.as_u32()) - else { - return true; - }; + let surrogate = row_key.surrogate().as_u32(); match overlay.get(coll_key, surrogate) { Some(Staged::Tombstone) => false, Some(Staged::Put(staged_body)) => { @@ -129,9 +120,9 @@ impl CoreLoop { } match staged { Staged::Put(body) => { - let hex_id = surrogate_to_doc_id(Surrogate::new(surrogate)); - if matches(&hex_id, body) { - rows.push((hex_id, body.clone())); + let key = StorageKey::for_surrogate(Surrogate::new(surrogate)); + if matches(&key, body) { + rows.push((key, body.clone())); seen.insert(surrogate); } } @@ -243,19 +234,18 @@ impl CoreLoop { /// own current-version + tombstone-aware semantics. /// /// Identity note: `DocumentEngine::index_lookup` returns each match's - /// storage key, which is the hex surrogate (`surrogate_to_doc_id`), NOT - /// the user-visible primary key — the secondary index stores the row's - /// hex-surrogate storage key as its document_id component. So this path - /// keys the overlay by surrogate exactly like `merge_overlay_into_scan`: - /// parse each base doc_id as hex to a surrogate, consult - /// `overlay.get(coll_key, surrogate)`, and append additions as - /// `surrogate_to_doc_id(..)` hex — the same identity the base list and the - /// handler's body fetch use. (The overlay's `doc_id_to_surrogate` map is - /// keyed by the PK, so `get_by_doc_id` would never match a hex key here.) + /// `StorageKey`, NOT the user-visible primary key — the secondary index + /// stores the row's storage key as its document_id component. So this + /// path keys the overlay by surrogate exactly like + /// `merge_overlay_into_scan`: read each base `StorageKey`'s surrogate, + /// consult `overlay.get(coll_key, surrogate)`, and append additions as + /// `StorageKey`s — the same identity the base list and the handler's + /// body fetch use. (The overlay's `doc_id_to_surrogate` map is keyed by + /// the PK, so `get_by_doc_id` would never match a storage key here.) pub(in crate::data::executor) fn merge_overlay_into_index_lookup( &self, params: IndexOverlayMergeParams<'_>, - doc_ids: &mut Vec, + doc_ids: &mut Vec, decode: &dyn Fn(&[u8]) -> crate::Result>, ) -> crate::Result<()> { let IndexOverlayMergeParams { @@ -315,11 +305,11 @@ impl CoreLoop { // passes finish. let predicate_err: std::cell::Cell> = std::cell::Cell::new(None); - let residual_matches = |doc_id: &str, body: &[u8]| -> bool { + let residual_matches = |row_key: &StorageKey, body: &[u8]| -> bool { if residual.is_empty() { return true; } - match matches_with_resolved_schema(strict_schema, residual, doc_id, body) { + match matches_with_resolved_schema(strict_schema, residual, row_key, body) { Ok(b) => b, Err(e) => { predicate_err.set(Some(e)); @@ -328,26 +318,17 @@ impl CoreLoop { } }; - // Base doc IDs are hex surrogates; track their surrogates so additions - // don't re-append a row the base index lookup already returned. - let mut seen: HashSet = doc_ids - .iter() - .filter_map(|id| { - crate::engine::document::store::doc_id_to_surrogate(id).map(|s| s.as_u32()) - }) - .collect(); + // Base doc IDs' surrogates, so additions don't re-append a row the + // base index lookup already returned. + let mut seen: HashSet = doc_ids.iter().map(|id| id.surrogate().as_u32()).collect(); - // Base-minus-superseded: resolve each base hex doc_id to its surrogate - // and consult the overlay. A tombstone drops it; a staged put re-checks - // whether the new body still equals the lookup value (an update may - // have moved the row off the indexed value); no overlay entry — or an - // unparseable key — keeps it as-is. + // Base-minus-superseded: resolve each base storage key to its + // surrogate and consult the overlay. A tombstone drops it; a staged + // put re-checks whether the new body still equals the lookup value + // (an update may have moved the row off the indexed value); no + // overlay entry keeps it as-is. doc_ids.retain(|doc_id| { - let Some(surrogate) = - crate::engine::document::store::doc_id_to_surrogate(doc_id).map(|s| s.as_u32()) - else { - return true; - }; + let surrogate = doc_id.surrogate().as_u32(); match overlay.get(coll_key, surrogate) { Some(Staged::Tombstone) => false, Some(Staged::Put(body)) => value_matches(body) && residual_matches(doc_id, body), @@ -356,18 +337,17 @@ impl CoreLoop { }); // Overlay additions: staged puts for surrogates the base lookup did - // not return, appended as hex `surrogate_to_doc_id(..)` when their - // staged body matches the lookup value — this surfaces a staged insert - // or an update-into-the-value. + // not return, appended when their staged body matches the lookup + // value — this surfaces a staged insert or an update-into-the-value. for (surrogate, staged) in overlay.iter_for_collection(coll_key) { if seen.contains(&surrogate) { continue; } match staged { Staged::Put(body) => { - let hex_id = surrogate_to_doc_id(Surrogate::new(surrogate)); - if value_matches(body) && residual_matches(&hex_id, body) { - doc_ids.push(hex_id); + let key = StorageKey::for_surrogate(Surrogate::new(surrogate)); + if value_matches(body) && residual_matches(&key, body) { + doc_ids.push(key); seen.insert(surrogate); } } @@ -390,26 +370,24 @@ impl CoreLoop { /// a row that was added by the merge (a staged insert/update) or whose /// base body was superseded by a staged update has no correct body in base /// storage — the staged `Put` bytes are the only current representation. - /// `doc_id` is the hex-surrogate storage key the index lookup returned, so - /// the overlay is consulted by surrogate (`get`), matching the identity - /// the merge used — `get_by_doc_id` is keyed by the PK and would not match. + /// `doc_id` is the storage key the index lookup returned, so the overlay + /// is consulted by surrogate (`get`), matching the identity the merge + /// used — `get_by_doc_id` is keyed by the PK and would not match. /// Returns `None` for a staged tombstone. Falls back to the lazy `base` /// closure (skipped whenever the overlay already has the answer) when the - /// key is unparseable or the surrogate has no staged mutation. + /// surrogate has no staged mutation. pub(in crate::data::executor) fn overlay_or_base_body( &self, txn_id: Option, coll_key: &(DatabaseId, TenantId, String), - doc_id: &str, + doc_id: &StorageKey, base: impl FnOnce() -> crate::Result>>, ) -> crate::Result>> { if let Some(txn_id) = txn_id { // Read-your-own-writes refreshes the lease (see the reaper). self.touch_overlay(txn_id); - if let Some(overlay) = self.txn_overlays.get(&txn_id) - && let Some(surrogate) = - crate::engine::document::store::doc_id_to_surrogate(doc_id).map(|s| s.as_u32()) - { + if let Some(overlay) = self.txn_overlays.get(&txn_id) { + let surrogate = doc_id.surrogate().as_u32(); match overlay.get(coll_key, surrogate) { Some(Staged::Put(body)) => return Ok(Some(body.clone())), Some(Staged::Tombstone) => return Ok(None), diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_delete.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_delete.rs index 402576d56..59e1455ed 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_delete.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_delete.rs @@ -12,6 +12,8 @@ //! indexes, graph edges) run only at COMMIT replay through the real apply //! path, exactly as a staged point delete defers its cascade today. +use nodedb_types::StorageKey; + use crate::bridge::envelope::{ErrorCode, Response}; use crate::bridge::scan_filter::ScanFilter; use crate::data::executor::core_loop::CoreLoop; @@ -85,14 +87,14 @@ impl CoreLoop { { // `merge_overlay_into_scan` takes an infallible - // `Fn(&str, &[u8]) -> bool` predicate, so a division/modulo-by- - // zero is captured via this `Cell` side-channel and checked once + // `Fn(&StorageKey, &[u8]) -> bool` predicate, so a division/modulo- + // by-zero is captured via this `Cell` side-channel and checked once // the merge returns. let raw_matches = self.strict_aware_matcher(database_id.as_u64(), tid, collection, &filters); let predicate_err: std::cell::Cell> = std::cell::Cell::new(None); - let matches = |doc_id: &str, body: &[u8]| match raw_matches(doc_id, body) { + let matches = |row_key: &StorageKey, body: &[u8]| match raw_matches(row_key, body) { Ok(b) => b, Err(e) => { predicate_err.set(Some(e)); @@ -115,7 +117,7 @@ impl CoreLoop { nodedb_types::WriteGateDecision::AdmitAll ) { for (row_key, body) in &rows { - let identity = crate::engine::document::store::identity_of(row_key); + let identity = row_key.to_identity(); if let Err(e) = self.stage_admit_write( rls_write_check, body, @@ -131,11 +133,12 @@ impl CoreLoop { let mut affected = 0u64; for (row_key, _body) in &rows { - let Ok(surrogate) = u32::from_str_radix(row_key, 16) else { - continue; - }; - self.txn_overlay_mut(txn_id) - .insert_tombstone(coll_key.clone(), surrogate, row_key); + let surrogate = row_key.surrogate().as_u32(); + self.txn_overlay_mut(txn_id).insert_tombstone( + coll_key.clone(), + surrogate, + &row_key.to_string(), + ); affected += 1; } diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs index 2fd72344c..f98d1162e 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs @@ -14,6 +14,7 @@ //! replay remains the sole durable apply. use nodedb_physical::physical_plan::UpdateValue; +use nodedb_types::StorageKey; use crate::bridge::envelope::{ErrorCode, Response}; use crate::bridge::scan_filter::ScanFilter; @@ -109,14 +110,14 @@ impl CoreLoop { // appends overlay-only rows that now match. { // `merge_overlay_into_scan` takes an infallible - // `Fn(&str, &[u8]) -> bool` predicate, so a division/modulo-by- - // zero is captured via this `Cell` side-channel and checked once + // `Fn(&StorageKey, &[u8]) -> bool` predicate, so a division/modulo- + // by-zero is captured via this `Cell` side-channel and checked once // the merge returns. let raw_matches = self.strict_aware_matcher(database_id.as_u64(), tid, collection, &filters); let predicate_err: std::cell::Cell> = std::cell::Cell::new(None); - let matches = |doc_id: &str, body: &[u8]| match raw_matches(doc_id, body) { + let matches = |row_key: &StorageKey, body: &[u8]| match raw_matches(row_key, body) { Ok(b) => b, Err(e) => { predicate_err.set(Some(e)); @@ -131,9 +132,7 @@ impl CoreLoop { let mut affected = 0u64; for (row_key, current_body) in &rows { - let Ok(surrogate) = u32::from_str_radix(row_key, 16) else { - continue; - }; + let surrogate = row_key.surrogate().as_u32(); let new_body = match self.stage_apply_update( database_id.as_u64(), tid, @@ -149,7 +148,7 @@ impl CoreLoop { // policy. A rejected row fails the statement rather than being // skipped: skipping would under-report `affected` while the rest of // the predicate's matches were still rewritten. - let identity = crate::engine::document::store::identity_of(row_key); + let identity = row_key.to_identity(); if let Err(e) = self.stage_admit_write( rls_write_check, &new_body, @@ -160,9 +159,13 @@ impl CoreLoop { ) { return self.response_error(task, e); } - if let Err(e) = - self.stage_bulk_put_capped(txn_id, &coll_key, surrogate, row_key, new_body) - { + if let Err(e) = self.stage_bulk_put_capped( + txn_id, + &coll_key, + surrogate, + &row_key.to_string(), + new_body, + ) { return self.response_error(task, e); } affected += 1; @@ -191,14 +194,14 @@ impl CoreLoop { tid: u64, collection: &str, filters: &[ScanFilter], - ) -> Result)>, Response> { + ) -> Result)>, Response> { let matching_ids = self .scan_matching_documents(database_id, tid, collection, filters) .map_err(|e| self.response_error(task, e))?; - let mut rows: Vec<(String, Vec)> = Vec::with_capacity(matching_ids.len()); + let mut rows: Vec<(StorageKey, Vec)> = Vec::with_capacity(matching_ids.len()); for key in matching_ids { if let Ok(Some(bytes)) = self.sparse.get(database_id, tid, collection, &key) { - rows.push((key.to_string(), bytes)); + rows.push((key, bytes)); } } Ok(rows) diff --git a/nodedb/src/data/executor/handlers/update_from_join_collect.rs b/nodedb/src/data/executor/handlers/update_from_join_collect.rs index 2a33db855..6e6382879 100644 --- a/nodedb/src/data/executor/handlers/update_from_join_collect.rs +++ b/nodedb/src/data/executor/handlers/update_from_join_collect.rs @@ -14,6 +14,7 @@ use redb::{ReadableDatabase, ReadableTable}; use std::collections::HashMap; +use nodedb_types::StorageKey; use nodedb_types::columnar::StrictSchema; use crate::bridge::scan_filter::ScanFilter; @@ -22,6 +23,7 @@ use crate::data::executor::core_loop::filter_match::matches_with_resolved_schema use crate::data::executor::doc_format; use crate::data::executor::handlers::update_from_join_source_map::json_value_to_string; use crate::data::executor::task::ExecutionTask; +use crate::engine::sparse::btree::{KeyedTable, invalid_storage_key_err}; use crate::types::{DatabaseId, TenantId, TxnId}; use nodedb_physical::physical_plan::UpdateValue; @@ -104,7 +106,7 @@ impl CoreLoop { })?; let mut rows: Vec = Vec::new(); - for (doc_id, current_bytes) in target_rows { + for (key, current_bytes) in target_rows { // A row the statement matched but cannot decode fails the // statement. Skipping it leaves the row untouched under a smaller // affected count that reports success. @@ -113,10 +115,10 @@ impl CoreLoop { .ok_or_else(|| { crate::diag::strict_row_undecodable( target_collection, - &doc_id, + &key.to_string(), "update_from_join_collect", ); - let identity = crate::engine::document::store::identity_of(&doc_id); + let identity = key.to_identity(); super::super::strict_format::undecodable_strict_row( target_collection, identity.as_str(), @@ -158,7 +160,7 @@ impl CoreLoop { .map_err(|e| crate::Error::Serialization { format: "msgpack".into(), detail: format!( - "literal assigned to \"{field}\" for document \"{doc_id}\" \ + "literal assigned to \"{field}\" for document \"{key}\" \ of collection \"{target_collection}\" does not decode: {e}" ), })?, @@ -212,13 +214,12 @@ impl CoreLoop { doc_format::encode_to_msgpack(&target_doc) }; - // The storage key is the hex-encoded surrogate on a surrogate-keyed - // row; parse it once here for the reindex + write-set (write path) - // and the expanded `PointPut`'s identity (RESOLVE path). - let surrogate = crate::engine::document::store::doc_id_to_surrogate(&doc_id); + // The storage key came typed off the target scan; the write-set + // reindex and the expanded `PointPut`'s identity (RESOLVE path) + // both need it, so it's carried through rather than re-parsed. rows.push(ResolvedUpdateRow { - doc_id, - surrogate, + doc_id: key.to_string(), + surrogate: Some(key.surrogate()), body: updated_bytes, old_body: current_bytes, doc: target_doc, @@ -239,7 +240,10 @@ impl CoreLoop { /// strict-aware matcher), and a staged put absent from base is appended when /// it passes the filters. `None` (autocommit) returns the base-filtered rows /// unchanged — byte-identical to the pre-staging behavior. - fn scan_target_rows(&self, args: ScanTargetRows<'_>) -> crate::Result)>> { + fn scan_target_rows( + &self, + args: ScanTargetRows<'_>, + ) -> crate::Result)>> { let ScanTargetRows { database_id, tid, @@ -267,26 +271,25 @@ impl CoreLoop { detail: format!("open table: {e}"), })?; - let mut rows: Vec<(String, Vec)> = Vec::new(); + let mut rows: Vec<(StorageKey, Vec)> = Vec::new(); if let Ok(range) = table.range(prefix.as_str()..end.as_str()) { for entry in range.flatten() { - let key = entry.0.value(); + let full_key = entry.0.value(); let value_bytes = entry.1.value(); - let Some(doc_id) = key.strip_prefix(&prefix) else { + let Some(rest) = full_key.strip_prefix(&prefix) else { continue; }; + let key = StorageKey::parse(rest).ok_or_else(|| { + invalid_storage_key_err(KeyedTable::Documents, target_collection, rest) + })?; // Goes through the same primitive the overlay half below uses, // so a schemaless row with no `id` field matches `WHERE id // ...` here exactly as it does once staged. - let matches = matches_with_resolved_schema( - strict_schema, - target_filters, - doc_id, - value_bytes, - ) - .map_err(crate::Error::from)?; + let matches = + matches_with_resolved_schema(strict_schema, target_filters, &key, value_bytes) + .map_err(crate::Error::from)?; if matches { - rows.push((doc_id.to_string(), value_bytes.to_vec())); + rows.push((key, value_bytes.to_vec())); } } } @@ -299,14 +302,14 @@ impl CoreLoop { // dropped, exactly as for a base row. if let Some(txn_id) = txn_id { // `merge_overlay_into_scan` takes an infallible - // `Fn(&str, &[u8]) -> bool` predicate, so a division/modulo-by- - // zero is captured via this `Cell` side-channel and checked once + // `Fn(&StorageKey, &[u8]) -> bool` predicate, so a division/modulo- + // by-zero is captured via this `Cell` side-channel and checked once // the merge returns. let raw_matches = self.strict_aware_matcher(database_id, tid, target_collection, target_filters); let predicate_err: std::cell::Cell> = std::cell::Cell::new(None); - let matches = |doc_id: &str, body: &[u8]| match raw_matches(doc_id, body) { + let matches = |row_key: &StorageKey, body: &[u8]| match raw_matches(row_key, body) { Ok(b) => b, Err(e) => { predicate_err.set(Some(e)); diff --git a/nodedb/src/engine/sparse/btree_scan.rs b/nodedb/src/engine/sparse/btree_scan.rs index 2f8ecf287..eba450258 100644 --- a/nodedb/src/engine/sparse/btree_scan.rs +++ b/nodedb/src/engine/sparse/btree_scan.rs @@ -238,7 +238,7 @@ impl SparseEngine { tenant_id: u64, collection: &str, field: &str, - doc_ids: &std::collections::HashSet, + doc_ids: &std::collections::HashSet, ) -> crate::Result> { let prefix = format!( "{}{field}:", @@ -264,7 +264,10 @@ impl SparseEngine { { let value = &rest[..colon_pos]; let doc_id = &rest[colon_pos + 1..]; - if doc_ids.contains(doc_id) { + let doc_key = StorageKey::parse(doc_id).ok_or_else(|| { + invalid_storage_key_err(KeyedTable::Indexes, collection, doc_id) + })?; + if doc_ids.contains(&doc_key) { *groups.entry(value.to_string()).or_default() += 1; } } From 4bfffbcef5a3f5b3627db83a890151b469d76fef Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 12 Sep 2026 12:15:03 +0800 Subject: [PATCH 12/17] fix(overlay): key the transaction overlay by row identity, not doc_id The overlay's insert/tombstone/TTL paths and every stage_write caller keyed staged rows by a raw string doc_id, so a bulk write and a point get for the same row could land on different overlay slots whenever the row carried a declared or default identity column instead of its surrogate. Overlay lookups, undo journal entries, and doc-id iteration now key by the typed RowIdentity everywhere. Add RowIdentity::of_stored_row / extract_pk_value / value_to_pk_string in nodedb-types::row_identity as the single place that derives a stored row's identity from its body, the DDL's declared primary key, and its storage key, replacing the equivalent private helper that lived in target_identity::pk. Calvin's bulk delete and bulk update overlay staging, KV TTL staging, and every DML stage_write path now derive and thread this identity instead of falling back to the storage key's decimal surrogate or the doc_id's string form. --- .../src/physical_plan/document/op.rs | 4 + nodedb-types/src/lib.rs | 3 +- nodedb-types/src/row_identity.rs | 107 +++++++++++++- .../calvin/tx_class/dependent_builder.rs | 1 + .../control/planner/rls_injection/document.rs | 1 + .../rls_injection/permission_tree/document.rs | 1 + .../sql_plan_convert/dml/insert/schema.rs | 5 +- .../dml/update_delete/delete.rs | 6 + .../native/dispatch/plan_builder/document.rs | 2 + .../server/shared/sql/staging_predicates.rs | 1 + .../predicate/txn_buffering/classify.rs | 3 + .../server/wal_dispatch/write_set_redo.rs | 1 + .../control/target_identity/document_id.rs | 4 +- nodedb/src/control/target_identity/pk.rs | 25 +--- .../src/control/target_identity/surrogate.rs | 4 +- .../wal_replication/decode/document.rs | 29 ++-- .../wal_replication/encode/document.rs | 3 +- .../wal_replication/encode/entry_document.rs | 4 + nodedb/src/data/executor/dispatch/document.rs | 3 + .../data/executor/handlers/control/calvin.rs | 1 + .../handlers/control/calvin_overlay_stage.rs | 21 +-- .../control/calvin_overlay_stage_bulk.rs | 89 ++++++++---- .../handlers/control/calvin_resolve.rs | 1 + .../handlers/document/resolve/dispatch.rs | 3 + .../src/data/executor/handlers/kv/crud/get.rs | 4 +- nodedb/src/data/executor/handlers/kv/ttl.rs | 4 +- .../src/data/executor/handlers/point/get.rs | 12 +- .../executor/handlers/point/overlay_lookup.rs | 18 ++- .../handlers/transaction/overlay/merge.rs | 18 +-- .../handlers/transaction/overlay/staged.rs | 136 ++++++++++++------ .../handlers/transaction/resolve/document.rs | 26 ++-- .../handlers/transaction/resolve/entry.rs | 107 +++++++++----- .../handlers/transaction/resolve/kv.rs | 15 +- .../handlers/transaction/stage_write/body.rs | 24 +++- .../transaction/stage_write/context.rs | 18 +-- .../transaction/stage_write/dispatch.rs | 54 +++++-- .../handlers/transaction/stage_write/mod.rs | 3 +- .../stage_write/stage_bulk_delete.rs | 35 +++-- .../stage_write/stage_bulk_update.rs | 23 +-- .../transaction/stage_write/stage_columnar.rs | 14 +- .../stage_write/stage_columnar_dml.rs | 58 +++++--- .../stage_columnar_resolved_dml.rs | 46 ++++-- .../transaction/stage_write/stage_kv.rs | 36 +++-- .../stage_write/stage_kv_atomic.rs | 6 +- .../stage_write/stage_kv_conflict.rs | 2 +- .../stage_write/stage_kv_delete.rs | 8 +- .../transaction/stage_write/stage_kv_ttl.rs | 2 +- .../transaction/stage_write/stage_spatial.rs | 21 ++- .../stage_write/stage_timeseries.rs | 62 ++++++-- .../transaction/stage_write/stage_upsert.rs | 2 +- .../executor_tests/test_ollp_verification.rs | 2 + .../tests/inproc/cases/trigger_execution.rs | 1 + 52 files changed, 757 insertions(+), 322 deletions(-) diff --git a/nodedb-physical/src/physical_plan/document/op.rs b/nodedb-physical/src/physical_plan/document/op.rs index d3c511440..3cccfa423 100644 --- a/nodedb-physical/src/physical_plan/document/op.rs +++ b/nodedb-physical/src/physical_plan/document/op.rs @@ -473,6 +473,10 @@ pub enum DocumentOp { /// See `PointPut::resolved_sum_targets`. #[serde(default)] resolved_sum_targets: Vec, + /// See `PointUpdate::declared_primary_key`. Names the column each + /// removed row's identity is read from when staged. + #[serde(default)] + declared_primary_key: Option, }, /// MERGE: join-based multi-action DML (INSERT/UPDATE/DELETE per WHEN diff --git a/nodedb-types/src/lib.rs b/nodedb-types/src/lib.rs index 2c8587839..8e71fca06 100644 --- a/nodedb-types/src/lib.rs +++ b/nodedb-types/src/lib.rs @@ -120,7 +120,8 @@ pub use quota::{ pub use result::{QueryResult, SearchResult, SubGraph}; pub use rls_write_check::{RlsWriteCheck, WriteGateDecision}; pub use row_identity::{ - RowIdentity, StorageKey, doc_id_to_surrogate, identity_of, surrogate_to_doc_id, + DEFAULT_IDENTITY_COLUMN, RowIdentity, StorageKey, doc_id_to_surrogate, extract_pk_value, + identity_of, surrogate_to_doc_id, value_to_pk_string, }; pub use sparse_vector::{SparseVector, SparseVectorError}; pub use sql_quote::{quote_ident, quote_literal}; diff --git a/nodedb-types/src/row_identity.rs b/nodedb-types/src/row_identity.rs index 0cff13b1d..5e24c23d1 100644 --- a/nodedb-types/src/row_identity.rs +++ b/nodedb-types/src/row_identity.rs @@ -23,7 +23,16 @@ //! 8-hex-character shape (a primary key `deadbeef` parses as a storage key), //! so a loose `String` cannot tell them apart. `StorageKey::parse` is the //! only place that shape gets reinterpreted as a surrogate. -use crate::Surrogate; +//! +//! The identity rule INSERT applies lives here too, so both planes derive a +//! stored row's identity the same way: [`DEFAULT_IDENTITY_COLUMN`], +//! [`extract_pk_value`], [`value_to_pk_string`], and +//! [`RowIdentity::of_stored_row`]. +use crate::{Surrogate, Value}; + +/// The column that carries a row's identity when the DDL declares no +/// `PRIMARY KEY`. INSERT and every stored-row identity derivation use it. +pub const DEFAULT_IDENTITY_COLUMN: &str = "id"; /// The redb key a document row is stored under. /// @@ -83,7 +92,7 @@ impl std::fmt::Display for StorageKey { /// /// A minted row renders its surrogate in decimal. A row with a declared /// `PRIMARY KEY` carries the user's own value. -#[derive(Debug, Clone, PartialEq, Eq)] +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash)] pub struct RowIdentity(String); impl RowIdentity { @@ -92,6 +101,22 @@ impl RowIdentity { Self(surrogate.as_u32().to_string()) } + /// The identity INSERT mints for a row, applied to a stored body. + /// + /// The identity column is `declared_primary_key`, else + /// [`DEFAULT_IDENTITY_COLUMN`]. A body carrying that column yields its + /// value. A body without it yields the decimal surrogate of `key`. + pub fn of_stored_row( + body: &[u8], + declared_primary_key: Option<&str>, + key: StorageKey, + ) -> RowIdentity { + let column = declared_primary_key.unwrap_or(DEFAULT_IDENTITY_COLUMN); + extract_pk_value(body, column) + .map(RowIdentity::from_user_key) + .unwrap_or_else(|| key.to_identity()) + } + /// Wrap a declared or client-supplied key as the row's identity. /// /// The value is taken verbatim and never interpreted as a storage key @@ -154,10 +179,88 @@ pub fn identity_of(doc_id: &str) -> RowIdentity { .unwrap_or_else(|| RowIdentity::from_user_key(doc_id)) } +/// Extract the stringified value of `field` from a MessagePack row body. +/// +/// Returns `None` when the body is not an object, lacks `field`, or the +/// value has no primary-key string form. +pub fn extract_pk_value(body: &[u8], field: &str) -> Option { + let Value::Object(obj) = crate::value_from_msgpack(body).ok()? else { + return None; + }; + value_to_pk_string(obj.get(field)?) +} + +/// Stringify a scalar value into its primary-key form. +/// +/// Matches the `sql_value_to_string` convention of the INSERT identity +/// path. Non-scalar values have no primary-key form and yield `None`. +pub fn value_to_pk_string(v: &Value) -> Option { + match v { + Value::String(s) => Some(s.clone()), + Value::Integer(n) => Some(n.to_string()), + Value::Float(f) => Some(f.to_string()), + Value::Bool(b) => Some(b.to_string()), + Value::Decimal(d) => Some(d.to_string()), + _ => None, + } +} + #[cfg(test)] mod tests { use super::*; + fn body(fields: &[(&str, Value)]) -> Vec { + let mut obj = std::collections::HashMap::new(); + for (name, value) in fields { + obj.insert((*name).to_string(), value.clone()); + } + crate::value_to_msgpack(&Value::Object(obj)).expect("encode msgpack") + } + + #[test] + fn of_stored_row_uses_declared_primary_key() { + let key = StorageKey::for_surrogate(Surrogate::new(9)); + let body = body(&[ + ("id", Value::String("ignored".into())), + ("sku", Value::Integer(42)), + ]); + assert_eq!( + RowIdentity::of_stored_row(&body, Some("sku"), key).as_str(), + "42" + ); + } + + #[test] + fn of_stored_row_falls_back_to_id_column() { + let key = StorageKey::for_surrogate(Surrogate::new(9)); + let body = body(&[("id", Value::String("user-1".into()))]); + assert_eq!( + RowIdentity::of_stored_row(&body, None, key).as_str(), + "user-1" + ); + } + + #[test] + fn of_stored_row_without_identity_column_is_decimal_surrogate() { + let key = StorageKey::for_surrogate(Surrogate::new(9)); + let body = body(&[("name", Value::String("x".into()))]); + assert_eq!(RowIdentity::of_stored_row(&body, None, key).as_str(), "9"); + assert_eq!( + RowIdentity::of_stored_row(&body, Some("sku"), key).as_str(), + "9" + ); + } + + #[test] + fn value_to_pk_string_rejects_non_scalars() { + assert_eq!(value_to_pk_string(&Value::Array(Vec::new())), None); + assert_eq!(value_to_pk_string(&Value::Null), None); + assert_eq!( + value_to_pk_string(&Value::Bool(true)).as_deref(), + Some("true") + ); + } + #[test] fn formats_zero_padded_lowercase() { assert_eq!( diff --git a/nodedb/src/control/planner/calvin/tx_class/dependent_builder.rs b/nodedb/src/control/planner/calvin/tx_class/dependent_builder.rs index 03c3f72e8..f906f561b 100644 --- a/nodedb/src/control/planner/calvin/tx_class/dependent_builder.rs +++ b/nodedb/src/control/planner/calvin/tx_class/dependent_builder.rs @@ -274,6 +274,7 @@ mod tests { rls_filters: vec![], rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), resolved_sum_targets: Vec::new(), + declared_primary_key: None, }), post_set_op: nodedb_physical::physical_task::PostSetOp::None, txn_id: None, diff --git a/nodedb/src/control/planner/rls_injection/document.rs b/nodedb/src/control/planner/rls_injection/document.rs index 3533a51a9..7593de0a4 100644 --- a/nodedb/src/control/planner/rls_injection/document.rs +++ b/nodedb/src/control/planner/rls_injection/document.rs @@ -361,6 +361,7 @@ mod tests { rls_filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), resolved_sum_targets: Vec::new(), + declared_primary_key: None, }); assert!(inject(&mut plan, &store).is_ok()); assert!(write_check(&plan).has_predicate()); diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/document.rs b/nodedb/src/control/planner/rls_injection/permission_tree/document.rs index 49ad4b100..260e7f154 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/document.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/document.rs @@ -227,6 +227,7 @@ mod tests { rls_filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), resolved_sum_targets: Vec::new(), + declared_primary_key: None, }); assert!(apply(&mut plan, &cache).is_ok()); match &plan { diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert/schema.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert/schema.rs index a132779a8..81b236e94 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/insert/schema.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert/schema.rs @@ -1,5 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 +pub(crate) use nodedb_types::DEFAULT_IDENTITY_COLUMN; use nodedb_types::columnar::{ColumnDef, ColumnType, ColumnarSchema}; /// Build a `ColumnarSchema` from raw catalog column-type strings. @@ -23,10 +24,6 @@ use nodedb_types::columnar::{ColumnDef, ColumnType, ColumnarSchema}; /// `bootstrap::data_plane::load_columnar_schema_seed`, which pre-registers /// each columnar-family collection's real schema before WAL replay so a /// fresh `MutationEngine` never falls back to type-lossy inference. -/// The identity column a columnar-family collection carries when its DDL -/// declares no `PRIMARY KEY`. The planner resolves the same name. -pub(crate) const DEFAULT_IDENTITY_COLUMN: &str = "id"; - pub(crate) fn build_columnar_schema( column_schema: &[(String, String)], identity_column: &str, diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs index 073e58865..6a92304b8 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs @@ -115,6 +115,10 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_delete( }); } + // A predicate delete stages each removed row under its client identity, + // read from the declared `PRIMARY KEY` column when the DDL names one. + let declared_primary_key = super::super::declared_primary_key_name(ctx, collection)?; + // Edge-bearing gate: a PK-equality delete on a collection with implicit // edges must not lower to a static `PointDelete` — that bypasses OLLP // and leaks the edge. Route as `BulkDelete` instead so the edge-bearing @@ -138,6 +142,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_delete( // Filled in by the materialized-sum resolution pass, which // recon-scans the rows this predicate matches. resolved_sum_targets: Vec::new(), + declared_primary_key, }), post_set_op: PostSetOp::None, txn_id: None, @@ -201,6 +206,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_delete( // Filled in by the materialized-sum resolution pass, which // recon-scans the rows this predicate matches. resolved_sum_targets: Vec::new(), + declared_primary_key, }), post_set_op: PostSetOp::None, txn_id: None, diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/document.rs b/nodedb/src/control/server/native/dispatch/plan_builder/document.rs index da7ecdcc5..d70e66853 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/document.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/document.rs @@ -415,6 +415,8 @@ pub(crate) fn build_bulk_delete( rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), // Filled in by the materialized-sum resolution pass. resolved_sum_targets: Vec::new(), + // See `build_update`: reads the declared PRIMARY KEY from the catalog. + declared_primary_key: declared_primary_key(ctx, collection)?, })) } diff --git a/nodedb/src/control/server/shared/sql/staging_predicates.rs b/nodedb/src/control/server/shared/sql/staging_predicates.rs index c2230d091..1f8aea47e 100644 --- a/nodedb/src/control/server/shared/sql/staging_predicates.rs +++ b/nodedb/src/control/server/shared/sql/staging_predicates.rs @@ -328,6 +328,7 @@ mod tests { rls_filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), resolved_sum_targets: Vec::new(), + declared_primary_key: None, }); assert!(is_stageable_write(&bulk_delete)); assert_eq!(staged_tag_kind(&bulk_delete, &[]), StagedTagKind::Delete); diff --git a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs index c54cd5c71..57f421195 100644 --- a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs +++ b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs @@ -632,6 +632,7 @@ mod tests { rls_filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, resolved_sum_targets: Vec::new(), + declared_primary_key: None, }), // BulkUpdate / BulkDelete: OLLP surrogate set present — the // Calvin-routed, not-buffered case. @@ -656,6 +657,7 @@ mod tests { rls_filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, resolved_sum_targets: Vec::new(), + declared_primary_key: None, }), // BulkUpdate / BulkDelete: OLLP edge set present, surrogates None — // the other half of the `Some` guard. @@ -690,6 +692,7 @@ mod tests { rls_filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, resolved_sum_targets: Vec::new(), + declared_primary_key: None, }), PhysicalPlan::Document(DocumentOp::MaterializeScan { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), diff --git a/nodedb/src/control/server/wal_dispatch/write_set_redo.rs b/nodedb/src/control/server/wal_dispatch/write_set_redo.rs index a0dfc1b97..1138cc9fd 100644 --- a/nodedb/src/control/server/wal_dispatch/write_set_redo.rs +++ b/nodedb/src/control/server/wal_dispatch/write_set_redo.rs @@ -206,6 +206,7 @@ mod tests { rls_filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), resolved_sum_targets: Vec::new(), + declared_primary_key: None, }); assert_eq!(plan_post_apply_redo(&plan).as_deref(), Some("docs")); } diff --git a/nodedb/src/control/target_identity/document_id.rs b/nodedb/src/control/target_identity/document_id.rs index 347ff6d26..8198c5c22 100644 --- a/nodedb/src/control/target_identity/document_id.rs +++ b/nodedb/src/control/target_identity/document_id.rs @@ -4,9 +4,9 @@ //! collection on behalf of another operation, and validate that a matched //! existing row carries a registered surrogate. -use nodedb_types::Surrogate; +use nodedb_types::{Surrogate, extract_pk_value}; -use super::pk::{TargetPk, extract_pk_value}; +use super::pk::TargetPk; use crate::engine::document::store::RowIdentity; /// The user-visible primary key (`document_id`) for a row written on this diff --git a/nodedb/src/control/target_identity/pk.rs b/nodedb/src/control/target_identity/pk.rs index dfda0591e..b1267e03c 100644 --- a/nodedb/src/control/target_identity/pk.rs +++ b/nodedb/src/control/target_identity/pk.rs @@ -4,8 +4,8 @@ //! assignment for a row written on its behalf (mirrors the plain-`INSERT` //! identity path). +use nodedb_types::CollectionType; use nodedb_types::columnar::DocumentMode; -use nodedb_types::{CollectionType, Value}; use crate::control::security::catalog::StoredCollection; @@ -48,7 +48,7 @@ pub(crate) fn resolve_target_pk( name: target .declared_primary_key .clone() - .unwrap_or_else(|| "id".to_string()), + .unwrap_or_else(|| nodedb_types::DEFAULT_IDENTITY_COLUMN.to_string()), declared: target.declared_primary_key.is_some(), }), CollectionType::KeyValue(_) | CollectionType::Columnar(_) => Err(crate::Error::PlanError { @@ -59,24 +59,3 @@ pub(crate) fn resolve_target_pk( }), } } - -/// Extract a stringified primary-key value from a MessagePack row body. -pub(super) fn extract_pk_value(body: &[u8], field: &str) -> Option { - let Value::Object(obj) = nodedb_types::value_from_msgpack(body).ok()? else { - return None; - }; - value_to_pk_string(obj.get(field)?) -} - -/// Stringify a scalar value into its primary-key byte form (mirrors the -/// `sql_value_to_string` convention used by the plain-INSERT identity path). -fn value_to_pk_string(v: &Value) -> Option { - match v { - Value::String(s) => Some(s.clone()), - Value::Integer(n) => Some(n.to_string()), - Value::Float(f) => Some(f.to_string()), - Value::Bool(b) => Some(b.to_string()), - Value::Decimal(d) => Some(d.to_string()), - _ => None, - } -} diff --git a/nodedb/src/control/target_identity/surrogate.rs b/nodedb/src/control/target_identity/surrogate.rs index 35e95f33b..f77b9894e 100644 --- a/nodedb/src/control/target_identity/surrogate.rs +++ b/nodedb/src/control/target_identity/surrogate.rs @@ -3,9 +3,9 @@ //! Assign a fresh, catalog-registered surrogate for a row written into a //! target collection on behalf of another operation. -use nodedb_types::{DatabaseId, Surrogate, TenantId}; +use nodedb_types::{DatabaseId, Surrogate, TenantId, extract_pk_value}; -use super::pk::{TargetPk, extract_pk_value}; +use super::pk::TargetPk; use crate::control::state::SharedState; /// Assign a fresh, registered surrogate for one written row on the TARGET's diff --git a/nodedb/src/control/wal_replication/decode/document.rs b/nodedb/src/control/wal_replication/decode/document.rs index 840f03412..9a64ce420 100644 --- a/nodedb/src/control/wal_replication/decode/document.rs +++ b/nodedb/src/control/wal_replication/decode/document.rs @@ -356,6 +356,8 @@ pub(super) fn bulk_dml( // No predicate on replay — see `point_delete`. rls_write_check: nodedb_types::RlsWriteCheck::already_decided_elsewhere(), resolved_sum_targets, + // Read off the record — see `point_update`. + declared_primary_key, }) } } @@ -586,6 +588,7 @@ mod tests { "acc-1", Surrogate::new(4242), )], + declared_primary_key: Some("sku".to_string()), }); let bytes = to_replicated_entry(tenant, DatabaseId::DEFAULT, vshard, &bulk) .expect("encode must not error") @@ -597,16 +600,24 @@ mod tests { match decoded { PhysicalPlan::Document(DocumentOp::BulkDelete { resolved_sum_targets, + declared_primary_key, .. - }) => assert_eq!( - resolved_sum_targets, - vec![ResolvedSumTarget::new( - "accounts", - "acc-1", - Surrogate::new(4242) - )], - "a replica re-derives which rows matched, never which target they credit" - ), + }) => { + assert_eq!( + resolved_sum_targets, + vec![ResolvedSumTarget::new( + "accounts", + "acc-1", + Surrogate::new(4242) + )], + "a replica re-derives which rows matched, never which target they credit" + ); + assert_eq!( + declared_primary_key.as_deref(), + Some("sku"), + "the declared primary key travels on the record" + ); + } other => panic!("expected BulkDelete, got {other:?}"), } } diff --git a/nodedb/src/control/wal_replication/encode/document.rs b/nodedb/src/control/wal_replication/encode/document.rs index f4b7f5432..4391efbd0 100644 --- a/nodedb/src/control/wal_replication/encode/document.rs +++ b/nodedb/src/control/wal_replication/encode/document.rs @@ -211,6 +211,7 @@ pub(super) fn bulk_delete( resolved_sum_targets: &[ResolvedSumTarget], returning: Option>, rls_filters: &[u8], + declared_primary_key: Option<&str>, ) -> ReplicatedWrite { ReplicatedWrite::BulkDml { collection: collection.to_owned(), @@ -221,7 +222,7 @@ pub(super) fn bulk_delete( resolved_sum_target_bindings: wire_target_bindings(resolved_sum_targets), returning, rls_filters: rls_filters.to_vec(), - declared_primary_key: None, + declared_primary_key: declared_primary_key.map(str::to_owned), } } diff --git a/nodedb/src/control/wal_replication/encode/entry_document.rs b/nodedb/src/control/wal_replication/encode/entry_document.rs index 0719d12cf..9d9f5d682 100644 --- a/nodedb/src/control/wal_replication/encode/entry_document.rs +++ b/nodedb/src/control/wal_replication/encode/entry_document.rs @@ -147,12 +147,16 @@ pub(super) fn document_write(op: &DocumentOp) -> Option { rls_write_check: _, // See `PointPut`. Matches are re-derived by every replica; target identity is not. resolved_sum_targets, + // Carried on the record so a staged replay keys each removed row + // by its identity — see `decode/document.rs`. + declared_primary_key, } => document::bulk_delete( collection.as_str(), filters, resolved_sum_targets, encode_returning(returning), rls_filters, + declared_primary_key.as_deref(), ), DocumentOp::BulkUpdate { collection, diff --git a/nodedb/src/data/executor/dispatch/document.rs b/nodedb/src/data/executor/dispatch/document.rs index 059424e9d..7dbbefa65 100644 --- a/nodedb/src/data/executor/dispatch/document.rs +++ b/nodedb/src/data/executor/dispatch/document.rs @@ -281,6 +281,9 @@ impl CoreLoop { rls_filters, rls_write_check, resolved_sum_targets, + // Read only by overlay staging; a durable delete removes rows + // by storage key. + declared_primary_key: _, } => self.execute_bulk_delete( task, tid, diff --git a/nodedb/src/data/executor/handlers/control/calvin.rs b/nodedb/src/data/executor/handlers/control/calvin.rs index e152e84fd..d7470a9c0 100644 --- a/nodedb/src/data/executor/handlers/control/calvin.rs +++ b/nodedb/src/data/executor/handlers/control/calvin.rs @@ -591,6 +591,7 @@ mod tests { rls_filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, resolved_sum_targets: Vec::new(), + declared_primary_key: None, }) } diff --git a/nodedb/src/data/executor/handlers/control/calvin_overlay_stage.rs b/nodedb/src/data/executor/handlers/control/calvin_overlay_stage.rs index d0dd59d1b..e55eceb85 100644 --- a/nodedb/src/data/executor/handlers/control/calvin_overlay_stage.rs +++ b/nodedb/src/data/executor/handlers/control/calvin_overlay_stage.rs @@ -13,6 +13,7 @@ //! (`resolve/entry.rs`). use nodedb_physical::physical_plan::{DocumentOp, GraphOp, PhysicalPlan, TimeseriesOp}; +use nodedb_types::RowIdentity; use crate::bridge::envelope::{ErrorCode, Response, Status}; use crate::data::executor::core_loop::CoreLoop; @@ -68,7 +69,7 @@ impl CoreLoop { tid, txn_id, collection.as_str(), - document_id, + RowIdentity::from_user_key(document_id.as_str()), *surrogate, ); let resp = self.stage_point_insert(&ctx, value, *if_absent); @@ -86,7 +87,7 @@ impl CoreLoop { tid, txn_id, collection.as_str(), - document_id, + RowIdentity::from_user_key(document_id.as_str()), *surrogate, ); let resp = self.stage_point_put(&ctx, value); @@ -104,7 +105,7 @@ impl CoreLoop { tid, txn_id, collection.as_str(), - document_id, + RowIdentity::from_user_key(document_id.as_str()), *surrogate, ); let resp = self.stage_point_delete(&ctx, rls_write_check); @@ -124,7 +125,7 @@ impl CoreLoop { tid, txn_id, collection.as_str(), - document_id, + RowIdentity::from_user_key(document_id.as_str()), *surrogate, ); let resp = self.stage_point_update( @@ -149,7 +150,7 @@ impl CoreLoop { tid, txn_id, collection.as_str(), - document_id, + RowIdentity::from_user_key(document_id.as_str()), *surrogate, ); let resp = @@ -160,16 +161,18 @@ impl CoreLoop { collection, ollp_predicted_surrogates, rls_write_check, + declared_primary_key, .. }) => self - .stage_calvin_bulk_delete( + .stage_calvin_bulk_delete(super::calvin_overlay_stage_bulk::CalvinBulkDeleteStage { task, tid, txn_id, - collection.as_str(), - ollp_predicted_surrogates.as_deref(), + collection: collection.as_str(), + ollp_predicted_surrogates: ollp_predicted_surrogates.as_deref(), rls_write_check, - ) + declared_primary_key: declared_primary_key.as_deref(), + }) .map_err(ErrorCode::from), PhysicalPlan::Document(DocumentOp::BulkUpdate { collection, diff --git a/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs b/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs index 466bf1226..b3496654f 100644 --- a/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs +++ b/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs @@ -41,9 +41,25 @@ use nodedb_types::Surrogate; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::bulk_dml::scan::ollp_predicted_doc_ids; use crate::data::executor::handlers::transaction::overlay::Staged; +use crate::data::executor::handlers::transaction::stage_write::stored_row_identity; use crate::data::executor::task::ExecutionTask; use crate::types::{DatabaseId, TenantId, TxnId}; +/// Borrowed inputs for [`CoreLoop::stage_calvin_bulk_delete`], grouped so the +/// method stays within the argument-count limit. +pub(in crate::data::executor) struct CalvinBulkDeleteStage<'a> { + pub task: &'a ExecutionTask, + pub tid: u64, + pub txn_id: TxnId, + pub collection: &'a str, + pub ollp_predicted_surrogates: Option<&'a [u32]>, + /// Compiled RLS write policy deciding each removed row's pre-image. + pub rls_write_check: &'a nodedb_types::RlsWriteCheck, + /// The collection's DDL-declared primary key, when it has one. Names the + /// column each removed row's identity is read from. + pub declared_primary_key: Option<&'a str>, +} + /// Loudly reject a Calvin bulk predicate plan that reached overlay staging /// without a predicted surrogate set. A Calvin-reachable bulk plan always /// carries one (injected at Control-Plane recon before dispatch); a plan @@ -82,26 +98,47 @@ impl CoreLoop { /// uses to derive `apply_ids`. NOT a live predicate rescan. pub(in crate::data::executor) fn stage_calvin_bulk_delete( &mut self, - task: &ExecutionTask, - tid: u64, - txn_id: TxnId, - collection: &str, - ollp_predicted_surrogates: Option<&[u32]>, - rls_write_check: &nodedb_types::RlsWriteCheck, + params: CalvinBulkDeleteStage<'_>, ) -> crate::Result<()> { + let CalvinBulkDeleteStage { + task, + tid, + txn_id, + collection, + ollp_predicted_surrogates, + rls_write_check, + declared_primary_key, + } = params; let Some(predicted) = ollp_predicted_surrogates else { return Err(missing_prediction_error(collection)); }; - let coll_key: (DatabaseId, TenantId, String) = ( - task.request.database_id, - TenantId::new(tid), - collection.to_string(), - ); + let database_id = task.request.database_id; + let coll_key: (DatabaseId, TenantId, String) = + (database_id, TenantId::new(tid), collection.to_string()); let mut predicted_sorted: Vec = predicted.to_vec(); predicted_sorted.sort_unstable(); let doc_ids = ollp_predicted_doc_ids(predicted); + // Each row's identity is read once from its current body, by the rule + // INSERT minted it with. A row with no current body removes nothing; + // its tombstone is keyed by the decimal surrogate. + let strict_schema = self.resolve_strict_schema(database_id.as_u64(), tid, collection); + let mut rows: Vec<(u32, nodedb_types::RowIdentity, Option>)> = + Vec::with_capacity(doc_ids.len()); + for (surrogate, doc_id) in predicted_sorted.into_iter().zip(doc_ids) { + let body = self + .sparse + .get(database_id.as_u64(), tid, collection, &doc_id)?; + let identity = match &body { + Some(body) => { + stored_row_identity(body, strict_schema.as_ref(), declared_primary_key, doc_id) + } + None => doc_id.to_identity(), + }; + rows.push((surrogate, identity, body)); + } + // Decide every predicted row's pre-deletion image against the write // policy BEFORE any tombstone is staged, so a rejected row cannot leave // the rows ahead of it hidden from the rest of the transaction. A row @@ -110,17 +147,13 @@ impl CoreLoop { rls_write_check.decision(), nodedb_types::WriteGateDecision::AdmitAll ) { - for doc_id in &doc_ids { - if let Some(body) = - self.sparse - .get(task.request.database_id.as_u64(), tid, collection, doc_id)? - { - let identity = doc_id.to_identity(); + for (_, identity, body) in &rows { + if let Some(body) = body { self.stage_admit_write( rls_write_check, - &body, - &identity, - task.request.database_id.as_u64(), + body, + identity, + database_id.as_u64(), tid, collection, )?; @@ -129,8 +162,8 @@ impl CoreLoop { } let overlay = self.txn_overlay_mut(txn_id); - for (surrogate, doc_id) in predicted_sorted.into_iter().zip(doc_ids) { - overlay.insert_tombstone(coll_key.clone(), surrogate, &doc_id.to_string()); + for (surrogate, identity, _body) in &rows { + overlay.insert_tombstone(coll_key.clone(), *surrogate, identity); } Ok(()) } @@ -173,6 +206,7 @@ impl CoreLoop { let mut predicted_sorted: Vec = predicted.to_vec(); predicted_sorted.sort_unstable(); + let strict_schema = self.resolve_strict_schema(database_id.as_u64(), tid, collection); for surrogate in predicted_sorted { let storage_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); @@ -217,7 +251,12 @@ impl CoreLoop { )?; // Decide the staged post-image against the write policy: this is // the row the Calvin flush will install. - let identity = storage_key.to_identity(); + let identity = stored_row_identity( + &new_body, + strict_schema.as_ref(), + declared_primary_key, + storage_key, + ); self.stage_admit_write( rls_write_check, &new_body, @@ -226,9 +265,7 @@ impl CoreLoop { tid, collection, )?; - // The overlay's doc-id side map is keyed by text. - let doc_id = storage_key.to_string(); - self.stage_bulk_put_capped(txn_id, &coll_key, surrogate, &doc_id, new_body)?; + self.stage_bulk_put_capped(txn_id, &coll_key, surrogate, &identity, new_body)?; } Ok(()) } diff --git a/nodedb/src/data/executor/handlers/control/calvin_resolve.rs b/nodedb/src/data/executor/handlers/control/calvin_resolve.rs index 9145acebc..f6ac2d30b 100644 --- a/nodedb/src/data/executor/handlers/control/calvin_resolve.rs +++ b/nodedb/src/data/executor/handlers/control/calvin_resolve.rs @@ -171,6 +171,7 @@ mod tests { rls_filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, resolved_sum_targets: Vec::new(), + declared_primary_key: None, }) } diff --git a/nodedb/src/data/executor/handlers/document/resolve/dispatch.rs b/nodedb/src/data/executor/handlers/document/resolve/dispatch.rs index 1875f1250..cfd52f4a0 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/dispatch.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/dispatch.rs @@ -136,6 +136,9 @@ impl CoreLoop { resolved_sum_targets, ollp_predicted_surrogates: _, ollp_predicted_edges: _, + // Read only by overlay staging; a resolved delete removes rows + // by storage key. + declared_primary_key: _, } => self.resolve_bulk_delete( task, ResolveBulkDelete { diff --git a/nodedb/src/data/executor/handlers/kv/crud/get.rs b/nodedb/src/data/executor/handlers/kv/crud/get.rs index 6aa99845d..4bce7041d 100644 --- a/nodedb/src/data/executor/handlers/kv/crud/get.rs +++ b/nodedb/src/data/executor/handlers/kv/crud/get.rs @@ -8,7 +8,7 @@ use super::types::KvGetParams; use crate::bridge::envelope::Response; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::transaction::overlay::{Staged, StagedTtl}; -use crate::data::executor::handlers::transaction::stage_write::hex_key; +use crate::data::executor::handlers::transaction::stage_write::kv_row_identity; use crate::data::executor::task::ExecutionTask; use crate::engine::kv::current_ms; use crate::types::TenantId; @@ -41,7 +41,7 @@ impl CoreLoop { TenantId::new(tid), collection.to_string(), ); - let doc_id = hex_key(key); + let doc_id = kv_row_identity(key); if let Some(overlay) = self.txn_overlays.get(&txn_id) { // A staged EXPIRE with an already-past instant makes the row // appear absent to a same-transaction read -- independent of diff --git a/nodedb/src/data/executor/handlers/kv/ttl.rs b/nodedb/src/data/executor/handlers/kv/ttl.rs index d6b5bf353..7c83866c4 100644 --- a/nodedb/src/data/executor/handlers/kv/ttl.rs +++ b/nodedb/src/data/executor/handlers/kv/ttl.rs @@ -7,7 +7,7 @@ use tracing::debug; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::transaction::overlay::{Staged, StagedTtl}; -use crate::data::executor::handlers::transaction::stage_write::hex_key; +use crate::data::executor::handlers::transaction::stage_write::kv_row_identity; use crate::data::executor::response_codec; use crate::data::executor::task::ExecutionTask; use crate::engine::kv::current_ms; @@ -156,7 +156,7 @@ impl CoreLoop { TenantId::new(tid), collection.to_string(), ); - let doc_id = hex_key(key); + let doc_id = kv_row_identity(key); if let Some(overlay) = self.txn_overlays.get(&txn_id) { let staged_value = overlay.get_by_doc_id(&coll_key, &doc_id); if matches!(staged_value, Some(Staged::Tombstone)) { diff --git a/nodedb/src/data/executor/handlers/point/get.rs b/nodedb/src/data/executor/handlers/point/get.rs index 5e787cdf5..1e5adfe9c 100644 --- a/nodedb/src/data/executor/handlers/point/get.rs +++ b/nodedb/src/data/executor/handlers/point/get.rs @@ -8,7 +8,7 @@ use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::scan_normalize::{sparse_body_to_msgpack, sparse_row_to_doc}; use crate::data::executor::task::ExecutionTask; -use nodedb_types::Surrogate; +use nodedb_types::{RowIdentity, Surrogate}; pub(in crate::data::executor) struct PointGetParams<'a> { pub tid: u64, @@ -87,9 +87,13 @@ impl CoreLoop { ); } } - } else if let Some(overlay_data) = - self.overlay_point_lookup(task, tid, collection, document_id, surrogate) - { + } else if let Some(overlay_data) = self.overlay_point_lookup( + task, + tid, + collection, + &RowIdentity::from_user_key(document_id), + surrogate, + ) { match overlay_data { Ok(data) => data, Err(response) => return response, diff --git a/nodedb/src/data/executor/handlers/point/overlay_lookup.rs b/nodedb/src/data/executor/handlers/point/overlay_lookup.rs index 6eb6beb94..9068345ab 100644 --- a/nodedb/src/data/executor/handlers/point/overlay_lookup.rs +++ b/nodedb/src/data/executor/handlers/point/overlay_lookup.rs @@ -12,14 +12,16 @@ use crate::bridge::envelope::Response; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::transaction::overlay::{Staged, StagedTtl}; -use crate::data::executor::handlers::transaction::stage_write::hex_key; +use crate::data::executor::handlers::transaction::stage_write::kv_row_identity; use crate::data::executor::task::ExecutionTask; use crate::engine::kv::current_ms; -use nodedb_types::Surrogate; +use nodedb_types::{RowIdentity, Surrogate}; impl CoreLoop { /// Consult the active transaction's staging overlay for a point-get. /// + /// `document_id` is the plan's resolved client identity of the row. + /// /// Returns `None` when there is no active transaction on this task, or /// the transaction has no overlay entry for this collection/surrogate — /// callers should fall through to the normal cache/base-storage lookup. @@ -37,7 +39,7 @@ impl CoreLoop { task: &ExecutionTask, tid: u64, collection: &str, - document_id: &str, + document_id: &RowIdentity, surrogate: Surrogate, ) -> Option, Response>> { let txn_id = task.request.txn_id?; @@ -49,9 +51,6 @@ impl CoreLoop { crate::types::TenantId::new(tid), collection.to_string(), ); - // A staged-only insert has no base surrogate yet, so the read plan's - // `surrogate` is unresolved (zero) — resolve by document id first, then - // fall back to the surrogate for rows that already exist in base. let overlay = self.txn_overlays.get(&txn_id)?; // A staged-only insert has no base surrogate yet, so the read plan's // `surrogate` is unresolved (zero) — resolve by document id first, then @@ -66,9 +65,8 @@ impl CoreLoop { } /// Consult the active transaction's staging overlay for a raw KV key - /// (hex-encoded into the overlay's doc-id, same as every KV staging - /// path -- see `stage_kv::hex_key`), for read-merge in `BatchGet` / - /// `FieldGet`. + /// (keyed by `kv_row_identity`, same as every KV staging path), for + /// read-merge in `BatchGet` / `FieldGet`. /// /// Unlike [`overlay_point_lookup`], which is tailored to a single /// point-get's not-found response shape, this returns a plain nested @@ -90,7 +88,7 @@ impl CoreLoop { crate::types::TenantId::new(tid), collection.to_string(), ); - let doc_id = hex_key(key); + let doc_id = kv_row_identity(key); let overlay = self.txn_overlays.get(&txn_id)?; // A staged EXPIRE with an already-past instant makes the row appear diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/merge.rs b/nodedb/src/data/executor/handlers/transaction/overlay/merge.rs index 4bce930c6..67b9e67d9 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/merge.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/merge.rs @@ -32,7 +32,7 @@ use std::collections::HashSet; use nodedb_types::columnar::StrictSchema; -use nodedb_types::{StorageKey, Surrogate}; +use nodedb_types::{RowIdentity, StorageKey, Surrogate}; use crate::bridge::scan_filter::ScanFilter; use crate::data::executor::core_loop::CoreLoop; @@ -138,8 +138,8 @@ impl CoreLoop { /// Unlike [`merge_overlay_into_scan`](Self::merge_overlay_into_scan), /// whose row identity is the Document scan's hex-surrogate row key, a /// KV row's scan identity is its raw key bytes -- so this merges by - /// [`hex_key`](super::super::stage_write::hex_key) identity instead, - /// via [`TxnOverlay::iter_doc_entries_for_collection`] and + /// [`kv_row_identity`](super::super::stage_write::kv_row_identity) + /// instead, via [`TxnOverlay::iter_doc_entries_for_collection`] and /// [`unhex_key`](super::super::stage_write::unhex_key) to recover the /// raw key bytes for a staged addition. `matches` is the SAME predicate /// the base KV scan applied, evaluated on the value bytes. @@ -156,13 +156,13 @@ impl CoreLoop { return; }; - let mut seen: HashSet = rows + let mut seen: HashSet = rows .iter() - .map(|(key, _)| super::super::stage_write::hex_key(key)) + .map(|(key, _)| super::super::stage_write::kv_row_identity(key)) .collect(); let now_ms = current_ms(); - let staged_expired = |doc_id: &str| -> bool { + let staged_expired = |doc_id: &RowIdentity| -> bool { matches!( overlay.get_ttl_by_doc_id(coll_key, doc_id), Some(StagedTtl::ExpireAt(t)) if t <= now_ms @@ -170,7 +170,7 @@ impl CoreLoop { }; rows.retain_mut(|(key, value)| { - let doc_id = super::super::stage_write::hex_key(key); + let doc_id = super::super::stage_write::kv_row_identity(key); // A staged EXPIRE with an already-past instant hides the row from // an in-transaction scan -- independent of whether the row's // VALUE was also staged this transaction (an `Expire` on a @@ -196,10 +196,10 @@ impl CoreLoop { if let Staged::Put(value) = staged && !staged_expired(doc_id) && matches(value) - && let Some(key) = super::super::stage_write::unhex_key(doc_id) + && let Some(key) = super::super::stage_write::unhex_key(doc_id.as_str()) { rows.push((key, value.clone())); - seen.insert(doc_id.to_string()); + seen.insert(doc_id.clone()); } } } diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/staged.rs b/nodedb/src/data/executor/handlers/transaction/overlay/staged.rs index 9f12bf02f..217442583 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/staged.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/staged.rs @@ -9,13 +9,16 @@ //! //! Keying rationale: the real storage key for a document is the SURROGATE //! (`u32`) — `apply_point_put` keys `sparse.versioned_put_in_txn` by -//! surrogate. `doc_id_to_surrogate` lets later units resolve a doc_id to a -//! staged surrogate for not-yet-persisted inserts (a doc_id that has no -//! durable surrogate yet because the insert itself is only staged). +//! surrogate. `doc_id_to_surrogate` resolves a row's client identity +//! ([`RowIdentity`]) to its staged surrogate, so a not-yet-persisted insert +//! is found by the identity a point read carries. A KV row's identity is +//! its raw key, hex encoded, taken verbatim (`stage_kv::kv_row_identity`). use std::cell::Cell; use std::collections::HashMap; +use nodedb_types::RowIdentity; + use crate::types::{DatabaseId, TenantId}; /// Per-core upper bound on the total staged-body bytes a single transaction's @@ -70,10 +73,10 @@ pub struct BitemporalStamp { pub struct CollectionOverlay { /// Staged mutation per surrogate — the authoritative storage key. by_surrogate: HashMap, - /// Resolves a doc_id to its staged surrogate, for inserts that have not - /// yet been made durable (and therefore have no other way to be looked - /// up by doc_id). - doc_id_to_surrogate: HashMap, + /// Resolves a row's client identity to its staged surrogate, for inserts + /// that have not yet been made durable (and therefore have no other way + /// to be looked up by identity). + doc_id_to_surrogate: HashMap, /// Staged KV TTL delta per surrogate — sibling to `by_surrogate`, never /// consulted by non-KV engines. See [`StagedTtl`]. ttl_by_surrogate: HashMap, @@ -104,7 +107,7 @@ impl CollectionOverlay { struct OverlayUndo { coll_key: (DatabaseId, TenantId, String), surrogate: u32, - doc_id: String, + doc_id: RowIdentity, /// Prior `by_surrogate` entry, or `None` if the slot was absent. prev_value: Option, /// Prior `ttl_by_surrogate` entry, or `None` if absent. @@ -167,7 +170,7 @@ impl TxnOverlay { &mut self, coll_key: &(DatabaseId, TenantId, String), surrogate: u32, - doc_id: &str, + doc_id: &RowIdentity, ) { let (prev_value, prev_ttl, prev_doc_binding) = match self.collections.get(coll_key) { Some(overlay) => ( @@ -180,7 +183,7 @@ impl TxnOverlay { self.journal.push(OverlayUndo { coll_key: coll_key.clone(), surrogate, - doc_id: doc_id.to_string(), + doc_id: doc_id.clone(), prev_value, prev_ttl, prev_doc_binding, @@ -192,7 +195,7 @@ impl TxnOverlay { &mut self, coll_key: (DatabaseId, TenantId, String), surrogate: u32, - doc_id: &str, + doc_id: &RowIdentity, body: Vec, ) { self.record_undo(&coll_key, surrogate, doc_id); @@ -200,7 +203,7 @@ impl TxnOverlay { overlay.by_surrogate.insert(surrogate, Staged::Put(body)); overlay .doc_id_to_surrogate - .insert(doc_id.to_string(), surrogate); + .insert(doc_id.clone(), surrogate); } /// Stage a tombstone (delete) for `surrogate` in the given collection. @@ -208,14 +211,14 @@ impl TxnOverlay { &mut self, coll_key: (DatabaseId, TenantId, String), surrogate: u32, - doc_id: &str, + doc_id: &RowIdentity, ) { self.record_undo(&coll_key, surrogate, doc_id); let overlay = self.collections.entry(coll_key).or_default(); overlay.by_surrogate.insert(surrogate, Staged::Tombstone); overlay .doc_id_to_surrogate - .insert(doc_id.to_string(), surrogate); + .insert(doc_id.clone(), surrogate); } /// Look up the staged mutation for `surrogate` in the given collection. @@ -234,7 +237,7 @@ impl TxnOverlay { pub fn get_by_doc_id( &self, coll_key: &(DatabaseId, TenantId, String), - doc_id: &str, + doc_id: &RowIdentity, ) -> Option<&Staged> { let overlay = self.collections.get(coll_key)?; let surrogate = overlay.doc_id_to_surrogate.get(doc_id)?; @@ -247,7 +250,7 @@ impl TxnOverlay { pub fn surrogate_for_doc_id( &self, coll_key: &(DatabaseId, TenantId, String), - doc_id: &str, + doc_id: &RowIdentity, ) -> Option { self.collections .get(coll_key)? @@ -260,12 +263,12 @@ impl TxnOverlay { /// given collection, binding `doc_id` to `surrogate` the same way /// `insert_put` / `insert_tombstone` do — a `GetTtl` (or a later /// `Expire`/`Persist`/`Incr` in the same transaction) resolves the same - /// slot by hex-encoded KV key. + /// slot by the KV row's identity. pub fn set_ttl( &mut self, coll_key: (DatabaseId, TenantId, String), surrogate: u32, - doc_id: &str, + doc_id: &RowIdentity, ttl: StagedTtl, ) { self.record_undo(&coll_key, surrogate, doc_id); @@ -273,7 +276,7 @@ impl TxnOverlay { overlay.ttl_by_surrogate.insert(surrogate, ttl); overlay .doc_id_to_surrogate - .insert(doc_id.to_string(), surrogate); + .insert(doc_id.clone(), surrogate); } /// Current length of the overlay undo journal — the savepoint marker a @@ -346,7 +349,7 @@ impl TxnOverlay { pub fn get_ttl_by_doc_id( &self, coll_key: &(DatabaseId, TenantId, String), - doc_id: &str, + doc_id: &RowIdentity, ) -> Option { let overlay = self.collections.get(coll_key)?; let surrogate = overlay.doc_id_to_surrogate.get(doc_id)?; @@ -409,17 +412,16 @@ impl TxnOverlay { .flat_map(|overlay| overlay.by_surrogate.iter().map(|(k, v)| (*k, v))) } - /// Iterate all staged `(doc_id, Staged)` pairs for a collection. + /// Iterate all staged `(identity, Staged)` pairs for a collection. /// /// Unlike [`iter_for_collection`](Self::iter_for_collection) (keyed by /// surrogate, the Document scan's row identity), this is keyed by the - /// overlay's doc-id -- the identity a KV scan merge needs, since a KV - /// row's scan identity is its raw key bytes (hex-encoded into the - /// doc-id), not a surrogate. + /// row's client identity -- the identity a KV scan merge needs, since a + /// KV row's scan identity is its raw key bytes, not a surrogate. pub fn iter_doc_entries_for_collection<'a>( &'a self, coll_key: &(DatabaseId, TenantId, String), - ) -> impl Iterator { + ) -> impl Iterator { self.collections .get(coll_key) .into_iter() @@ -431,7 +433,7 @@ impl TxnOverlay { overlay .by_surrogate .get(surrogate) - .map(|staged| (doc_id.as_str(), staged)) + .map(|staged| (doc_id, staged)) }) }) } @@ -474,6 +476,10 @@ mod tests { (DatabaseId::new(1), TenantId::new(1), coll.to_string()) } + fn id(text: &str) -> RowIdentity { + RowIdentity::from_user_key(text) + } + #[test] fn empty_overlay_has_no_entries() { let overlay = TxnOverlay::new(); @@ -481,14 +487,14 @@ mod tests { assert_eq!(overlay.len(), 0); assert_eq!(overlay.memory_size_estimate(), 0); assert!(overlay.get(&key("users"), 1).is_none()); - assert!(overlay.get_by_doc_id(&key("users"), "abc").is_none()); + assert!(overlay.get_by_doc_id(&key("users"), &id("abc")).is_none()); assert_eq!(overlay.iter_for_collection(&key("users")).count(), 0); } #[test] fn insert_put_and_lookup() { let mut overlay = TxnOverlay::new(); - overlay.insert_put(key("users"), 7, "doc-1", vec![1, 2, 3]); + overlay.insert_put(key("users"), 7, &id("doc-1"), vec![1, 2, 3]); assert!(!overlay.is_empty()); assert_eq!(overlay.len(), 1); @@ -498,7 +504,7 @@ mod tests { Some(&Staged::Put(vec![1, 2, 3])) ); assert_eq!( - overlay.get_by_doc_id(&key("users"), "doc-1"), + overlay.get_by_doc_id(&key("users"), &id("doc-1")), Some(&Staged::Put(vec![1, 2, 3])) ); let collected: Vec<_> = overlay.iter_for_collection(&key("users")).collect(); @@ -508,12 +514,51 @@ mod tests { #[test] fn insert_tombstone_and_lookup() { let mut overlay = TxnOverlay::new(); - overlay.insert_tombstone(key("users"), 9, "doc-2"); + overlay.insert_tombstone(key("users"), 9, &id("doc-2")); assert_eq!(overlay.get(&key("users"), 9), Some(&Staged::Tombstone)); assert_eq!(overlay.memory_size_estimate(), 0); } + #[test] + fn bulk_staged_row_is_found_by_the_point_get_identity() { + // A bulk path keys the row by `RowIdentity::of_stored_row`. A point + // get carries the plan's resolved identity: the `id` column value. + // Both must land on the same overlay slot. + let mut body_obj = HashMap::new(); + body_obj.insert( + "id".to_string(), + nodedb_types::Value::String("user-7".into()), + ); + body_obj.insert( + "name".to_string(), + nodedb_types::Value::String("ann".into()), + ); + let body = nodedb_types::value_to_msgpack(&nodedb_types::Value::Object(body_obj)) + .expect("encode msgpack"); + let storage_key = nodedb_types::StorageKey::for_surrogate(nodedb_types::Surrogate::new(7)); + let bulk_identity = RowIdentity::of_stored_row(&body, None, storage_key); + + let mut overlay = TxnOverlay::new(); + overlay.insert_put(key("users"), 7, &bulk_identity, body.clone()); + + let point_get_identity = RowIdentity::from_user_key("user-7"); + assert_eq!( + overlay.get_by_doc_id(&key("users"), &point_get_identity), + Some(&Staged::Put(body)) + ); + assert_eq!( + overlay.surrogate_for_doc_id(&key("users"), &point_get_identity), + Some(7) + ); + assert!( + overlay + .get_by_doc_id(&key("users"), &storage_key.to_identity()) + .is_none(), + "a row carrying an `id` column is never keyed by its surrogate" + ); + } + // ── KV TTL delta (`StagedTtl`) ────────────────────────────────────── // // `KvOp::Expire` / `KvOp::Persist` / `GetTtl` have no SQL or native-DSL @@ -531,14 +576,14 @@ mod tests { #[test] fn set_ttl_and_get_ttl_round_trip() { let mut overlay = TxnOverlay::new(); - overlay.set_ttl(key("cache"), 3, "6b6579", StagedTtl::ExpireAt(5_000)); + overlay.set_ttl(key("cache"), 3, &id("6b6579"), StagedTtl::ExpireAt(5_000)); assert_eq!( overlay.get_ttl(&key("cache"), 3), Some(StagedTtl::ExpireAt(5_000)) ); assert_eq!( - overlay.get_ttl_by_doc_id(&key("cache"), "6b6579"), + overlay.get_ttl_by_doc_id(&key("cache"), &id("6b6579")), Some(StagedTtl::ExpireAt(5_000)) ); } @@ -546,8 +591,8 @@ mod tests { #[test] fn set_ttl_persist_overrides_prior_expire() { let mut overlay = TxnOverlay::new(); - overlay.set_ttl(key("cache"), 3, "6b6579", StagedTtl::ExpireAt(5_000)); - overlay.set_ttl(key("cache"), 3, "6b6579", StagedTtl::Persist); + overlay.set_ttl(key("cache"), 3, &id("6b6579"), StagedTtl::ExpireAt(5_000)); + overlay.set_ttl(key("cache"), 3, &id("6b6579"), StagedTtl::Persist); assert_eq!(overlay.get_ttl(&key("cache"), 3), Some(StagedTtl::Persist)); } @@ -556,7 +601,10 @@ mod tests { fn get_ttl_none_when_nothing_staged() { let overlay = TxnOverlay::new(); assert_eq!(overlay.get_ttl(&key("cache"), 3), None); - assert_eq!(overlay.get_ttl_by_doc_id(&key("cache"), "6b6579"), None); + assert_eq!( + overlay.get_ttl_by_doc_id(&key("cache"), &id("6b6579")), + None + ); } #[test] @@ -565,11 +613,15 @@ mod tests { // (only a base row exists) must still resolve by doc_id -- `set_ttl` // binds `doc_id_to_surrogate` itself, independent of `insert_put`. let mut overlay = TxnOverlay::new(); - overlay.set_ttl(key("cache"), 42, "6b6579", StagedTtl::ExpireAt(9_999)); + overlay.set_ttl(key("cache"), 42, &id("6b6579"), StagedTtl::ExpireAt(9_999)); - assert!(overlay.get_by_doc_id(&key("cache"), "6b6579").is_none()); + assert!( + overlay + .get_by_doc_id(&key("cache"), &id("6b6579")) + .is_none() + ); assert_eq!( - overlay.get_ttl_by_doc_id(&key("cache"), "6b6579"), + overlay.get_ttl_by_doc_id(&key("cache"), &id("6b6579")), Some(StagedTtl::ExpireAt(9_999)) ); } @@ -577,7 +629,7 @@ mod tests { #[test] fn ttl_delta_is_per_collection() { let mut overlay = TxnOverlay::new(); - overlay.set_ttl(key("a"), 1, "6b", StagedTtl::ExpireAt(1_000)); + overlay.set_ttl(key("a"), 1, &id("6b"), StagedTtl::ExpireAt(1_000)); assert_eq!(overlay.get_ttl(&key("b"), 1), None); } @@ -586,10 +638,10 @@ mod tests { let mut overlay = TxnOverlay::new(); let retained = key("retained"); let post_marker = key("post_marker"); - overlay.insert_put(retained.clone(), 7, "stable", vec![1, 2, 3]); + overlay.insert_put(retained.clone(), 7, &id("stable"), vec![1, 2, 3]); let marker = overlay.journal_len(); - overlay.insert_put(post_marker.clone(), 9, "temporary", vec![4, 5]); + overlay.insert_put(post_marker.clone(), 9, &id("temporary"), vec![4, 5]); overlay.rollback_to(marker); assert!( @@ -603,7 +655,7 @@ mod tests { "the pre-savepoint body must remain byte-exact" ); assert_eq!( - overlay.get_by_doc_id(&retained, "stable"), + overlay.get_by_doc_id(&retained, &id("stable")), Some(&Staged::Put(vec![1, 2, 3])) ); assert_eq!(overlay.journal_len(), marker); diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/document.rs b/nodedb/src/data/executor/handlers/transaction/resolve/document.rs index e10a59fc0..31607935b 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/document.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/document.rs @@ -42,12 +42,12 @@ //! ## Determinism //! //! The overlay keys slots by surrogate in a `HashMap`, so entries are collected -//! into a `BTreeMap` keyed by the overlay doc-id (the user primary key) before -//! emitting. Two replicas resolving the same transaction produce byte-identical -//! redo ops. +//! into a `BTreeMap` keyed by the row's client identity before emitting. Two +//! replicas resolving the same transaction produce byte-identical redo ops. use std::collections::BTreeMap; +use nodedb_types::RowIdentity; use nodedb_types::columnar::StrictSchema; use nodedb_types::sync::wire::SyncProvenance; use nodedb_wal::record::RecordType; @@ -71,7 +71,7 @@ pub(super) fn serialize_document_collection( strict_schema: Option<&StrictSchema>, ops: &mut Vec, ) -> crate::Result<()> { - let mut entries: BTreeMap = BTreeMap::new(); + let mut entries: BTreeMap<&RowIdentity, (u32, &Staged)> = BTreeMap::new(); for (doc_id, staged) in overlay.iter_doc_entries_for_collection(coll_key) { let surrogate = overlay .surrogate_for_doc_id(coll_key, doc_id) @@ -80,7 +80,7 @@ pub(super) fn serialize_document_collection( "document resolve: staged doc-id '{doc_id}' has no bound surrogate" ), })?; - entries.insert(doc_id.to_string(), (surrogate, staged)); + entries.insert(doc_id, (surrogate, staged)); } for (doc_id, (surrogate, staged)) in entries { @@ -191,12 +191,16 @@ mod tests { zerompk::to_msgpack_vec(&Value::Object(obj)).expect("encode msgpack") } + fn id(text: &str) -> RowIdentity { + RowIdentity::from_user_key(text) + } + #[test] fn strict_put_emits_msgpack_not_binary_tuple() { let schema = strict_schema(); let tuple = strict_tuple(7, "elephant"); let mut overlay = TxnOverlay::new(); - overlay.insert_put(coll_key("docs"), 7, "row1", tuple.clone()); + overlay.insert_put(coll_key("docs"), 7, &id("row1"), tuple.clone()); let mut ops = Vec::new(); serialize_document_collection(&overlay, &coll_key("docs"), "docs", Some(&schema), &mut ops) @@ -233,7 +237,7 @@ mod tests { fn schemaless_put_emits_body_verbatim() { let body = schemaless_body("alice"); let mut overlay = TxnOverlay::new(); - overlay.insert_put(coll_key("notes"), 3, "userpk", body.clone()); + overlay.insert_put(coll_key("notes"), 3, &id("userpk"), body.clone()); let mut ops = Vec::new(); serialize_document_collection(&overlay, &coll_key("notes"), "notes", None, &mut ops) @@ -252,7 +256,7 @@ mod tests { #[test] fn tombstone_emits_delete_carrying_surrogate() { let mut overlay = TxnOverlay::new(); - overlay.insert_tombstone(coll_key("notes"), 11, "gone"); + overlay.insert_tombstone(coll_key("notes"), 11, &id("gone")); let mut ops = Vec::new(); serialize_document_collection(&overlay, &coll_key("notes"), "notes", None, &mut ops) @@ -272,9 +276,9 @@ mod tests { #[test] fn entries_emit_in_deterministic_doc_id_order() { let mut overlay = TxnOverlay::new(); - overlay.insert_put(coll_key("notes"), 30, "c", schemaless_body("c")); - overlay.insert_put(coll_key("notes"), 10, "a", schemaless_body("a")); - overlay.insert_put(coll_key("notes"), 20, "b", schemaless_body("b")); + overlay.insert_put(coll_key("notes"), 30, &id("c"), schemaless_body("c")); + overlay.insert_put(coll_key("notes"), 10, &id("a"), schemaless_body("a")); + overlay.insert_put(coll_key("notes"), 20, &id("b"), schemaless_body("b")); let mut ops = Vec::new(); serialize_document_collection(&overlay, &coll_key("notes"), "notes", None, &mut ops) diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs b/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs index 86e9971c6..84ae0204a 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs @@ -9,6 +9,7 @@ use std::collections::{BTreeMap, BTreeSet}; use nodedb_physical::physical_plan::{DocumentOp, KvOp, PhysicalPlan}; +use nodedb_types::RowIdentity; use crate::bridge::envelope::Response; use crate::data::executor::core_loop::CoreLoop; @@ -127,13 +128,13 @@ impl CoreLoop { TenantId::new(tid), collection.clone(), ); - let mut puts: Vec<(String, u32)> = match self.txn_overlays.get(&txn_id) { + let mut puts: Vec<(&RowIdentity, u32)> = match self.txn_overlays.get(&txn_id) { Some(overlay) => overlay .iter_doc_entries_for_collection(&coll_key) .filter_map(|(doc_id, staged)| match staged { Staged::Put(_) => overlay .surrogate_for_doc_id(&coll_key, doc_id) - .map(|surrogate| (doc_id.to_string(), surrogate)), + .map(|surrogate| (doc_id, surrogate)), Staged::Tombstone => None, }) .collect(), @@ -381,10 +382,9 @@ mod tests { ArrayOp, ColumnarInsertIntent, ColumnarOp, DocumentOp, GraphOp, KvOp, MetaOp, ReturningColumns, ReturningSpec, StorageMode, TimeseriesOp, UpdateValue, VectorOp, }; - use nodedb_types::QualifiedCollection; - use nodedb_types::Surrogate; use nodedb_types::columnar::{ColumnDef, ColumnType, StrictSchema}; use nodedb_types::sync::wire::SyncProvenance; + use nodedb_types::{QualifiedCollection, RowIdentity, Surrogate}; use crate::data::executor::handlers::graph::EdgePutParams; use crate::data::executor::strict_format; @@ -394,7 +394,7 @@ mod tests { use crate::bridge::envelope::{PhysicalPlan, Priority, Request, Status}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::transaction::overlay::{Staged, StagedTtl}; - use crate::data::executor::handlers::transaction::stage_write::hex_key; + use crate::data::executor::handlers::transaction::stage_write::kv_row_identity; use crate::data::executor::task::ExecutionTask; use crate::types::{ DatabaseId, ReadConsistency, RequestId, TenantId, TraceId, TxnId, VShardId, @@ -511,7 +511,7 @@ mod tests { let overlay_bytes = match core .txn_overlays .get(&txn) - .and_then(|o| o.get_by_doc_id(&coll_key("counters"), &hex_key(b"c"))) + .and_then(|o| o.get_by_doc_id(&coll_key("counters"), &kv_row_identity(b"c"))) .expect("staged incr present") { Staged::Put(v) => v.clone(), @@ -548,11 +548,16 @@ mod tests { let expire_at = 1_700_000_000_000u64; { let overlay = core.txn_overlay_mut(txn); - overlay.insert_put(coll_key("sessions"), 7, &hex_key(b"s1"), b"v1".to_vec()); + overlay.insert_put( + coll_key("sessions"), + 7, + &kv_row_identity(b"s1"), + b"v1".to_vec(), + ); overlay.set_ttl( coll_key("sessions"), 7, - &hex_key(b"s1"), + &kv_row_identity(b"s1"), StagedTtl::ExpireAt(expire_at), ); } @@ -586,8 +591,12 @@ mod tests { let task = make_task(); let txn = TxnId::new(3); - core.txn_overlay_mut(txn) - .insert_put(coll_key("kvc"), 9, &hex_key(b"k9"), b"body".to_vec()); + core.txn_overlay_mut(txn).insert_put( + coll_key("kvc"), + 9, + &kv_row_identity(b"k9"), + b"body".to_vec(), + ); let resp = core.execute_resolve_txn(&task, TID, txn, &[kv_write_plan("kvc")]); let redo = decode_redo(&resp); @@ -610,7 +619,7 @@ mod tests { let txn = TxnId::new(4); core.txn_overlay_mut(txn) - .insert_tombstone(coll_key("kvc"), 11, &hex_key(b"gone")); + .insert_tombstone(coll_key("kvc"), 11, &kv_row_identity(b"gone")); let resp = core.execute_resolve_txn( &task, @@ -660,7 +669,7 @@ mod tests { core.txn_overlay_mut(txn).insert_put( coll_key("kvc"), 1, - &hex_key(b"k"), + &kv_row_identity(b"k"), b"staged".to_vec(), ); @@ -683,8 +692,11 @@ mod tests { // A DELETE ... RETURNING stages like any other point delete: the overlay // holds a tombstone, and resolve serializes it from there. - core.txn_overlay_mut(txn) - .insert_tombstone(coll_key("notes"), surrogate, "gone"); + core.txn_overlay_mut(txn).insert_tombstone( + coll_key("notes"), + surrogate, + &RowIdentity::from_user_key("gone"), + ); let doc_plan = PhysicalPlan::Document(DocumentOp::PointDelete { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "notes"), @@ -880,6 +892,7 @@ mod tests { rls_filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, resolved_sum_targets: Vec::new(), + declared_primary_key: None, }); let resp = core.execute_stage_write(&task, TID, &plan); @@ -913,8 +926,18 @@ mod tests { // leaves them; a RETURNING clause does not change the overlay contents. { let overlay = src.txn_overlay_mut(txn); - overlay.insert_put(coll_key("notes"), 1, "u1", schemaless_body("bob")); - overlay.insert_put(coll_key("notes"), 2, "u2", schemaless_body("bob")); + overlay.insert_put( + coll_key("notes"), + 1, + &RowIdentity::from_user_key("u1"), + schemaless_body("bob"), + ); + overlay.insert_put( + coll_key("notes"), + 2, + &RowIdentity::from_user_key("u2"), + schemaless_body("bob"), + ); } // Serializes the staged post-images from the overlay. @@ -1389,7 +1412,7 @@ mod tests { src.txn_overlay_mut(txn).insert_put( coll_key("sdocs"), surrogate, - "row1", + &RowIdentity::from_user_key("row1"), strict_tuple(surrogate as i64, "elephant"), ); @@ -1437,8 +1460,12 @@ mod tests { let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); let body = schemaless_body("alice"); - src.txn_overlay_mut(txn) - .insert_put(coll_key("notes"), surrogate, "userpk", body.clone()); + src.txn_overlay_mut(txn).insert_put( + coll_key("notes"), + surrogate, + &RowIdentity::from_user_key("userpk"), + body.clone(), + ); let resp = src.execute_resolve_txn(&task, TID, txn, &[doc_put_plan("notes")]); let redo = decode_redo(&resp); @@ -1469,8 +1496,11 @@ mod tests { let surrogate = 11u32; let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); - src.txn_overlay_mut(txn) - .insert_tombstone(coll_key("notes"), surrogate, "gone"); + src.txn_overlay_mut(txn).insert_tombstone( + coll_key("notes"), + surrogate, + &RowIdentity::from_user_key("gone"), + ); let delete_plan = PhysicalPlan::Document(DocumentOp::PointDelete { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "notes"), @@ -1598,7 +1628,7 @@ mod tests { core.txn_overlay_mut(txn).insert_put( coll_key("notes"), surrogate, - "userpk", + &RowIdentity::from_user_key("userpk"), schemaless_body("staged"), ); @@ -1627,11 +1657,11 @@ mod tests { { let overlay = src.txn_overlay_mut(txn); - overlay.insert_put(coll_key("kvc"), 1, &hex_key(b"k"), b"V".to_vec()); + overlay.insert_put(coll_key("kvc"), 1, &kv_row_identity(b"k"), b"V".to_vec()); overlay.insert_put( coll_key("notes"), doc_surrogate, - "userpk", + &RowIdentity::from_user_key("userpk"), schemaless_body("bob"), ); } @@ -1685,14 +1715,19 @@ mod tests { let expire_at = crate::engine::kv::current_ms() + 3_600_000; { let overlay = src.txn_overlay_mut(txn); - overlay.insert_put(coll_key("kvc"), 1, &hex_key(b"live"), b"V".to_vec()); + overlay.insert_put(coll_key("kvc"), 1, &kv_row_identity(b"live"), b"V".to_vec()); overlay.set_ttl( coll_key("kvc"), 1, - &hex_key(b"live"), + &kv_row_identity(b"live"), StagedTtl::ExpireAt(expire_at), ); - overlay.insert_put(coll_key("kvc"), 2, &hex_key(b"plain"), b"P".to_vec()); + overlay.insert_put( + coll_key("kvc"), + 2, + &kv_row_identity(b"plain"), + b"P".to_vec(), + ); } let resp = src.execute_resolve_txn(&task, TID, txn, &[kv_write_plan("kvc")]); @@ -2052,7 +2087,7 @@ mod tests { overlay.insert_put( coll_key("notes"), doc_surrogate, - "userpk", + &RowIdentity::from_user_key("userpk"), schemaless_body("carol"), ); } @@ -2779,8 +2814,12 @@ mod tests { let txn = TxnId::new(47); // Stage the KV write into the overlay (overlay-driven serializer). - src.txn_overlay_mut(txn) - .insert_put(coll_key("kvc"), 1, &hex_key(b"k"), b"V".to_vec()); + src.txn_overlay_mut(txn).insert_put( + coll_key("kvc"), + 1, + &kv_row_identity(b"k"), + b"V".to_vec(), + ); let mut row = std::collections::HashMap::new(); row.insert("a".to_string(), nodedb_types::Value::Integer(7)); @@ -3114,8 +3153,12 @@ mod tests { let txn = TxnId::new(44); let surrogate = 13u32; - src.txn_overlay_mut(txn) - .insert_put(coll_key("kvc"), 1, &hex_key(b"k"), b"V".to_vec()); + src.txn_overlay_mut(txn).insert_put( + coll_key("kvc"), + 1, + &kv_row_identity(b"k"), + b"V".to_vec(), + ); let plans = [ kv_write_plan("kvc"), diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/kv.rs b/nodedb/src/data/executor/handlers/transaction/resolve/kv.rs index ca2390335..ef2db7d21 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/kv.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/kv.rs @@ -28,12 +28,13 @@ //! ## Determinism //! //! The overlay keys slots by surrogate in a `HashMap`, so entries are collected -//! into a `BTreeMap` keyed by the overlay doc-id (lowercase-hex of the KV key) -//! before emitting. Two replicas resolving the same transaction produce +//! into a `BTreeMap` keyed by the row's identity (`kv_row_identity` of the KV +//! key) before emitting. Two replicas resolving the same transaction produce //! byte-identical redo ops. use std::collections::BTreeMap; +use nodedb_types::RowIdentity; use nodedb_wal::record::RecordType; use crate::control::server::wal_dispatch_kv::encode::encode_kv_put; @@ -55,18 +56,18 @@ pub(super) fn serialize_kv_collection( collection: &str, ops: &mut Vec, ) -> crate::Result<()> { - let mut entries: BTreeMap = BTreeMap::new(); + let mut entries: BTreeMap<&RowIdentity, &Staged> = BTreeMap::new(); for (doc_id, staged) in overlay.iter_doc_entries_for_collection(coll_key) { - entries.insert(doc_id.to_string(), staged); + entries.insert(doc_id, staged); } for (doc_id, staged) in entries { - let key = unhex_key(&doc_id).ok_or_else(|| crate::Error::Internal { + let key = unhex_key(doc_id.as_str()).ok_or_else(|| crate::Error::Internal { detail: format!("kv resolve: overlay doc-id '{doc_id}' is not valid hex"), })?; match staged { Staged::Put(value) => { - let expire_at_ms = match overlay.get_ttl_by_doc_id(coll_key, &doc_id) { + let expire_at_ms = match overlay.get_ttl_by_doc_id(coll_key, doc_id) { Some(StagedTtl::ExpireAt(ms)) => Some(ms), Some(StagedTtl::Persist) | None => None, }; @@ -77,7 +78,7 @@ pub(super) fn serialize_kv_collection( // this entry through, which is not a shape to paper over. let surrogate = overlay - .surrogate_for_doc_id(coll_key, &doc_id) + .surrogate_for_doc_id(coll_key, doc_id) .ok_or_else(|| crate::Error::Internal { detail: format!( "kv resolve: overlay has no surrogate for staged doc-id '{doc_id}'" diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/body.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/body.rs index 58e8dc033..979a46b07 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/body.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/body.rs @@ -11,7 +11,8 @@ //! configured). use nodedb_physical::physical_plan::{StorageMode, UpdateValue}; -use nodedb_types::Surrogate; +use nodedb_types::columnar::StrictSchema; +use nodedb_types::{RowIdentity, StorageKey, Surrogate}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::generated; @@ -19,6 +20,27 @@ use crate::data::executor::handlers::merge_helpers::check_declared_pk_not_null; use crate::data::executor::{doc_format, strict_format}; use crate::types::TenantId; +/// The client identity of a stored row, by the rule INSERT mints it with. +/// +/// A schemaless body is MessagePack and is read as is. A strict body is a +/// Binary Tuple and is decoded through `strict_schema` first. A strict body +/// that fails to decode carries no readable identity column and yields the +/// decimal surrogate. +pub(in crate::data::executor) fn stored_row_identity( + body: &[u8], + strict_schema: Option<&StrictSchema>, + declared_primary_key: Option<&str>, + key: StorageKey, +) -> RowIdentity { + match strict_schema { + Some(schema) => match strict_format::binary_tuple_to_msgpack(body, schema) { + Some(msgpack) => RowIdentity::of_stored_row(&msgpack, declared_primary_key, key), + None => key.to_identity(), + }, + None => RowIdentity::of_stored_row(body, declared_primary_key, key), + } +} + impl CoreLoop { /// Encode a PointPut / PointInsert body into its stored form, mirroring /// `apply_point_put`'s encoding pipeline (generated columns, `_rowid` diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/context.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/context.rs index df314cf5c..70ce891bc 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/context.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/context.rs @@ -2,9 +2,7 @@ //! Shared routing context for a single staged point write. -use std::borrow::Cow; - -use nodedb_types::Surrogate; +use nodedb_types::{RowIdentity, Surrogate}; use crate::data::executor::task::ExecutionTask; use crate::types::{DatabaseId, TenantId, TxnId}; @@ -15,18 +13,16 @@ pub(super) type CollKey = (DatabaseId, TenantId, String); /// The invariant routing identity of one staged point write, bundled so the /// per-op helpers stay within a sane argument count. /// -/// `document_id` is the overlay's doc-id key: for Document ops it borrows -/// the plan's own document id; for KV ops (which have no document id) it -/// owns the [`hex_key`](super::stage_kv::hex_key)-encoded KV key instead -- -/// `Cow` lets both engines share this one context type without allocating -/// on the Document path or leaking on the KV path. +/// `document_id` is the row's client identity, the overlay's doc-id key. +/// A Document op carries the plan's resolved identity. A KV op carries +/// [`kv_row_identity`](super::stage_kv::kv_row_identity) of its raw key. pub(in crate::data::executor) struct StageCtx<'a> { pub task: &'a ExecutionTask, pub tid: u64, pub database_id: u64, pub txn_id: TxnId, pub collection: &'a str, - pub document_id: Cow<'a, str>, + pub document_id: RowIdentity, pub surrogate: Surrogate, pub coll_key: CollKey, } @@ -37,7 +33,7 @@ impl<'a> StageCtx<'a> { tid: u64, txn_id: TxnId, collection: &'a str, - document_id: impl Into>, + document_id: RowIdentity, surrogate: Surrogate, ) -> Self { let coll_key = ( @@ -51,7 +47,7 @@ impl<'a> StageCtx<'a> { database_id: task.request.database_id.as_u64(), txn_id, collection, - document_id: document_id.into(), + document_id, surrogate, coll_key, } diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/dispatch.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/dispatch.rs index f05737b98..536808569 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/dispatch.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/dispatch.rs @@ -4,6 +4,7 @@ //! path, compute its real affected-row count, and record it in the overlay. use nodedb_physical::physical_plan::{ColumnarOp, DocumentOp, GraphOp, SpatialOp, TimeseriesOp}; +use nodedb_types::RowIdentity; use super::constraint::OverlayPk; use super::context::StageCtx; @@ -214,6 +215,7 @@ impl CoreLoop { | PhysicalPlan::ClusterEvent(_) => return self.stage_not_point_write(task), }; + // The plan's `document_id` is the row's resolved client identity. match doc_op { DocumentOp::PointInsert { collection, @@ -223,8 +225,14 @@ impl CoreLoop { surrogate, .. } => { - let ctx = - StageCtx::new(task, tid, txn_id, collection.as_str(), document_id, *surrogate); + let ctx = StageCtx::new( + task, + tid, + txn_id, + collection.as_str(), + RowIdentity::from_user_key(document_id.as_str()), + *surrogate, + ); self.stage_point_insert(&ctx, value, *if_absent) } DocumentOp::PointPut { @@ -234,8 +242,14 @@ impl CoreLoop { surrogate, .. } => { - let ctx = - StageCtx::new(task, tid, txn_id, collection.as_str(), document_id, *surrogate); + let ctx = StageCtx::new( + task, + tid, + txn_id, + collection.as_str(), + RowIdentity::from_user_key(document_id.as_str()), + *surrogate, + ); self.stage_point_put(&ctx, value) } DocumentOp::PointDelete { @@ -245,8 +259,14 @@ impl CoreLoop { rls_write_check, .. } => { - let ctx = - StageCtx::new(task, tid, txn_id, collection.as_str(), document_id, *surrogate); + let ctx = StageCtx::new( + task, + tid, + txn_id, + collection.as_str(), + RowIdentity::from_user_key(document_id.as_str()), + *surrogate, + ); self.stage_point_delete(&ctx, rls_write_check) } DocumentOp::PointUpdate { @@ -258,8 +278,14 @@ impl CoreLoop { declared_primary_key, .. } => { - let ctx = - StageCtx::new(task, tid, txn_id, collection.as_str(), document_id, *surrogate); + let ctx = StageCtx::new( + task, + tid, + txn_id, + collection.as_str(), + RowIdentity::from_user_key(document_id.as_str()), + *surrogate, + ); self.stage_point_update(&ctx, updates, rls_write_check, declared_primary_key.as_deref()) } // Predicate UPDATE staged like a point update, resolved against @@ -299,6 +325,7 @@ impl CoreLoop { rls_write_check, // See the `BulkUpdate` arm: staging carries no delta. resolved_sum_targets: _, + declared_primary_key, } => self.stage_bulk_delete(StageBulkDeleteParams { task, tid, @@ -306,6 +333,7 @@ impl CoreLoop { collection: collection.as_str(), filter_bytes: filters, rls_write_check, + declared_primary_key: declared_primary_key.as_deref(), }), // `UPSERT INTO`: resolve the current body under base ∪ overlay, @@ -320,8 +348,14 @@ impl CoreLoop { rls_write_check, .. } => { - let ctx = - StageCtx::new(task, tid, txn_id, collection.as_str(), document_id, *surrogate); + let ctx = StageCtx::new( + task, + tid, + txn_id, + collection.as_str(), + RowIdentity::from_user_key(document_id.as_str()), + *surrogate, + ); self.stage_document_upsert(&ctx, value, on_conflict_updates, rls_write_check) } diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/mod.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/mod.rs index bfedb92ec..8e543a051 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/mod.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/mod.rs @@ -34,6 +34,7 @@ mod stage_spatial; mod stage_timeseries; mod stage_upsert; +pub(in crate::data::executor) use body::stored_row_identity; pub(in crate::data::executor) use context::StageCtx; pub(in crate::data::executor) use stage_bulk_delete::StageBulkDeleteParams; pub(in crate::data::executor) use stage_bulk_update::StageBulkUpdateParams; @@ -45,6 +46,6 @@ pub(in crate::data::executor) use stage_columnar_resolved_dml::{ StageColumnarResolvedDeleteParams, StageColumnarResolvedUpdateParams, }; pub(in crate::data::executor) use stage_graph::GRAPH_LABEL_COLL_KEY; -pub(in crate::data::executor) use stage_kv::{hex_key, unhex_key}; +pub(in crate::data::executor) use stage_kv::{kv_row_identity, unhex_key}; pub(in crate::data::executor) use stage_spatial::StageSpatialInsertParams; pub(in crate::data::executor) use stage_timeseries::StageTimeseriesInsertParams; diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_delete.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_delete.rs index 59e1455ed..3584e5dfb 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_delete.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_delete.rs @@ -14,6 +14,7 @@ use nodedb_types::StorageKey; +use super::body::stored_row_identity; use crate::bridge::envelope::{ErrorCode, Response}; use crate::bridge::scan_filter::ScanFilter; use crate::data::executor::core_loop::CoreLoop; @@ -32,6 +33,9 @@ pub(in crate::data::executor) struct StageBulkDeleteParams<'a> { /// Compiled RLS write policy gating each matched row's removal, decided /// against its pre-deletion image. pub rls_write_check: &'a nodedb_types::RlsWriteCheck, + /// Declared `PRIMARY KEY` column of the collection, `None` otherwise. + /// Names the column each removed row's identity is read from. + pub declared_primary_key: Option<&'a str>, } impl CoreLoop { @@ -50,6 +54,7 @@ impl CoreLoop { collection, filter_bytes, rls_write_check, + declared_primary_key, } = params; let database_id = task.request.database_id; let coll_key: (DatabaseId, TenantId, String) = @@ -112,16 +117,31 @@ impl CoreLoop { // it already hidden from the rest of the transaction. Each row's // current BASE ∪ OVERLAY body is the pre-deletion image the policy // decides — the only image a delete has. + // Each row's identity is read from the pre-image the delete removes, + // by the same rule INSERT minted it with. + let strict_schema = self.resolve_strict_schema(database_id.as_u64(), tid, collection); + let identified: Vec<(StorageKey, nodedb_types::RowIdentity, &Vec)> = rows + .iter() + .map(|(row_key, body)| { + let identity = stored_row_identity( + body, + strict_schema.as_ref(), + declared_primary_key, + *row_key, + ); + (*row_key, identity, body) + }) + .collect(); + if !matches!( rls_write_check.decision(), nodedb_types::WriteGateDecision::AdmitAll ) { - for (row_key, body) in &rows { - let identity = row_key.to_identity(); + for (_, identity, body) in &identified { if let Err(e) = self.stage_admit_write( rls_write_check, body, - &identity, + identity, database_id.as_u64(), tid, collection, @@ -132,13 +152,10 @@ impl CoreLoop { } let mut affected = 0u64; - for (row_key, _body) in &rows { + for (row_key, identity, _body) in &identified { let surrogate = row_key.surrogate().as_u32(); - self.txn_overlay_mut(txn_id).insert_tombstone( - coll_key.clone(), - surrogate, - &row_key.to_string(), - ); + self.txn_overlay_mut(txn_id) + .insert_tombstone(coll_key.clone(), surrogate, identity); affected += 1; } diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs index f98d1162e..c470b822b 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_update.rs @@ -14,8 +14,9 @@ //! replay remains the sole durable apply. use nodedb_physical::physical_plan::UpdateValue; -use nodedb_types::StorageKey; +use nodedb_types::{RowIdentity, StorageKey}; +use super::body::stored_row_identity; use crate::bridge::envelope::{ErrorCode, Response}; use crate::bridge::scan_filter::ScanFilter; use crate::data::executor::core_loop::CoreLoop; @@ -130,6 +131,7 @@ impl CoreLoop { } } + let strict_schema = self.resolve_strict_schema(database_id.as_u64(), tid, collection); let mut affected = 0u64; for (row_key, current_body) in &rows { let surrogate = row_key.surrogate().as_u32(); @@ -148,7 +150,12 @@ impl CoreLoop { // policy. A rejected row fails the statement rather than being // skipped: skipping would under-report `affected` while the rest of // the predicate's matches were still rewritten. - let identity = row_key.to_identity(); + let identity = stored_row_identity( + &new_body, + strict_schema.as_ref(), + declared_primary_key, + *row_key, + ); if let Err(e) = self.stage_admit_write( rls_write_check, &new_body, @@ -159,13 +166,9 @@ impl CoreLoop { ) { return self.response_error(task, e); } - if let Err(e) = self.stage_bulk_put_capped( - txn_id, - &coll_key, - surrogate, - &row_key.to_string(), - new_body, - ) { + if let Err(e) = + self.stage_bulk_put_capped(txn_id, &coll_key, surrogate, &identity, new_body) + { return self.response_error(task, e); } affected += 1; @@ -220,7 +223,7 @@ impl CoreLoop { txn_id: TxnId, coll_key: &(DatabaseId, TenantId, String), surrogate: u32, - doc_id: &str, + doc_id: &RowIdentity, body: Vec, ) -> crate::Result<()> { let current = self diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar.rs index 7bb626162..4193003a7 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar.rs @@ -9,10 +9,10 @@ //! plan is still replayed through `execute_columnar_insert` inside the //! COMMIT `TransactionBatch`, which remains the sole durable apply. //! -//! Row identity: a columnar row has no separate primary document id — it is -//! surrogate-identified. The overlay's doc-id side-map therefore uses -//! `surrogate_to_doc_id` (hex), matching the identity `execute_columnar_scan` -//! uses for its own rows (`scan_memtable_rows_with_surrogates`). +//! Row identity: the overlay's identity side-map holds the row's primary-key +//! value when the schema declares one, else the decimal surrogate +//! (`columnar_row_identity`). The overlay slot itself is the surrogate, +//! matching `execute_columnar_scan` (`scan_memtable_rows_with_surrogates`). //! //! Row body encoding: each row's schema-ordered `Vec` is wrapped as a //! `Value::Array` and encoded via `nodedb_types::value_to_msgpack` — decoded @@ -48,11 +48,11 @@ use nodedb_types::columnar::schema::{TS_SYSTEM, TS_VALID_FROM, TS_VALID_UNTIL}; use nodedb_types::value::Value; use super::context::StageCtx; +use super::stage_columnar_dml::columnar_row_identity; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::columnar_write::ndb_field_to_value; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; use crate::types::{TenantId, TxnId}; /// Inputs for [`CoreLoop::stage_columnar_insert`]. Bundled because the raw @@ -259,6 +259,7 @@ impl CoreLoop { let mut staged = 0usize; for (surrogate, values) in resolved { + let identity = columnar_row_identity(&schema, &values, surrogate.as_u32()); let body = match nodedb_types::value_to_msgpack(&Value::Array(values)) { Ok(b) => b, Err(e) => { @@ -271,8 +272,7 @@ impl CoreLoop { } }; - let doc_id = surrogate_to_doc_id(surrogate); - let ctx = StageCtx::new(task, tid, txn_id, collection, doc_id, surrogate); + let ctx = StageCtx::new(task, tid, txn_id, collection, identity, surrogate); if let Err(e) = self.stage_put_capped(&ctx, body) { return self.response_error(task, e); } diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_dml.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_dml.rs index f73723a17..6c9e09b5d 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_dml.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_dml.rs @@ -43,9 +43,9 @@ //! handlers resolve their matching set, so the in-transaction view matches the //! post-commit view. -use nodedb_types::Surrogate; use nodedb_types::columnar::ColumnarSchema; use nodedb_types::value::Value; +use nodedb_types::{RowIdentity, Surrogate, value_to_pk_string}; use crate::bridge::envelope::{ErrorCode, Response}; use crate::bridge::scan_filter::ScanFilter; @@ -55,9 +55,25 @@ use crate::data::executor::handlers::columnar_read::filter::row_matches_filters; use crate::data::executor::handlers::transaction::overlay::ColumnarOverlayMergeParams; use crate::data::executor::response_codec; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; use crate::types::{TenantId, TxnId}; +/// The client identity of a columnar row: the schema's primary-key value +/// when the row carries one that stringifies, else the decimal surrogate. +pub(super) fn columnar_row_identity( + schema: &ColumnarSchema, + row: &[Value], + surrogate: u32, +) -> RowIdentity { + schema + .columns + .iter() + .position(|c| c.primary_key) + .and_then(|idx| row.get(idx)) + .and_then(value_to_pk_string) + .map(RowIdentity::from_user_key) + .unwrap_or_else(|| RowIdentity::for_surrogate(Surrogate::new(surrogate))) +} + /// Routing identity + payload for one staged columnar predicate `DELETE`. pub(in crate::data::executor) struct StageColumnarDeleteParams<'a> { pub task: &'a ExecutionTask, @@ -109,6 +125,11 @@ impl CoreLoop { collection.to_string(), ); + let schema = match self.columnar_engine_schema(task, tid, collection) { + Ok(s) => s, + Err(resp) => return resp, + }; + let affected_rows = match self.columnar_txn_matching_rows(task, tid, txn_id, collection, filter_bytes) { Ok(rows) => rows, @@ -121,28 +142,22 @@ impl CoreLoop { if !matches!( rls_write_check.decision(), nodedb_types::WriteGateDecision::AdmitAll + ) && let Err(response) = self.stage_admit_columnar_rows( + task, + rls_write_check, + affected_rows.iter().map(|(_, row)| row.as_slice()), + &schema, + tid, + collection, ) { - let schema = match self.columnar_engine_schema(task, tid, collection) { - Ok(s) => s, - Err(resp) => return resp, - }; - if let Err(response) = self.stage_admit_columnar_rows( - task, - rls_write_check, - affected_rows.iter().map(|(_, row)| row.as_slice()), - &schema, - tid, - collection, - ) { - return response; - } + return response; } let affected = affected_rows.len(); - for (surrogate, _row) in affected_rows { - let doc_id = surrogate_to_doc_id(Surrogate::new(surrogate)); + for (surrogate, row) in affected_rows { + let identity = columnar_row_identity(&schema, &row, surrogate); self.txn_overlay_mut(txn_id) - .insert_tombstone(coll_key.clone(), surrogate, &doc_id); + .insert_tombstone(coll_key.clone(), surrogate, &identity); } self.stage_columnar_dml_response(task, affected) @@ -214,6 +229,7 @@ impl CoreLoop { } for (surrogate, new_row) in new_rows { + let identity = columnar_row_identity(&schema, &new_row, surrogate); let body = match nodedb_types::value_to_msgpack(&Value::Array(new_row)) { Ok(b) => b, Err(e) => { @@ -225,8 +241,8 @@ impl CoreLoop { ); } }; - let doc_id = surrogate_to_doc_id(Surrogate::new(surrogate)); - if let Err(e) = self.stage_bulk_put_capped(txn_id, &coll_key, surrogate, &doc_id, body) + if let Err(e) = + self.stage_bulk_put_capped(txn_id, &coll_key, surrogate, &identity, body) { return self.response_error(task, e); } diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_resolved_dml.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_resolved_dml.rs index 4a6485196..13cdc7e82 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_resolved_dml.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_resolved_dml.rs @@ -38,16 +38,29 @@ use std::collections::HashMap; use nodedb_columnar::pk_index::encode_pk; use nodedb_query::scan_filter::FilterOp; -use nodedb_types::Surrogate; use nodedb_types::value::Value; +use nodedb_types::{RowIdentity, value_to_pk_string}; use crate::bridge::envelope::{ErrorCode, Response}; use crate::bridge::scan_filter::ScanFilter; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; use crate::types::{TenantId, TxnId}; +/// The client identity of a resolved columnar row: its shipped primary-key +/// value. The PK is mandatory on the resolved variants, so a value with no +/// primary-key string form is a plan error, never a fallback. +fn resolved_pk_identity(collection: &str, pk: &Value) -> crate::Result { + value_to_pk_string(pk) + .map(RowIdentity::from_user_key) + .ok_or_else(|| crate::Error::PlanError { + detail: format!( + "columnar resolved DML on '{collection}': primary key value {pk:?} has no \ + primary-key string form" + ), + }) +} + /// Routing identity + payload for a staged columnar `ResolvedUpdate`. pub(in crate::data::executor) struct StageColumnarResolvedUpdateParams<'a> { pub task: &'a ExecutionTask, @@ -131,10 +144,14 @@ impl CoreLoop { for (surrogate, row) in &matched { by_pk.insert(encode_pk(&row[pk_idx]), *surrogate); } - let mut resolved: Vec<(u32, &Vec)> = Vec::with_capacity(rows.len()); + let mut resolved: Vec<(u32, RowIdentity, &Vec)> = Vec::with_capacity(rows.len()); for (pk, new_row) in rows { + let identity = match resolved_pk_identity(collection, pk) { + Ok(identity) => identity, + Err(e) => return self.response_error(task, e), + }; match by_pk.get(&encode_pk(pk)) { - Some(surrogate) => resolved.push((*surrogate, new_row)), + Some(surrogate) => resolved.push((*surrogate, identity, new_row)), None => return self.response_error(task, ErrorCode::OllpRetryRequired), } } @@ -144,7 +161,7 @@ impl CoreLoop { if let Err(response) = self.stage_admit_columnar_rows( task, rls_write_check, - resolved.iter().map(|(_, row)| row.as_slice()), + resolved.iter().map(|(_, _, row)| row.as_slice()), &schema, tid, collection, @@ -153,7 +170,7 @@ impl CoreLoop { } let affected = resolved.len(); - for (surrogate, new_row) in resolved { + for (surrogate, identity, new_row) in resolved { let body = match nodedb_types::value_to_msgpack(&Value::Array(new_row.clone())) { Ok(b) => b, Err(e) => { @@ -165,8 +182,8 @@ impl CoreLoop { ); } }; - let doc_id = surrogate_to_doc_id(Surrogate::new(surrogate)); - if let Err(e) = self.stage_bulk_put_capped(txn_id, &coll_key, surrogate, &doc_id, body) + if let Err(e) = + self.stage_bulk_put_capped(txn_id, &coll_key, surrogate, &identity, body) { return self.response_error(task, e); } @@ -233,19 +250,22 @@ impl CoreLoop { // Drift check BEFORE tombstoning anything: every shipped PK must // resolve to a surrogate in the current in-transaction view. - let mut surrogates: Vec = Vec::with_capacity(pks.len()); + let mut surrogates: Vec<(u32, RowIdentity)> = Vec::with_capacity(pks.len()); for pk in pks { + let identity = match resolved_pk_identity(collection, pk) { + Ok(identity) => identity, + Err(e) => return self.response_error(task, e), + }; match by_pk.get(&encode_pk(pk)) { - Some(surrogate) => surrogates.push(*surrogate), + Some(surrogate) => surrogates.push((*surrogate, identity)), None => return self.response_error(task, ErrorCode::OllpRetryRequired), } } let affected = surrogates.len(); - for surrogate in surrogates { - let doc_id = surrogate_to_doc_id(Surrogate::new(surrogate)); + for (surrogate, identity) in surrogates { self.txn_overlay_mut(txn_id) - .insert_tombstone(coll_key.clone(), surrogate, &doc_id); + .insert_tombstone(coll_key.clone(), surrogate, &identity); } self.stage_columnar_dml_response(task, affected) diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs index 68a08073b..7ba817446 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs @@ -2,12 +2,12 @@ //! Statement-time staging for KV point puts: `Put`, `Insert`, //! `InsertIfAbsent`. Sibling files stage the rest of the fourteen -//! stageable `KvOp`s. A KV row's overlay doc-id is the lowercase-hex -//! encoding of its raw key ([`hex_key`]), applied symmetrically here and -//! in the read-merge paths. +//! stageable `KvOp`s. A KV row's overlay identity is +//! [`kv_row_identity`] of its raw key, applied symmetrically here and in +//! the read-merge paths. use nodedb_physical::physical_plan::KvOp; -use nodedb_types::Surrogate; +use nodedb_types::{RowIdentity, Surrogate}; use super::context::StageCtx; use crate::bridge::envelope::Response; @@ -17,10 +17,8 @@ use crate::data::executor::task::ExecutionTask; use crate::engine::kv::current_ms; use crate::types::TxnId; -/// Lowercase-hex encode a raw KV key for use as the overlay's doc-id. -/// Applied symmetrically here (stage) and in the read-merge paths that -/// resolve a KV key back to its overlay entry. -pub(in crate::data::executor) fn hex_key(key: &[u8]) -> String { +/// Lowercase-hex encode a raw KV key. [`unhex_key`] is the inverse. +fn hex_key(key: &[u8]) -> String { let mut s = String::with_capacity(key.len() * 2); for b in key { s.push_str(&format!("{b:02x}")); @@ -28,6 +26,13 @@ pub(in crate::data::executor) fn hex_key(key: &[u8]) -> String { s } +/// The overlay identity of a KV row: its raw key, hex encoded, taken +/// verbatim. Every KV staging writer and read-merge reader builds the +/// identity through this one function. +pub(in crate::data::executor) fn kv_row_identity(raw_key: &[u8]) -> RowIdentity { + RowIdentity::from_user_key(hex_key(raw_key)) +} + /// Decode a lowercase-hex KV overlay doc-id back to raw key bytes, the /// inverse of [`hex_key`]. Returns `None` for malformed hex. pub(in crate::data::executor) fn unhex_key(s: &str) -> Option> { @@ -179,8 +184,8 @@ impl CoreLoop { } } - /// Build the shared [`StageCtx`] routing bundle for a KV write, keying - /// the overlay's doc-id by [`hex_key`] rather than a document primary key. + /// Build the shared [`StageCtx`] routing bundle for a KV write, keyed by + /// [`kv_row_identity`] rather than a document primary key. fn kv_stage_ctx<'a>( &self, task: &'a ExecutionTask, @@ -190,9 +195,14 @@ impl CoreLoop { key: &[u8], surrogate: Surrogate, ) -> StageCtx<'a> { - // `StageCtx.document_id` is `Cow` so a KV row's overlay doc-id - // can be an owned hex string here, with no borrow from `task`. - StageCtx::new(task, tid, txn_id, collection, hex_key(key), surrogate) + StageCtx::new( + task, + tid, + txn_id, + collection, + kv_row_identity(key), + surrogate, + ) } // ── Put: upsert, no existence check ───────────────────────────────────── diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs index d42faaee1..db5445acd 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs @@ -40,7 +40,7 @@ use nodedb_physical::physical_plan::KvOp; use nodedb_types::Surrogate; use super::context::StageCtx; -use super::stage_kv::hex_key; +use super::stage_kv::kv_row_identity; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::transaction::overlay::StagedTtl; @@ -159,7 +159,7 @@ impl CoreLoop { collection: &'a str, key: &[u8], ) -> StageCtx<'a> { - let doc_id = hex_key(key); + let doc_id = kv_row_identity(key); let coll_key = ( task.request.database_id, crate::types::TenantId::new(tid), @@ -242,7 +242,7 @@ impl CoreLoop { self.stage_admit_write( rls_write_check, image, - &crate::engine::document::store::RowIdentity::from_user_key(ctx.document_id.as_ref()), + &ctx.document_id, ctx.database_id, ctx.tid, ctx.collection, diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_conflict.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_conflict.rs index d0a9e8ef7..9883e8ed1 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_conflict.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_conflict.rs @@ -94,7 +94,7 @@ impl CoreLoop { if let Err(e) = self.stage_admit_write( rls_write_check, &stored_bytes, - &crate::engine::document::store::RowIdentity::from_user_key(ctx.document_id.as_ref()), + &ctx.document_id, ctx.database_id, ctx.tid, ctx.collection, diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_delete.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_delete.rs index ce34fded4..b15e0a424 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_delete.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_delete.rs @@ -25,7 +25,7 @@ use crate::data::executor::task::ExecutionTask; use crate::engine::kv::current_ms; use crate::types::TxnId; -use super::stage_kv::hex_key; +use super::stage_kv::kv_row_identity; impl CoreLoop { /// Tombstone every present key in the overlay, after deciding each row it @@ -42,7 +42,7 @@ impl CoreLoop { let did = task.request.database_id; let mut deleted = 0usize; for key in keys { - let doc_id = hex_key(key); + let doc_id = kv_row_identity(key); let coll_key = ( did, crate::types::TenantId::new(tid), @@ -116,9 +116,7 @@ impl CoreLoop { && let Err(e) = self.stage_admit_write( rls_write_check, &body, - &crate::engine::document::store::RowIdentity::from_user_key( - doc_id.as_str(), - ), + &doc_id, did.as_u64(), tid, collection, diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_ttl.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_ttl.rs index 0ed287301..e1928b1b2 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_ttl.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_ttl.rs @@ -9,7 +9,7 @@ //! sibling to `Staged`, declared in `overlay::staged`) rather than the //! shared `Staged::Put`/`Tombstone` every engine's read-merge uses. A //! same-transaction `GetTtl` (`kv/ttl.rs::execute_kv_get_ttl`) consults this -//! same map keyed by the same [`super::stage_kv::hex_key`] identity. +//! same map keyed by the same [`super::stage_kv::kv_row_identity`]. //! //! Both handlers reuse [`CoreLoop::kv_atomic_stage_ctx`] (the same //! surrogate-resolution `Incr` / `Cas` / `GetSet` use) to bind a stable diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_spatial.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_spatial.rs index 5ce90170e..d087cafd6 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_spatial.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_spatial.rs @@ -25,8 +25,8 @@ //! plan is still replayed through `execute_spatial_insert` / //! `execute_spatial_delete` inside the COMMIT `TransactionBatch`. -use nodedb_types::Surrogate; use nodedb_types::geometry::Geometry; +use nodedb_types::{RowIdentity, Surrogate}; use super::context::StageCtx; use crate::bridge::envelope::{ErrorCode, Response}; @@ -88,7 +88,14 @@ impl CoreLoop { } }; - let ctx = StageCtx::new(task, tid, txn_id, collection, doc_id, surrogate); + let ctx = StageCtx::new( + task, + tid, + txn_id, + collection, + RowIdentity::for_surrogate(surrogate), + surrogate, + ); if let Err(e) = self.stage_put_capped(&ctx, body) { return self.response_error(task, e); } @@ -105,8 +112,14 @@ impl CoreLoop { collection: &str, surrogate: Surrogate, ) -> Response { - let doc_id = surrogate_to_doc_id(surrogate); - let ctx = StageCtx::new(task, tid, txn_id, collection, doc_id, surrogate); + let ctx = StageCtx::new( + task, + tid, + txn_id, + collection, + RowIdentity::for_surrogate(surrogate), + surrogate, + ); self.txn_overlay_mut(ctx.txn_id).insert_tombstone( ctx.coll_key.clone(), ctx.surrogate.0, diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries.rs index f2a5cc40b..5fb54863f 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries.rs @@ -28,8 +28,8 @@ //! (a hash of measurement + tags), which is not a cross-engine surrogate. For //! staging, each row is keyed by the per-row `Surrogate` the planner minted //! via `assign_fresh` (`convert_timeseries_ingest`) — a fresh unique id per -//! row so every staged INSERT occupies its own overlay slot. `surrogate_to_doc_id` -//! (hex) is used only for the overlay's doc-id side-map. +//! row so every staged INSERT occupies its own overlay slot. The overlay's +//! identity side-map holds that surrogate's decimal `RowIdentity`. //! //! Row body encoding: each row's `{field => value}` map is stored VERBATIM //! (the exact column names the INSERT used) and encoded via @@ -45,14 +45,13 @@ use std::collections::HashMap; -use nodedb_types::Surrogate; use nodedb_types::value::Value; +use nodedb_types::{RowIdentity, Surrogate}; use super::context::StageCtx; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; use crate::types::TxnId; /// Inputs for [`CoreLoop::stage_timeseries_insert`]. Bundled because the raw @@ -204,8 +203,14 @@ impl CoreLoop { } }; - let doc_id = surrogate_to_doc_id(surrogate); - let ctx = StageCtx::new(task, tid, txn_id, collection, doc_id, surrogate); + let ctx = StageCtx::new( + task, + tid, + txn_id, + collection, + RowIdentity::for_surrogate(surrogate), + surrogate, + ); if let Err(e) = self.stage_put_capped(&ctx, body) { return self.response_error(task, e); } @@ -397,8 +402,14 @@ impl CoreLoop { .get(&txn_id) .map(|overlay| overlay.journal_len()); for (surrogate, body) in surrogates.iter().copied().zip(encoded_rows) { - let doc_id = surrogate_to_doc_id(surrogate); - let ctx = StageCtx::new(task, tid, txn_id, collection, doc_id, surrogate); + let ctx = StageCtx::new( + task, + tid, + txn_id, + collection, + RowIdentity::for_surrogate(surrogate), + surrogate, + ); if let Err(error) = self.stage_put_capped(&ctx, body) { self.rollback_canonical_ilp_stage(txn_id, prior_marker); return self.response_error(task, error); @@ -633,7 +644,12 @@ mod tests { let existing = TxnId::new(81); { let overlay = core.txn_overlay_mut(existing); - overlay.insert_put(collection.clone(), 1, "1", b"prior".to_vec()); + overlay.insert_put( + collection.clone(), + 1, + &RowIdentity::from_user_key("1"), + b"prior".to_vec(), + ); } let prior_marker = core .txn_overlays @@ -642,8 +658,18 @@ mod tests { .journal_len(); { let overlay = core.txn_overlay_mut(existing); - overlay.insert_put(collection.clone(), 2, "2", b"first-new".to_vec()); - overlay.insert_put(collection.clone(), 3, "3", b"second-new".to_vec()); + overlay.insert_put( + collection.clone(), + 2, + &RowIdentity::from_user_key("2"), + b"first-new".to_vec(), + ); + overlay.insert_put( + collection.clone(), + 3, + &RowIdentity::from_user_key("3"), + b"second-new".to_vec(), + ); } core.rollback_canonical_ilp_stage(existing, Some(prior_marker)); let overlay = core @@ -662,8 +688,18 @@ mod tests { let created = TxnId::new(82); { let overlay = core.txn_overlay_mut(created); - overlay.insert_put(collection.clone(), 4, "4", b"created-a".to_vec()); - overlay.insert_put(collection, 5, "5", b"created-b".to_vec()); + overlay.insert_put( + collection.clone(), + 4, + &RowIdentity::from_user_key("4"), + b"created-a".to_vec(), + ); + overlay.insert_put( + collection, + 5, + &RowIdentity::from_user_key("5"), + b"created-b".to_vec(), + ); } assert_eq!(metrics.active_txn_overlays.load(Ordering::Relaxed), 2); core.rollback_canonical_ilp_stage(created, None); diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs index 8fb1ec83a..1d100b7d3 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs @@ -66,7 +66,7 @@ impl CoreLoop { if let Err(e) = self.stage_admit_write( rls_write_check, &stored_bytes, - &crate::engine::document::store::RowIdentity::from_user_key(ctx.document_id.as_ref()), + &ctx.document_id, ctx.database_id, ctx.tid, ctx.collection, diff --git a/nodedb/tests/inproc/cases/executor_tests/test_ollp_verification.rs b/nodedb/tests/inproc/cases/executor_tests/test_ollp_verification.rs index 05cd8be42..9948252af 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_ollp_verification.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_ollp_verification.rs @@ -166,6 +166,7 @@ fn bulk_delete_plan(predicted: Option>) -> PhysicalPlan { rls_filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, resolved_sum_targets: Vec::new(), + declared_primary_key: None, }) } @@ -187,6 +188,7 @@ fn bulk_delete_plan_with_edges( rls_filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, resolved_sum_targets: Vec::new(), + declared_primary_key: None, }) } diff --git a/nodedb/tests/inproc/cases/trigger_execution.rs b/nodedb/tests/inproc/cases/trigger_execution.rs index b720000e7..3a9a222e6 100644 --- a/nodedb/tests/inproc/cases/trigger_execution.rs +++ b/nodedb/tests/inproc/cases/trigger_execution.rs @@ -212,6 +212,7 @@ fn classify_bulk_delete() { rls_filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, resolved_sum_targets: Vec::new(), + declared_primary_key: None, }); let info = classify_dml_write(&plan).unwrap(); assert_eq!(info.collection, "logs"); From 0debd4b2f3001ce6072788b84ca94a2f209114b8 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 12 Sep 2026 14:08:33 +0800 Subject: [PATCH 13/17] refactor(executor): thread StorageKey through the write-apply path Replace document_id string params and surrogate_to_doc_id conversions with StorageKey across point put/delete/update, upsert, merge, bulk DML, transaction undo/overlay, WAL dispatch, and replay handlers. StorageKey is derived once at the entry point and passed through, removing redundant surrogate-to-string conversions along call chains. --- nodedb/src/control/insert_select/copy_rows.rs | 8 +- .../planner/materialized_sum/settle.rs | 4 +- nodedb/src/control/server/sync/fts_handler.rs | 6 +- .../src/control/server/wal_dispatch/text.rs | 6 +- .../server/wal_dispatch/write_set_redo.rs | 18 +- .../server/wal_dispatch_fts_spatial.rs | 8 +- .../core_loop/vector_index_rebuild.rs | 5 +- .../executor/dispatch/array/surrogate_scan.rs | 4 +- .../enforcement/materialized_sum/rmw.rs | 4 +- .../handlers/bulk_dml/delete_cascade.rs | 2 +- .../data/executor/handlers/bulk_dml/update.rs | 3 +- .../handlers/control/crdt_materialize.rs | 3 +- .../handlers/document/resolve/apply_row.rs | 6 +- .../handlers/document/write/batch_insert.rs | 2 +- .../executor/handlers/merge/target_docs.rs | 17 +- .../merge_orchestrated/apply/insert_rows.rs | 9 +- .../merge_orchestrated/apply/update_rows.rs | 204 ++++++++---------- .../merge_orchestrated/apply_support.rs | 23 +- .../merge_orchestrated/delete_arms.rs | 189 +++++++--------- .../handlers/merge_orchestrated/plan.rs | 28 +-- .../handlers/merge_orchestrated/resolve.rs | 6 +- .../executor/handlers/point/apply_delete.rs | 8 +- .../executor/handlers/point/apply_put/core.rs | 16 +- .../handlers/point/apply_put/index.rs | 6 +- .../handlers/point/apply_put/sparse.rs | 66 +++--- .../handlers/point/apply_put/types.rs | 3 +- .../handlers/point/apply_put/vector/put.rs | 24 ++- .../handlers/point/apply_put/vector/remove.rs | 10 +- .../handlers/point/apply_put/vector/types.rs | 2 +- .../data/executor/handlers/point/insert.rs | 3 +- .../src/data/executor/handlers/point/put.rs | 4 +- .../executor/handlers/point/update/exec.rs | 3 +- .../point/update_reindex_secondary.rs | 14 +- .../handlers/point/update_reindex_sparse.rs | 11 +- .../handlers/point/update_reindex_vector.rs | 15 +- .../data/executor/handlers/spatial_sync.rs | 10 +- .../src/data/executor/handlers/text_search.rs | 2 +- .../transaction/overlay/spatial_merge.rs | 6 +- .../transaction/stage_write/stage_spatial.rs | 4 +- .../handlers/transaction/sub_plan_doc/put.rs | 4 +- .../handlers/transaction/undo/graph_node.rs | 8 +- .../handlers/transaction/undo/rollback.rs | 4 +- nodedb/src/data/executor/handlers/truncate.rs | 2 +- .../handlers/update_from_join_write.rs | 3 +- .../executor/handlers/upsert/exec/dispatch.rs | 5 - .../executor/handlers/upsert/exec/insert.rs | 4 +- .../handlers/upsert/exec/overwrite.rs | 6 +- .../src/data/executor/handlers/write_batch.rs | 4 +- .../executor/wal_replay_document_vector.rs | 20 +- .../data/executor/wal_replay_redo_document.rs | 13 +- 50 files changed, 395 insertions(+), 440 deletions(-) diff --git a/nodedb/src/control/insert_select/copy_rows.rs b/nodedb/src/control/insert_select/copy_rows.rs index 79d9540e1..ddfcf1b71 100644 --- a/nodedb/src/control/insert_select/copy_rows.rs +++ b/nodedb/src/control/insert_select/copy_rows.rs @@ -19,7 +19,7 @@ use crate::control::state::SharedState; use crate::control::target_identity::{ TargetPk, assign_target_surrogate, bare_collection_name, resolve_target_pk, }; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::StorageKey; /// Resolved, per-statement copy context shared across every scanned page. pub(crate) struct CopySpec { @@ -125,7 +125,11 @@ pub(crate) fn assign_page_rows( &spec.target_pk, &value, )?; - out.push((surrogate_to_doc_id(surrogate), value, surrogate)); + out.push(( + StorageKey::for_surrogate(surrogate).to_string(), + value, + surrogate, + )); *remaining -= 1; } Ok(out) diff --git a/nodedb/src/control/planner/materialized_sum/settle.rs b/nodedb/src/control/planner/materialized_sum/settle.rs index c13b4c7f5..2fbf2521a 100644 --- a/nodedb/src/control/planner/materialized_sum/settle.rs +++ b/nodedb/src/control/planner/materialized_sum/settle.rs @@ -71,7 +71,7 @@ use nodedb_types::id::TxnId; use crate::control::server::shared::session::read_set::{ EngineTag, ReadKey, ReadOrigin, ReadSetEntry, }; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::StorageKey; use crate::query::{sum_target_is_co_resident, sum_target_vshard}; use crate::types::{DatabaseId, KeyRepr, Lsn, TenantId}; @@ -231,7 +231,7 @@ pub(super) fn balance_task(spec: BalanceTaskSpec<'_>) -> PhysicalTask { spec.database_id, &spec.binding.target_collection, ), - document_id: surrogate_to_doc_id(spec.surrogate), + document_id: StorageKey::for_surrogate(spec.surrogate).to_string(), surrogate: spec.surrogate, column: spec.binding.target_column.clone(), // The exact decimal, as a string: the balance is stored as one for diff --git a/nodedb/src/control/server/sync/fts_handler.rs b/nodedb/src/control/server/sync/fts_handler.rs index d19cfceb3..e8dc1187e 100644 --- a/nodedb/src/control/server/sync/fts_handler.rs +++ b/nodedb/src/control/server/sync/fts_handler.rs @@ -97,7 +97,8 @@ impl<'a> FtsDispatcher for SharedStateFtsDispatcher<'a> { // (same as what the DP uses for storage). We encode the original // doc_id (the Lite-side external key) into the WAL payload so replay // can re-derive the surrogate via the same assigner. - let surrogate_hex = crate::engine::document::store::surrogate_to_doc_id(surrogate); + let surrogate_hex = + crate::engine::document::store::StorageKey::for_surrogate(surrogate).to_string(); let fts_index_payload = nodedb_wal::record::FtsIndexPayload::new( prov.clone(), &collection, @@ -152,7 +153,8 @@ impl<'a> FtsDispatcher for SharedStateFtsDispatcher<'a> { &collection, )?; - let surrogate_hex = crate::engine::document::store::surrogate_to_doc_id(surrogate); + let surrogate_hex = + crate::engine::document::store::StorageKey::for_surrogate(surrogate).to_string(); let fts_delete_payload = nodedb_wal::record::FtsDeletePayload::new(prov.clone(), &collection, &surrogate_hex); let wal_lsn = wal_append_fts_delete( diff --git a/nodedb/src/control/server/wal_dispatch/text.rs b/nodedb/src/control/server/wal_dispatch/text.rs index 1cd8bd24b..20d27c9e2 100644 --- a/nodedb/src/control/server/wal_dispatch/text.rs +++ b/nodedb/src/control/server/wal_dispatch/text.rs @@ -36,7 +36,8 @@ pub(crate) fn wal_append_text_op( text, provenance, } => { - let doc_id = crate::engine::document::store::surrogate_to_doc_id(*surrogate); + let doc_id = + crate::engine::document::store::StorageKey::for_surrogate(*surrogate).to_string(); let prov = provenance.clone().unwrap_or_default(); let payload = nodedb_wal::record::FtsIndexPayload::new(prov, collection.as_str(), &doc_id, text); @@ -53,7 +54,8 @@ pub(crate) fn wal_append_text_op( surrogate, provenance, } => { - let doc_id = crate::engine::document::store::surrogate_to_doc_id(*surrogate); + let doc_id = + crate::engine::document::store::StorageKey::for_surrogate(*surrogate).to_string(); let prov = provenance.clone().unwrap_or_default(); let payload = nodedb_wal::record::FtsDeletePayload::new(prov, collection.as_str(), &doc_id); diff --git a/nodedb/src/control/server/wal_dispatch/write_set_redo.rs b/nodedb/src/control/server/wal_dispatch/write_set_redo.rs index 1138cc9fd..298eff040 100644 --- a/nodedb/src/control/server/wal_dispatch/write_set_redo.rs +++ b/nodedb/src/control/server/wal_dispatch/write_set_redo.rs @@ -8,7 +8,7 @@ //! Control Plane mints the durable redo here. use crate::bridge::envelope::{PhysicalPlan, Response, Status, WriteSetEntry}; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::StorageKey; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; use crate::wal::manager::WalManager; use nodedb_physical::physical_plan::DocumentOp; @@ -54,8 +54,8 @@ pub fn plan_post_apply_redo(plan: &PhysicalPlan) -> Option { } /// Append a document redo record for each write-set entry, returning the last -/// allocated LSN. Each entry is keyed by `surrogate_to_doc_id(surrogate)` so -/// replay keys on the same identity. Called under the write-admission guard. +/// allocated LSN. Each entry is keyed by `StorageKey::for_surrogate(surrogate)` +/// so replay keys on the same identity. Called under the write-admission guard. pub fn append_write_set_redo( wal: &WalManager, tenant_id: TenantId, @@ -67,7 +67,7 @@ pub fn append_write_set_redo( let mut last: Option = None; for entry in write_set { let entry_collection = entry.collection.as_deref().unwrap_or(collection); - let doc_id = surrogate_to_doc_id(Surrogate::new(entry.surrogate)); + let doc_id = StorageKey::for_surrogate(Surrogate::new(entry.surrogate)).to_string(); // A cross-collection entry homes to a different vShard, so it's re-derived // per entry rather than reusing the caller-hoisted `vshard_id`. let entry_vshard_id = match &entry.collection { @@ -272,7 +272,10 @@ mod tests { ) .expect("decode put payload"); assert_eq!(collection, "docs"); - assert_eq!(document_id, surrogate_to_doc_id(Surrogate::new(9))); + assert_eq!( + document_id, + StorageKey::for_surrogate(Surrogate::new(9)).to_string() + ); assert_eq!(value, vec![1, 2, 3]); assert_eq!(surrogate, 9); } @@ -303,7 +306,10 @@ mod tests { zerompk::from_msgpack::<(String, String, Option, u32)>(&record.payload) .expect("decode delete payload"); assert_eq!(collection, "docs"); - assert_eq!(document_id, surrogate_to_doc_id(Surrogate::new(9))); + assert_eq!( + document_id, + StorageKey::for_surrogate(Surrogate::new(9)).to_string() + ); assert_eq!(surrogate, 9); } diff --git a/nodedb/src/control/server/wal_dispatch_fts_spatial.rs b/nodedb/src/control/server/wal_dispatch_fts_spatial.rs index 5d74545bb..a277a2c21 100644 --- a/nodedb/src/control/server/wal_dispatch_fts_spatial.rs +++ b/nodedb/src/control/server/wal_dispatch_fts_spatial.rs @@ -22,8 +22,8 @@ use crate::wal::manager::WalManager; /// This is the ONE builder for the shape: both the sync `dispatch_insert` /// autocommit WAL append and the transaction-resolve serializer call it so /// producer and `replay_spatial_wal` never drift. `doc_id` is derived from -/// `surrogate` via `surrogate_to_doc_id`, matching the hex-encoded key both -/// the R-tree entry and the sparse document body are keyed by. +/// `surrogate` via `StorageKey::for_surrogate`, matching the hex-encoded key +/// both the R-tree entry and the sparse document body are keyed by. pub(crate) fn encode_spatial_put_payload( collection: &str, field: &str, @@ -31,7 +31,7 @@ pub(crate) fn encode_spatial_put_payload( geometry: &Geometry, provenance: &SyncProvenance, ) -> crate::Result { - let doc_id = crate::engine::document::store::surrogate_to_doc_id(surrogate); + let doc_id = crate::engine::document::store::StorageKey::for_surrogate(surrogate).to_string(); let geometry_bytes = zerompk::to_msgpack_vec(geometry).map_err(|e| crate::Error::Serialization { format: "msgpack".into(), @@ -54,7 +54,7 @@ pub(crate) fn encode_spatial_delete_payload( surrogate: Surrogate, provenance: &SyncProvenance, ) -> nodedb_wal::record::SpatialDeletePayload { - let doc_id = crate::engine::document::store::surrogate_to_doc_id(surrogate); + let doc_id = crate::engine::document::store::StorageKey::for_surrogate(surrogate).to_string(); nodedb_wal::record::SpatialDeletePayload::new(provenance.clone(), collection, field, doc_id) } diff --git a/nodedb/src/data/executor/core_loop/vector_index_rebuild.rs b/nodedb/src/data/executor/core_loop/vector_index_rebuild.rs index 0ba258ea5..0d1efbbbe 100644 --- a/nodedb/src/data/executor/core_loop/vector_index_rebuild.rs +++ b/nodedb/src/data/executor/core_loop/vector_index_rebuild.rs @@ -99,7 +99,8 @@ impl CoreLoop { let mut rebuilt = 0usize; for (surrogate, value) in docs { - let doc_id = crate::engine::document::store::surrogate_to_doc_id(surrogate); + let storage_key = + crate::engine::document::store::StorageKey::for_surrogate(surrogate); // Same as WAL replay: the document is already durable, so a // width mismatch from before the forward-path check existed is // reported and skipped rather than aborting the rebuild. @@ -107,7 +108,7 @@ impl CoreLoop { database_id: db, tid: tenant_id, collection: &collection, - document_id: &doc_id, + storage_key, surrogate, value: &value, wal_lsn: 0, diff --git a/nodedb/src/data/executor/dispatch/array/surrogate_scan.rs b/nodedb/src/data/executor/dispatch/array/surrogate_scan.rs index 5e2601f98..98572f31b 100644 --- a/nodedb/src/data/executor/dispatch/array/surrogate_scan.rs +++ b/nodedb/src/data/executor/dispatch/array/surrogate_scan.rs @@ -18,7 +18,7 @@ use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::response_codec::encode_raw_document_rows; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::StorageKey; impl CoreLoop { pub(in crate::data::executor) fn dispatch_array_surrogate_bitmap_scan( @@ -99,7 +99,7 @@ impl CoreLoop { if sur.as_u32() == 0 { continue; } - let hex = surrogate_to_doc_id(*sur); + let hex = StorageKey::for_surrogate(*sur).to_string(); // Empty msgpack map as the row body — the consumer // (`collect_surrogates`) only reads `id`. rows.push((hex, vec![0x80])); diff --git a/nodedb/src/data/executor/enforcement/materialized_sum/rmw.rs b/nodedb/src/data/executor/enforcement/materialized_sum/rmw.rs index a357c97ad..858696bd7 100644 --- a/nodedb/src/data/executor/enforcement/materialized_sum/rmw.rs +++ b/nodedb/src/data/executor/enforcement/materialized_sum/rmw.rs @@ -90,7 +90,7 @@ impl CoreLoop { params: &BalanceRmw<'_>, ) -> crate::Result { let storage_key = nodedb_types::StorageKey::for_surrogate(params.surrogate); - // `PointPutParams` still carries the document id as text. + // Rendered once, for the not-an-object error message below. let document_id = storage_key.to_string(); // The TARGET collection's encoding is resolved from `doc_configs`, not @@ -157,7 +157,7 @@ impl CoreLoop { database_id: params.database_id, tid: params.tid, collection: params.target_collection, - document_id: &document_id, + storage_key, surrogate: params.surrogate, value: &body, index_text: true, diff --git a/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs b/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs index da7551ae1..0ebb807e6 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs @@ -106,7 +106,7 @@ impl CoreLoop { // soft-delete those nodes and drop the reverse-map entry, or the // leaked vector keeps scoring in KNN search in the same process. if has_vectors { - self.remove_document_vector_indexes(database_id, tid, collection, doc_id); + self.remove_document_vector_indexes(database_id, tid, collection, storage_key); } self.doc_cache.invalidate( task.request.database_id.as_u64(), diff --git a/nodedb/src/data/executor/handlers/bulk_dml/update.rs b/nodedb/src/data/executor/handlers/bulk_dml/update.rs index 372de8e25..e4eb9c11b 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/update.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/update.rs @@ -349,8 +349,7 @@ impl CoreLoop { database_id, tid, collection, - row_key: doc_id, - surrogate, + storage_key, new_body: &updated_bytes, is_strict: strict_schema.is_some(), has_vectors, diff --git a/nodedb/src/data/executor/handlers/control/crdt_materialize.rs b/nodedb/src/data/executor/handlers/control/crdt_materialize.rs index b24d665ff..c56af9c9f 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_materialize.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_materialize.rs @@ -107,7 +107,6 @@ impl CoreLoop { ) { let database_id = task.request.database_id.as_u64(); let storage_key = StorageKey::for_surrogate(surrogate); - let row_key = storage_key.to_string(); let txn = match self.sparse.begin_write() { Ok(t) => t, @@ -123,7 +122,7 @@ impl CoreLoop { database_id, tid, collection, - document_id: row_key.as_str(), + storage_key, surrogate, value, index_text, diff --git a/nodedb/src/data/executor/handlers/document/resolve/apply_row.rs b/nodedb/src/data/executor/handlers/document/resolve/apply_row.rs index ee8ebf5a0..46710cf76 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/apply_row.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/apply_row.rs @@ -57,14 +57,12 @@ impl CoreLoop { } = put; let database_id = task.request.database_id.as_u64(); let storage_key = StorageKey::for_surrogate(surrogate); - let row_key = storage_key.to_string(); - let row_key = row_key.as_str(); let has_vectors = self.collection_has_vectors(database_id, tid, collection); // HNSW insert appends rather than replaces, so the prior embedding // must come out first or KNN keeps scoring both. if has_vectors && precondition.is_some() { - self.remove_document_vector_indexes(database_id, tid, collection, row_key); + self.remove_document_vector_indexes(database_id, tid, collection, storage_key); } let txn = self.sparse.begin_write().map_err(ErrorCode::from)?; @@ -74,7 +72,7 @@ impl CoreLoop { database_id, tid, collection, - document_id: row_key, + storage_key, surrogate, value, index_text: true, diff --git a/nodedb/src/data/executor/handlers/document/write/batch_insert.rs b/nodedb/src/data/executor/handlers/document/write/batch_insert.rs index 02443ff5b..cf6af8102 100644 --- a/nodedb/src/data/executor/handlers/document/write/batch_insert.rs +++ b/nodedb/src/data/executor/handlers/document/write/batch_insert.rs @@ -201,7 +201,7 @@ impl CoreLoop { database_id, tid, collection, - document_id: &row_key, + storage_key: key, surrogate, value: effective_value, index_text: true, diff --git a/nodedb/src/data/executor/handlers/merge/target_docs.rs b/nodedb/src/data/executor/handlers/merge/target_docs.rs index 8131565f4..2f4afbf61 100644 --- a/nodedb/src/data/executor/handlers/merge/target_docs.rs +++ b/nodedb/src/data/executor/handlers/merge/target_docs.rs @@ -25,18 +25,18 @@ impl CoreLoop { /// folds the transaction's staging overlay: a staged tombstone hides its base /// row, a staged put replaces the base body, and a staged put absent from /// base is appended — so an in-transaction MERGE resolved at COMMIT sees rows - /// staged by earlier statements in the same transaction. The `doc_id` this - /// produces is the storage key's text, matching the overlay's surrogate - /// keying, so staged and base bodies (same canonical stored form — Binary - /// Tuple for a strict target, MessagePack for a schemaless one) are merged - /// like-for-like and decoded identically downstream by `decode_target`. + /// staged by earlier statements in the same transaction. The `StorageKey` + /// this produces matches the overlay's surrogate keying, so staged and + /// base bodies (same canonical stored form — Binary Tuple for a strict + /// target, MessagePack for a schemaless one) are merged like-for-like and + /// decoded identically downstream by `decode_target`. pub(in crate::data::executor) fn collect_target_docs( &self, database_id: u64, tid: u64, collection: &str, txn_id: Option, - ) -> crate::Result)>> { + ) -> crate::Result)>> { let prefix = crate::engine::sparse::btree::coll_prefix(database_id, tid, collection); let end = format!("{prefix}\u{ffff}"); @@ -81,9 +81,6 @@ impl CoreLoop { ); self.merge_overlay_into_scan(txn_id, &coll_key, &mut docs, &|_, _| true); } - Ok(docs - .into_iter() - .map(|(key, body)| (key.to_string(), body)) - .collect()) + Ok(docs) } } diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/insert_rows.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/insert_rows.rs index fef8936d6..ede0e5691 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/insert_rows.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/insert_rows.rs @@ -14,7 +14,7 @@ use crate::data::executor::enforcement::write_hook; use crate::data::executor::handlers::point::apply_put::PointPutParams; use crate::data::executor::handlers::transaction::undo::UndoEntry; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::StorageKey; use nodedb_types::Surrogate; use super::super::abort::MergeAbort; @@ -99,7 +99,8 @@ impl CoreLoop { })); } }; - let row_key = surrogate_to_doc_id(surrogate); + let storage_key = StorageKey::for_surrogate(surrogate); + let row_key = storage_key.to_string(); applied_keys.push(row_key.clone()); match self.apply_point_put( txn, @@ -107,7 +108,7 @@ impl CoreLoop { database_id, tid, collection, - document_id: &row_key, + storage_key, surrogate, value: &ins.body, index_text: true, @@ -163,7 +164,7 @@ impl CoreLoop { }); } if returning { - match returning_doc(&ins.body, &row_key) { + match returning_doc(&ins.body, &storage_key) { Ok(doc) => returned_docs.push(doc), Err(e) => { return Err(self.abort_merge_apply(MergeAbort { diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs index 065997d27..e6694db60 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs @@ -12,7 +12,6 @@ use crate::data::executor::enforcement::write_hook; use crate::data::executor::handlers::point::apply_put::PointPutParams; use crate::data::executor::handlers::transaction::undo::UndoEntry; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; use super::super::abort::MergeAbort; use super::super::apply_support::{MergePutEvent, record_put_index_undo, returning_doc}; @@ -76,115 +75,70 @@ impl CoreLoop { } = tally; for upd in updates { - match upd.surrogate { - Some(surrogate) => { - let row_key = surrogate_to_doc_id(surrogate); - applied_keys.push(row_key.clone()); - // `apply_point_put`'s vector step APPENDS (it never replaces), - // so an in-place UPDATE must first soft-delete the surrogate's - // prior embedding or the stale vector keeps scoring in KNN - // search. Push each removal as a `DeleteVector` undo BEFORE the - // put's `InsertVector` undos so an abort undeletes the old - // vector after removing the new one (reverse order). - if has_vectors { - for d in self.remove_document_vector_indexes( - database_id, - tid, - collection, - &row_key, - ) { - undo_log.push(UndoEntry::DeleteVector { - index_key: d.index_key, - vector_id: d.vector_id, - collection: d.collection, - field: d.field, - doc_id: d.doc_id, - }); - } - } - match self.apply_point_put( + let surrogate = upd.key.surrogate(); + let row_key = upd.key.to_string(); + applied_keys.push(row_key.clone()); + // `apply_point_put`'s vector step APPENDS (it never replaces), + // so an in-place UPDATE must first soft-delete the surrogate's + // prior embedding or the stale vector keeps scoring in KNN + // search. Push each removal as a `DeleteVector` undo BEFORE the + // put's `InsertVector` undos so an abort undeletes the old + // vector after removing the new one (reverse order). + if has_vectors { + for d in self.remove_document_vector_indexes(database_id, tid, collection, upd.key) + { + undo_log.push(UndoEntry::DeleteVector { + index_key: d.index_key, + vector_id: d.vector_id, + collection: d.collection, + field: d.field, + doc_id: d.doc_id, + }); + } + } + match self.apply_point_put( + txn, + PointPutParams { + database_id, + tid, + collection, + storage_key: upd.key, + surrogate, + value: &upd.body, + index_text: true, + user_roles: &task.request.user_roles, + enforce: true, + wal_lsn: task.wal_lsn(), + resolved_targets: resolved_sum_targets, + }, + ) { + Ok(mut outcome) => { + record_put_index_undo(undo_log, &mut outcome); + // The arm's materialized-sum delta is folded inside + // the SAME transaction the arm's row lands in, so a + // moved total rolls back with the row that moved it. + // Both images come from the plan: the classifier held + // the pre-image already, so nothing is re-read. + match write_hook::run( + self, txn, - PointPutParams { + &write_hook::HookCtx { database_id, tid, collection, - document_id: &row_key, - surrogate, - value: &upd.body, - index_text: true, - user_roles: &task.request.user_roles, - enforce: true, - wal_lsn: task.wal_lsn(), resolved_targets: resolved_sum_targets, + deferred_sum_targets: &[], + wal_lsn: task.wal_lsn(), + }, + write_hook::WriteImages::Update { + old: write_hook::ImageBody::Submitted(&upd.old_body), + new: write_hook::ImageBody::Submitted(&upd.body), }, ) { - Ok(mut outcome) => { - record_put_index_undo(undo_log, &mut outcome); - // The arm's materialized-sum delta is folded inside - // the SAME transaction the arm's row lands in, so a - // moved total rolls back with the row that moved it. - // Both images come from the plan: the classifier held - // the pre-image already, so nothing is re-read. - match write_hook::run( - self, - txn, - &write_hook::HookCtx { - database_id, - tid, - collection, - resolved_targets: resolved_sum_targets, - deferred_sum_targets: &[], - wal_lsn: task.wal_lsn(), - }, - write_hook::WriteImages::Update { - old: write_hook::ImageBody::Submitted(&upd.old_body), - new: write_hook::ImageBody::Submitted(&upd.body), - }, - ) { - Ok(enforcement) => { - write_set.extend(write_hook::target_write_set( - &enforcement.target_writes, - )); - balanced_entries.extend(enforcement.balanced_entries); - } - Err(e) => { - return Err(self.abort_merge_apply(MergeAbort { - task, - database_id, - tid, - collection, - applied_keys: applied_keys.as_slice(), - undo_log: std::mem::take(undo_log), - err: e.into(), - })); - } - } - if has_vectors { - write_set.push(WriteSetEntry { - surrogate: surrogate.as_u32(), - is_delete: false, - value: upd.body.clone(), - collection: None, - }); - } - if returning { - match returning_doc(&upd.body, &row_key) { - Ok(doc) => returned_docs.push(doc), - Err(e) => { - return Err(self.abort_merge_apply(MergeAbort { - task, - database_id, - tid, - collection, - applied_keys: applied_keys.as_slice(), - undo_log: std::mem::take(undo_log), - err: e.into(), - })); - } - } - } - put_events.push((row_key, upd.body.as_slice(), outcome.prior_value)); - *affected += 1; + Ok(enforcement) => { + write_set + .extend(write_hook::target_write_set(&enforcement.target_writes)); + balanced_entries.extend(enforcement.balanced_entries); } Err(e) => { return Err(self.abort_merge_apply(MergeAbort { @@ -198,14 +152,34 @@ impl CoreLoop { })); } } + if has_vectors { + write_set.push(WriteSetEntry { + surrogate: surrogate.as_u32(), + is_delete: false, + value: upd.body.clone(), + collection: None, + }); + } + if returning { + match returning_doc(&upd.body, &upd.key) { + Ok(doc) => returned_docs.push(doc), + Err(e) => { + return Err(self.abort_merge_apply(MergeAbort { + task, + database_id, + tid, + collection, + applied_keys: applied_keys.as_slice(), + undo_log: std::mem::take(undo_log), + err: e.into(), + })); + } + } + } + put_events.push((row_key, upd.body.as_slice(), outcome.prior_value)); + *affected += 1; } - None => { - // A target row whose `doc_id` does not parse as a storage - // key: `put_in_txn` addresses DOCUMENTS rows by - // `StorageKey` only, and the workspace carries no - // on-disk-format compatibility burden for a row shape - // that predates surrogate keying, so this arm is refused - // rather than written through a raw string key. + Err(e) => { return Err(self.abort_merge_apply(MergeAbort { task, database_id, @@ -213,15 +187,7 @@ impl CoreLoop { collection, applied_keys: applied_keys.as_slice(), undo_log: std::mem::take(undo_log), - err: crate::Error::Storage { - engine: "document".into(), - detail: format!( - "MERGE UPDATE target row '{}' in '{collection}' has no \ - surrogate storage key", - upd.doc_id - ), - } - .into(), + err: e.into(), })); } } diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/apply_support.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/apply_support.rs index 63e81af1e..168e0c99e 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/apply_support.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/apply_support.rs @@ -7,6 +7,7 @@ use crate::data::executor::handlers::point::apply_put::PointPutOutcome; use crate::data::executor::handlers::rls_write_gate; use crate::data::executor::handlers::transaction::undo::UndoEntry; +use crate::engine::document::store::StorageKey; use super::plan::MergePlanActions; @@ -68,14 +69,10 @@ pub(super) fn gate_merge_arms( let doc_arms = plan .updates .iter() - .map(|u| (u.body.as_slice(), u.doc_id.as_str())) - .chain( - plan.deletes - .iter() - .map(|d| (d.body.as_slice(), d.doc_id.as_str())), - ); - for (body, doc_id) in doc_arms { - let identity = crate::engine::document::store::identity_of(doc_id); + .map(|u| (u.body.as_slice(), u.key)) + .chain(plan.deletes.iter().map(|d| (d.body.as_slice(), d.key))); + for (body, key) in doc_arms { + let identity = key.to_identity(); rls_write_gate::admit_stored_row(rls_write_check, body, &identity, None, tid, collection)?; } for insert in &plan.inserts { @@ -97,15 +94,15 @@ pub(super) fn gate_merge_arms( /// reads. Same shape the point and bulk DML RETURNING paths emit, so a MERGE /// row projects identically. /// -/// `doc_id` is the row's storage key, every caller's `MergeUpdate::doc_id`, -/// `MergeDelete::doc_id`, or a minted insert key from `surrogate_to_doc_id`. -/// This function converts it to the client-visible identity before decoding. +/// `key` is the row's storage key: every caller's `MergeUpdate::key`, +/// `MergeDelete::key`, or a freshly minted insert key. This function converts +/// it to the client-visible identity before decoding. /// /// The schema argument is `None` unconditionally: a merge plan's captured /// bodies are MessagePack for BOTH storage modes (`collect_merge_plan` decodes /// a strict target's Binary Tuple and re-encodes the resolved row before the /// apply pass ever sees it), so the strict decoder would have nothing to read. -pub(super) fn returning_doc(body: &[u8], doc_id: &str) -> crate::Result { - let identity = crate::engine::document::store::identity_of(doc_id); +pub(super) fn returning_doc(body: &[u8], key: &StorageKey) -> crate::Result { + let identity = key.to_identity(); super::super::returning_doc::from_stored(body, &identity, None) } diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs index 34369c884..f8d7526ce 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs @@ -17,7 +17,6 @@ use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::enforcement::write_hook; use crate::data::executor::handlers::point::apply_delete::PointDeleteParams; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::StorageKey; use super::apply_support::returning_doc; use super::plan::MergeDelete; @@ -72,124 +71,102 @@ impl CoreLoop { } = tally; for del in deletes { - match del.surrogate { - Some(surrogate) => { - // One write txn per arm: the removal and its index cascades - // commit together, and a failing arm drops the txn - // un-committed so it leaves nothing behind. - let txn = match self.sparse.begin_write() { - Ok(txn) => txn, - Err(e) => return Err(self.response_error(task, e)), - }; - match self.apply_point_delete( + let surrogate = del.key.surrogate(); + let row_key = del.key.to_string(); + // One write txn per arm: the removal and its index cascades + // commit together, and a failing arm drops the txn + // un-committed so it leaves nothing behind. + let txn = match self.sparse.begin_write() { + Ok(txn) => txn, + Err(e) => return Err(self.response_error(task, e)), + }; + match self.apply_point_delete( + &txn, + PointDeleteParams { + database_id, + tid, + collection, + document_id: &row_key, + surrogate, + user_roles: &task.request.user_roles, + enforce: true, + resolved_targets, + }, + ) { + Ok(outcome) => { + // A DELETE arm takes the removed row's contribution + // back off its target, folded inside THIS arm's + // transaction so the debit and the removal commit + // together. The pre-image is the plan's captured + // body — the only image a delete has. + match write_hook::run( + self, &txn, - PointDeleteParams { + &write_hook::HookCtx { database_id, tid, collection, - document_id: &del.doc_id, - surrogate, - user_roles: &task.request.user_roles, - enforce: true, resolved_targets, + deferred_sum_targets: &[], + wal_lsn: task.wal_lsn(), + }, + write_hook::WriteImages::Delete { + old: write_hook::ImageBody::Submitted(&del.body), }, ) { - Ok(outcome) => { - // A DELETE arm takes the removed row's contribution - // back off its target, folded inside THIS arm's - // transaction so the debit and the removal commit - // together. The pre-image is the plan's captured - // body — the only image a delete has. - match write_hook::run( - self, - &txn, - &write_hook::HookCtx { - database_id, - tid, - collection, - resolved_targets, - deferred_sum_targets: &[], - wal_lsn: task.wal_lsn(), - }, - write_hook::WriteImages::Delete { - old: write_hook::ImageBody::Submitted(&del.body), - }, - ) { - // The arm's BALANCED contribution is NOT settled - // here: these arms commit one transaction each, - // after the caller's phase-A commit, so a - // violation found here could no longer be - // undone. The caller accounts every delete - // arm's pre-image before phase A runs and - // judges the whole MERGE there. - Ok(enforcement) => write_set.extend(write_hook::target_write_set( - &enforcement.target_writes, - )), - // Dropping `txn` un-committed reverses the - // removal and every target it had debited. + // The arm's BALANCED contribution is NOT settled + // here: these arms commit one transaction each, + // after the caller's phase-A commit, so a + // violation found here could no longer be + // undone. The caller accounts every delete + // arm's pre-image before phase A runs and + // judges the whole MERGE there. + Ok(enforcement) => write_set + .extend(write_hook::target_write_set(&enforcement.target_writes)), + // Dropping `txn` un-committed reverses the + // removal and every target it had debited. + Err(e) => return Err(self.response_error(task, e)), + } + if let Err(e) = txn.commit() { + return Err(self.response_error( + task, + crate::Error::Storage { + engine: "sparse".into(), + detail: format!("merge delete commit: {e}"), + }, + )); + } + if outcome.prior_value.is_some() { + *affected += 1; + // A DELETE arm returns the PRE-image — the row + // as it was classified, since nothing survives + // the delete to project. Taken from the plan's + // captured body rather than `prior_value`, which + // is the raw stored form (Binary Tuple on a + // strict target) and would need re-decoding. + if returning { + match returning_doc(&del.body, &del.key) { + Ok(doc) => returned_docs.push(doc), Err(e) => return Err(self.response_error(task, e)), } - if let Err(e) = txn.commit() { - return Err(self.response_error( - task, - crate::Error::Storage { - engine: "sparse".into(), - detail: format!("merge delete commit: {e}"), - }, - )); - } - if outcome.prior_value.is_some() { - *affected += 1; - // A DELETE arm returns the PRE-image — the row - // as it was classified, since nothing survives - // the delete to project. Taken from the plan's - // captured body rather than `prior_value`, which - // is the raw stored form (Binary Tuple on a - // strict target) and would need re-decoding. - if returning { - match returning_doc(&del.body, &del.doc_id) { - Ok(doc) => returned_docs.push(doc), - Err(e) => return Err(self.response_error(task, e)), - } - } - if has_vectors { - write_set.push(WriteSetEntry { - surrogate: surrogate.as_u32(), - is_delete: true, - value: Vec::new(), - collection: None, - }); - } - } - self.emit_document_delete_event( - task, - collection, - StorageKey::for_surrogate(surrogate).to_identity(), - outcome.prior_value.as_deref(), - ); } - Err(e) => return Err(self.response_error(task, e)), + if has_vectors { + write_set.push(WriteSetEntry { + surrogate: surrogate.as_u32(), + is_delete: true, + value: Vec::new(), + collection: None, + }); + } } - } - None => { - // A target row whose `doc_id` does not parse as a storage - // key: `delete` addresses DOCUMENTS rows by `StorageKey` - // only, and the workspace carries no on-disk-format - // compatibility burden for a row shape that predates - // surrogate keying, so this arm is refused rather than - // removed through a raw string key. - return Err(self.response_error( + self.emit_document_delete_event( task, - crate::Error::Storage { - engine: "document".into(), - detail: format!( - "MERGE DELETE target row '{}' in '{collection}' has no \ - surrogate storage key", - del.doc_id - ), - }, - )); + collection, + del.key.to_identity(), + outcome.prior_value.as_deref(), + ); } + Err(e) => return Err(self.response_error(task, e)), } } Ok(()) diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/plan.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/plan.rs index be17625b1..2a11a1eb6 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/plan.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/plan.rs @@ -9,11 +9,10 @@ use std::collections::HashSet; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::doc_format; use crate::data::executor::doc_format::encode_resolved_wire_body as encode_doc_body; -use crate::engine::document::store::{RowIdentity, doc_id_to_surrogate}; +use crate::engine::document::store::{RowIdentity, StorageKey}; use nodedb_physical::physical_plan::document::merge_types::{ MergeActionOp, MergeClauseKind as MergeClauseKindOp, }; -use nodedb_types::Surrogate; use super::super::merge::MergeParams; use super::super::merge_helpers::{ @@ -22,11 +21,8 @@ use super::super::merge_helpers::{ /// A matched / not-matched-by-source UPDATE arm resolved to a rewrite. pub(super) struct MergeUpdate { - /// Existing target storage key (the surrogate hex). - pub(super) doc_id: String, - /// The row's registered surrogate, parsed from `doc_id`. `None` only for - /// legacy non-surrogate rows that predate surrogate-keyed storage. - pub(super) surrogate: Option, + /// Existing target row's storage key. + pub(super) key: StorageKey, /// Post-update document as MessagePack (pre-strict-encoding). pub(super) body: Vec, /// The target row as it stood BEFORE the arm, as MessagePack. An UPDATE @@ -37,8 +33,7 @@ pub(super) struct MergeUpdate { /// A matched / not-matched-by-source DELETE arm resolved to a removal. pub(super) struct MergeDelete { - pub(super) doc_id: String, - pub(super) surrogate: Option, + pub(super) key: StorageKey, /// The deleted target row as MessagePack, so the Control-Plane expander can /// extract its primary key when rewriting the delete into a concrete /// `PointDelete` for an in-transaction MERGE at COMMIT. @@ -123,12 +118,9 @@ impl CoreLoop { // matching the legacy walk's `&serde_json::Value::Null`. let null_source = serde_json::Value::Null; - for (doc_id, bytes) in &target_docs { - let surrogate = doc_id_to_surrogate(doc_id); - let identity = match surrogate { - Some(s) => RowIdentity::for_surrogate(s), - None => RowIdentity::from_user_key(doc_id.clone()), - }; + for (key, bytes) in &target_docs { + let key = *key; + let identity = key.to_identity(); let target_doc = decode_target(&identity, bytes, &strict_schema)?; let join_val = target_doc .get(params.target_join_col) @@ -171,15 +163,13 @@ impl CoreLoop { pk, )?; updates.push(MergeUpdate { - doc_id: doc_id.clone(), - surrogate, + key, body: encode_doc_body(&updated), old_body: encode_doc_body(&target_doc), }); } MergeActionOp::Delete => deletes.push(MergeDelete { - doc_id: doc_id.clone(), - surrogate, + key, body: encode_doc_body(&target_doc), }), // INSERT is not a target-row arm; DoNothing is a no-op. diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/resolve.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/resolve.rs index e78d02bd1..1d12054ac 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/resolve.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/resolve.rs @@ -46,8 +46,8 @@ impl CoreLoop { .into_iter() .map(|u| { ( - u.doc_id, - u.surrogate.map(|s| s.as_u32()), + u.key.to_string(), + Some(u.key.surrogate().as_u32()), u.body, u.old_body, ) @@ -56,7 +56,7 @@ impl CoreLoop { let deletes: Vec<(String, Option, Vec)> = plan .deletes .into_iter() - .map(|d| (d.doc_id, d.surrogate.map(|s| s.as_u32()), d.body)) + .map(|d| (d.key.to_string(), Some(d.key.surrogate().as_u32()), d.body)) .collect(); let inserts: Vec<(String, Vec)> = plan .inserts diff --git a/nodedb/src/data/executor/handlers/point/apply_delete.rs b/nodedb/src/data/executor/handlers/point/apply_delete.rs index 1eef38f7f..d377b67b6 100644 --- a/nodedb/src/data/executor/handlers/point/apply_delete.rs +++ b/nodedb/src/data/executor/handlers/point/apply_delete.rs @@ -144,9 +144,9 @@ impl CoreLoop { } = params; let _ = user_roles; - let row_key = crate::engine::document::store::surrogate_to_doc_id(surrogate); - let row_key = row_key.as_str(); let storage_key = crate::engine::document::store::StorageKey::for_surrogate(surrogate); + let row_key = storage_key.to_string(); + let row_key = row_key.as_str(); let bitemporal = self.is_bitemporal(database_id, tid, collection); let config_key = ( crate::types::DatabaseId::new(database_id), @@ -401,13 +401,13 @@ impl CoreLoop { // looked up by its exact key rather than scanning the whole map on // every delete. Shared with the PointUpdate re-index path. let vector_deletes = - self.remove_document_vector_indexes(database_id, tid, collection, row_key); + self.remove_document_vector_indexes(database_id, tid, collection, storage_key); // Sparse inverted-index cleanup, mirroring the dense-vector cascade // above: drop this document's sparse posting entries under the same hex // surrogate row key the put path indexed them by. A no-op unless the // strict schema declares a `SparseVector` column. - self.remove_document_sparse_indexes(database_id, tid, collection, row_key); + self.remove_document_sparse_indexes(database_id, tid, collection, storage_key); // Invalidate document cache. self.doc_cache diff --git a/nodedb/src/data/executor/handlers/point/apply_put/core.rs b/nodedb/src/data/executor/handlers/point/apply_put/core.rs index bbdc93d7c..102602234 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/core.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/core.rs @@ -36,7 +36,7 @@ impl CoreLoop { database_id, tid, collection, - document_id, + storage_key, surrogate, value, index_text, @@ -45,8 +45,6 @@ impl CoreLoop { wal_lsn, resolved_targets, } = params; - // `surrogate` IS the storage key: no parse, no failure arm. - let storage_key = crate::engine::document::store::StorageKey::for_surrogate(surrogate); let config_key = ( crate::types::DatabaseId::new(database_id), crate::types::TenantId::new(tid), @@ -191,7 +189,7 @@ impl CoreLoop { // Recorded here, at the detection site — an fsync'd // report survives a restart, unlike a log line. crate::diag::fts_index_update_failed(&e, collection, surrogate.as_u32()); - warn!(core = self.core_id, %collection, %document_id, error = %e, "inverted index update failed; rejecting the write"); + warn!(core = self.core_id, %collection, %storage_key, error = %e, "inverted index update failed; rejecting the write"); return Err(e); } } @@ -298,20 +296,20 @@ impl CoreLoop { } let spatial_inserts = - self.apply_point_put_spatial(database_id, tid, collection, document_id, value); + self.apply_point_put_spatial(database_id, tid, collection, storage_key, value); let vector_inserts = self.apply_point_put_vector_indexes( crate::data::executor::handlers::point::apply_put::VectorIndexPutParams { database_id, tid, collection, - document_id, + storage_key, surrogate, value, wal_lsn: wal_lsn.map(|l| l.as_u64()).unwrap_or(0), }, )?; // No-op unless the strict schema declares a `SparseVector` column. - self.apply_point_put_sparse_indexes(database_id, tid, collection, document_id, value); + self.apply_point_put_sparse_indexes(database_id, tid, collection, storage_key, value); Ok(PointPutOutcome { prior_value: prior, @@ -500,7 +498,7 @@ mod tests { database_id: DatabaseId::DEFAULT.as_u64(), tid: TID, collection: COLL, - document_id: &row_key, + storage_key: crate::engine::document::store::StorageKey::for_surrogate(SURROGATE), surrogate: SURROGATE, value: BODY, index_text: true, @@ -541,7 +539,7 @@ mod tests { database_id: DatabaseId::DEFAULT.as_u64(), tid: TID, collection: COLL, - document_id: &row_key, + storage_key: crate::engine::document::store::StorageKey::for_surrogate(SURROGATE), surrogate: SURROGATE, value: BODY, index_text: false, diff --git a/nodedb/src/data/executor/handlers/point/apply_put/index.rs b/nodedb/src/data/executor/handlers/point/apply_put/index.rs index ac5d0d39d..0f4804ad1 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/index.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/index.rs @@ -9,6 +9,7 @@ use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::doc_format; use crate::data::executor::spatial_key::SpatialIndexKey; +use crate::engine::document::store::StorageKey; impl CoreLoop { /// Spatial R-tree + columnar ingest side-effect: parse geometry fields, @@ -25,7 +26,7 @@ impl CoreLoop { database_id: u64, tid: u64, collection: &str, - document_id: &str, + storage_key: StorageKey, value: &[u8], ) -> Vec<( ( @@ -36,6 +37,9 @@ impl CoreLoop { ), u64, )> { + // Rendered once here; every reverse-map / hash use below shares it. + let document_id = storage_key.to_string(); + let document_id = document_id.as_str(); let mut inserts = Vec::new(); // Re-indexing a document must REPLACE, not append: `RTree::insert` // blindly pushes a fresh entry even when one with this `entry_id` diff --git a/nodedb/src/data/executor/handlers/point/apply_put/sparse.rs b/nodedb/src/data/executor/handlers/point/apply_put/sparse.rs index 06ed8b4eb..d20576690 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/sparse.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/sparse.rs @@ -10,6 +10,7 @@ //! the dense-vector path needs. use crate::data::executor::core_loop::CoreLoop; +use crate::engine::document::store::StorageKey; impl CoreLoop { /// Strict-schema `SparseVector` column names declared on `collection`, or @@ -77,15 +78,15 @@ impl CoreLoop { /// Sparse inverted-index side-effect: for every declared `SparseVector` /// column, extract its string literal from the document body, parse it, and /// upsert it into the corresponding `SparseInvertedIndex` keyed by - /// `document_id`. + /// `storage_key`'s rendered text. /// - /// `document_id` is the hex-surrogate storage `row_key` — the SAME id the - /// delete path (`remove_document_sparse_indexes`) and the engine handler + /// `storage_key` is the SAME id the delete path + /// (`remove_document_sparse_indexes`) and the engine handler /// (`execute_sparse_insert` / `execute_sparse_search`) key on, so a search /// reads back exactly what this write wrote. The index `insert` is an /// upsert (it removes the doc's prior entries first), so a second put for - /// the same `document_id` replaces rather than duplicates. A missing field, - /// a non-string value, or an unparseable literal is skipped — mirroring the + /// the same key replaces rather than duplicates. A missing field, a + /// non-string value, or an unparseable literal is skipped — mirroring the /// dense-vector path's silent skip of malformed fields. /// /// No-op (byte-identical to a collection without sparse columns) when @@ -95,7 +96,7 @@ impl CoreLoop { database_id: u64, tid: u64, collection: &str, - document_id: &str, + storage_key: StorageKey, value: &[u8], ) { let sparse_fields = self.strict_sparse_fields(database_id, tid, collection); @@ -109,6 +110,8 @@ impl CoreLoop { return; }; + // Rendered once here — the inverted index is keyed by text. + let document_id = storage_key.to_string(); for field in &sparse_fields { let Some(nodedb_types::Value::String(literal)) = obj.get(field) else { continue; @@ -117,7 +120,7 @@ impl CoreLoop { continue; }; self.get_or_create_sparse_index(database_id, tid, collection, field) - .insert(document_id, &sv); + .insert(&document_id, &sv); // Sparse indexes are in-memory with no redb store behind them; the // checkpoint that persists them fires only on a dirty mark, exactly // as the standalone `execute_sparse_insert` handler flags it. @@ -125,24 +128,28 @@ impl CoreLoop { } } - /// Drop every sparse-index posting entry a document produced, keyed by its - /// hex-surrogate storage `row_key`. Shared by the PointDelete cascade - /// (which orphans a removed row's sparse entries) and the PointUpdate - /// re-index (which clears the old literal before inserting the new one). - /// Mirrors `remove_document_vector_indexes`. No-op when the collection - /// declares no sparse columns. + /// Drop every sparse-index posting entry a document produced. Shared by + /// the PointDelete cascade (which orphans a removed row's sparse entries) + /// and the PointUpdate re-index (which clears the old literal before + /// inserting the new one). Mirrors `remove_document_vector_indexes`. + /// No-op when the collection declares no sparse columns. pub(in crate::data::executor) fn remove_document_sparse_indexes( &mut self, database_id: u64, tid: u64, collection: &str, - row_key: &str, + storage_key: StorageKey, ) { let sparse_fields = self.strict_sparse_fields(database_id, tid, collection); + if sparse_fields.is_empty() { + return; + } + // The sparse index keys postings by the rendered storage key. + let row_key = storage_key.to_string(); for field in &sparse_fields { if self .get_or_create_sparse_index(database_id, tid, collection, field) - .delete(row_key) + .delete(&row_key) { self.checkpoint_coordinator.mark_dirty("vector", 1); } @@ -154,7 +161,7 @@ impl CoreLoop { mod tests { use super::*; use crate::bridge::dispatch::{BridgeRequest, BridgeResponse}; - use crate::engine::document::store::{CollectionConfig, surrogate_to_doc_id}; + use crate::engine::document::store::CollectionConfig; use nodedb_bridge::buffer::{Consumer, Producer, RingBuffer}; use nodedb_physical::physical_plan::StorageMode; use nodedb_types::columnar::{ColumnDef, ColumnType, StrictSchema}; @@ -238,12 +245,17 @@ mod tests { let tid = 1u64; let collection = "docs"; let field = "terms"; - let row_key = surrogate_to_doc_id(Surrogate::new(1)); register_strict_sparse(core, tid, collection, field); let doc = doc_with_sparse(field, "{3:0.5, 7:1.5}"); - core.apply_point_put_sparse_indexes(db, tid, collection, &row_key, &doc); + core.apply_point_put_sparse_indexes( + db, + tid, + collection, + StorageKey::for_surrogate(Surrogate::new(1)), + &doc, + ); assert_eq!( doc_count(core, db, tid, collection, field), @@ -263,7 +275,6 @@ mod tests { let tid = 1u64; let collection = "docs"; let field = "terms"; - let row_key = surrogate_to_doc_id(Surrogate::new(1)); register_strict_sparse(core, tid, collection, field); @@ -271,14 +282,14 @@ mod tests { db, tid, collection, - &row_key, + StorageKey::for_surrogate(Surrogate::new(1)), &doc_with_sparse(field, "{3:0.5, 7:1.5}"), ); core.apply_point_put_sparse_indexes( db, tid, collection, - &row_key, + StorageKey::for_surrogate(Surrogate::new(1)), &doc_with_sparse(field, "{1:0.9}"), ); @@ -300,7 +311,6 @@ mod tests { let tid = 1u64; let collection = "docs"; let field = "terms"; - let row_key = surrogate_to_doc_id(Surrogate::new(1)); register_strict_sparse(core, tid, collection, field); @@ -308,12 +318,17 @@ mod tests { db, tid, collection, - &row_key, + StorageKey::for_surrogate(Surrogate::new(1)), &doc_with_sparse(field, "{3:0.5, 7:1.5}"), ); assert_eq!(doc_count(core, db, tid, collection, field), 1); - core.remove_document_sparse_indexes(db, tid, collection, &row_key); + core.remove_document_sparse_indexes( + db, + tid, + collection, + StorageKey::for_surrogate(Surrogate::new(1)), + ); assert_eq!( doc_count(core, db, tid, collection, field), 0, @@ -331,7 +346,6 @@ mod tests { let db = 0u64; let tid = 1u64; let collection = "plain"; - let row_key = surrogate_to_doc_id(Surrogate::new(1)); // No strict sparse schema registered. assert!(!core.collection_has_sparse( @@ -344,7 +358,7 @@ mod tests { db, tid, collection, - &row_key, + StorageKey::for_surrogate(Surrogate::new(1)), &doc_with_sparse("terms", "{3:0.5}"), ); diff --git a/nodedb/src/data/executor/handlers/point/apply_put/types.rs b/nodedb/src/data/executor/handlers/point/apply_put/types.rs index 9a8aaa214..51e046b11 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/types.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/types.rs @@ -7,6 +7,7 @@ use nodedb_types::Surrogate; use crate::bridge::envelope::ErrorCode; use crate::data::executor::spatial_key::SpatialIndexKey; +use crate::engine::document::store::StorageKey; use nodedb_physical::physical_plan::ResolvedSumTarget; /// Parameters for [`CoreLoop::apply_point_put`](crate::data::executor::core_loop::CoreLoop::apply_point_put). @@ -14,7 +15,7 @@ pub(in crate::data::executor) struct PointPutParams<'a> { pub database_id: u64, pub tid: u64, pub collection: &'a str, - pub document_id: &'a str, + pub storage_key: StorageKey, pub surrogate: Surrogate, pub value: &'a [u8], /// Whether to index the document's text into the inverted BM25 index. diff --git a/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs b/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs index f08393d53..9970821c6 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs @@ -38,11 +38,15 @@ impl CoreLoop { database_id, tid, collection, - document_id, + storage_key, surrogate, value, wal_lsn, } = params; + // Rendered once here — `vector_doc_map` and its undo entries are + // keyed by text. + let document_id = storage_key.to_string(); + let document_id = document_id.as_str(); let mut inserts: Vec = Vec::new(); // Vector index: if the strict schema declares Vector(dim) columns, @@ -402,7 +406,7 @@ mod tests { let tid = 1u64; let collection = "docs"; let surrogate = Surrogate::new(1); - let row_key = crate::engine::document::store::surrogate_to_doc_id(surrogate); + let storage_key = crate::engine::document::store::StorageKey::for_surrogate(surrogate); register_bare_field(core, db_id, tid, collection); @@ -411,7 +415,7 @@ mod tests { database_id: db_id, tid, collection, - document_id: &row_key, + storage_key, surrogate, value: &first, wal_lsn: 0, @@ -423,7 +427,7 @@ mod tests { database_id: db_id, tid, collection, - document_id: &row_key, + storage_key, surrogate, value: &second, wal_lsn: 0, @@ -457,7 +461,7 @@ mod tests { let tid = 1u64; let collection = "docs"; let surrogate = Surrogate::new(1); - let row_key = crate::engine::document::store::surrogate_to_doc_id(surrogate); + let storage_key = crate::engine::document::store::StorageKey::for_surrogate(surrogate); register_named_field(core, db_id, tid, collection, "embedding"); register_named_field(core, db_id, tid, collection, "title_vec"); @@ -470,7 +474,7 @@ mod tests { database_id: db_id, tid, collection, - document_id: &row_key, + storage_key, surrogate, value: &doc, wal_lsn: 0, @@ -506,7 +510,7 @@ mod tests { let tid = 1u64; let collection = "docs"; let surrogate = Surrogate::new(1); - let row_key = crate::engine::document::store::surrogate_to_doc_id(surrogate); + let storage_key = crate::engine::document::store::StorageKey::for_surrogate(surrogate); register_bare_field(core, db_id, tid, collection); @@ -523,7 +527,7 @@ mod tests { database_id: db_id, tid, collection, - document_id: &row_key, + storage_key, surrogate, value: &body, wal_lsn: 0, @@ -549,7 +553,7 @@ mod tests { let tid = 1u64; let collection = "docs"; let surrogate = Surrogate::new(1); - let row_key = crate::engine::document::store::surrogate_to_doc_id(surrogate); + let storage_key = crate::engine::document::store::StorageKey::for_surrogate(surrogate); register_bare_field(core, db_id, tid, collection); @@ -564,7 +568,7 @@ mod tests { database_id: db_id, tid, collection, - document_id: &row_key, + storage_key, surrogate, value: &body, wal_lsn: 0, diff --git a/nodedb/src/data/executor/handlers/point/apply_put/vector/remove.rs b/nodedb/src/data/executor/handlers/point/apply_put/vector/remove.rs index 6ac39a69a..07352f78a 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/vector/remove.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/vector/remove.rs @@ -3,6 +3,7 @@ //! Soft-delete a document's prior vector nodes, per field or whole-document. use crate::data::executor::core_loop::CoreLoop; +use crate::engine::document::store::StorageKey; use super::types::VectorIndexDelta; @@ -64,7 +65,7 @@ impl CoreLoop { database_id: u64, tid: u64, collection: &str, - row_key: &str, + storage_key: StorageKey, ) -> Vec { let strict_fields = self.strict_vector_fields(database_id, tid, collection); let candidate_fields: Vec = if !strict_fields.is_empty() { @@ -73,13 +74,18 @@ impl CoreLoop { self.schemaless_vector_field_names(database_id, tid, collection) }; let mut vector_deletes = Vec::with_capacity(candidate_fields.len()); + if candidate_fields.is_empty() { + return vector_deletes; + } + // The vector reverse map keys nodes by the rendered storage key. + let row_key = storage_key.to_string(); for field in candidate_fields { if let Some(delta) = self.remove_document_vector_index_field( database_id, tid, collection, &field, - row_key, + &row_key, ) { vector_deletes.push(delta); } diff --git a/nodedb/src/data/executor/handlers/point/apply_put/vector/types.rs b/nodedb/src/data/executor/handlers/point/apply_put/vector/types.rs index 1e98c630c..8731fc7a8 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/vector/types.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/vector/types.rs @@ -11,7 +11,7 @@ pub(in crate::data::executor) struct VectorIndexPutParams<'a> { pub database_id: u64, pub tid: u64, pub collection: &'a str, - pub document_id: &'a str, + pub storage_key: crate::engine::document::store::StorageKey, pub surrogate: nodedb_types::Surrogate, pub value: &'a [u8], pub wal_lsn: u64, diff --git a/nodedb/src/data/executor/handlers/point/insert.rs b/nodedb/src/data/executor/handlers/point/insert.rs index f6c469e9a..b62745f77 100644 --- a/nodedb/src/data/executor/handlers/point/insert.rs +++ b/nodedb/src/data/executor/handlers/point/insert.rs @@ -66,7 +66,6 @@ impl CoreLoop { deferred_sum_targets, } = p; let storage_key = StorageKey::for_surrogate(surrogate); - let row_key = storage_key.to_string(); let document_identity = RowIdentity::from_user_key(document_id); debug!( core = self.core_id, @@ -172,7 +171,7 @@ impl CoreLoop { database_id: task.request.database_id.as_u64(), tid, collection, - document_id: &row_key, + storage_key, surrogate, value: effective_value, index_text: true, diff --git a/nodedb/src/data/executor/handlers/point/put.rs b/nodedb/src/data/executor/handlers/point/put.rs index 221c2d375..5eb6859a8 100644 --- a/nodedb/src/data/executor/handlers/point/put.rs +++ b/nodedb/src/data/executor/handlers/point/put.rs @@ -49,8 +49,6 @@ impl CoreLoop { resolved_sum_targets, } = params; let storage_key = StorageKey::for_surrogate(surrogate); - let row_key = storage_key.to_string(); - let row_key = row_key.as_str(); let document_identity = RowIdentity::from_user_key(document_id); debug!(core = self.core_id, %collection, %document_id, "point put"); @@ -102,7 +100,7 @@ impl CoreLoop { database_id, tid, collection, - document_id: row_key, + storage_key, surrogate, value: effective_value, index_text: true, diff --git a/nodedb/src/data/executor/handlers/point/update/exec.rs b/nodedb/src/data/executor/handlers/point/update/exec.rs index 3e7930045..dc82ad854 100644 --- a/nodedb/src/data/executor/handlers/point/update/exec.rs +++ b/nodedb/src/data/executor/handlers/point/update/exec.rs @@ -255,8 +255,7 @@ impl CoreLoop { database_id, tid, collection, - row_key, - surrogate, + storage_key, new_body: &updated_bytes, is_strict, has_vectors, diff --git a/nodedb/src/data/executor/handlers/point/update_reindex_secondary.rs b/nodedb/src/data/executor/handlers/point/update_reindex_secondary.rs index f69dc5a79..f1d77ce29 100644 --- a/nodedb/src/data/executor/handlers/point/update_reindex_secondary.rs +++ b/nodedb/src/data/executor/handlers/point/update_reindex_secondary.rs @@ -8,17 +8,16 @@ //! inverted index. Both are maintained here, together, so a caller cannot //! remember one and forget the other. -use nodedb_types::Surrogate; - use crate::data::executor::core_loop::CoreLoop; +use crate::engine::document::store::StorageKey; /// Inputs to [`CoreLoop::update_reindex_vector_and_sparse`]. pub(in crate::data::executor) struct UpdateSecondaryReindex<'a> { pub database_id: u64, pub tid: u64, pub collection: &'a str, - pub row_key: &'a str, - pub surrogate: Surrogate, + /// The row's storage key, shared by the vector and sparse reindex paths. + pub storage_key: StorageKey, pub new_body: &'a [u8], pub is_strict: bool, /// Precomputed by the caller so a loop over N rows pays the @@ -27,7 +26,7 @@ pub(in crate::data::executor) struct UpdateSecondaryReindex<'a> { } impl CoreLoop { - /// Re-index `row_key`'s vectors and sparse literal from its new body. + /// Re-index the row's vectors and sparse literal from its new body. /// /// Each half is a no-op when the collection declares nothing of that kind. /// A vector whose width disagrees with the index is an error, so an @@ -41,8 +40,7 @@ impl CoreLoop { database_id: p.database_id, tid: p.tid, collection: p.collection, - row_key: p.row_key, - surrogate: p.surrogate, + storage_key: p.storage_key, new_body: p.new_body, is_strict: p.is_strict, has_vectors: p.has_vectors, @@ -53,7 +51,7 @@ impl CoreLoop { database_id: p.database_id, tid: p.tid, collection: p.collection, - row_key: p.row_key, + storage_key: p.storage_key, new_body: p.new_body, is_strict: p.is_strict, has_sparse, diff --git a/nodedb/src/data/executor/handlers/point/update_reindex_sparse.rs b/nodedb/src/data/executor/handlers/point/update_reindex_sparse.rs index 8bcddaeeb..885562ea3 100644 --- a/nodedb/src/data/executor/handlers/point/update_reindex_sparse.rs +++ b/nodedb/src/data/executor/handlers/point/update_reindex_sparse.rs @@ -17,15 +17,16 @@ //! with insert and delete. use crate::data::executor::core_loop::CoreLoop; +use crate::engine::document::store::StorageKey; /// Inputs for [`CoreLoop::update_reindex_sparse_indexes`]. pub(in crate::data::executor) struct UpdateSparseReindex<'a> { pub database_id: u64, pub tid: u64, pub collection: &'a str, - /// Hex-surrogate storage key (matches the sparse-index doc-id keying used - /// by the put and delete paths). - pub row_key: &'a str, + /// Storage key (matches the sparse-index doc-id keying used by the put + /// and delete paths). + pub storage_key: StorageKey, /// The freshly-written stored body (Binary Tuple for strict, MessagePack /// for schemaless). Decoded storage-mode-aware to extract the new literal. pub new_body: &'a [u8], @@ -64,7 +65,7 @@ impl CoreLoop { // clears the `SparseVector` field must not leave the stale literal // searchable, and the re-insert below only re-adds fields present in // the new body. - self.remove_document_sparse_indexes(p.database_id, p.tid, p.collection, p.row_key); + self.remove_document_sparse_indexes(p.database_id, p.tid, p.collection, p.storage_key); // Re-extract from the new body via the exact put-time path. Sparse // extraction reads MessagePack; strict bodies are stored as Binary @@ -93,7 +94,7 @@ impl CoreLoop { p.new_body }; - self.apply_point_put_sparse_indexes(p.database_id, p.tid, p.collection, p.row_key, mp); + self.apply_point_put_sparse_indexes(p.database_id, p.tid, p.collection, p.storage_key, mp); Ok(()) } } diff --git a/nodedb/src/data/executor/handlers/point/update_reindex_vector.rs b/nodedb/src/data/executor/handlers/point/update_reindex_vector.rs index 74507c080..cb6b2ca04 100644 --- a/nodedb/src/data/executor/handlers/point/update_reindex_vector.rs +++ b/nodedb/src/data/executor/handlers/point/update_reindex_vector.rs @@ -19,17 +19,16 @@ use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::point::apply_put::VectorIndexPutParams; -use nodedb_types::Surrogate; +use crate::engine::document::store::StorageKey; /// Inputs for [`CoreLoop::update_reindex_vector_indexes`]. pub(in crate::data::executor) struct UpdateVectorReindex<'a> { pub database_id: u64, pub tid: u64, pub collection: &'a str, - /// Hex-surrogate storage key (matches the `vector_doc_map` keying used by - /// the put and delete paths). - pub row_key: &'a str, - pub surrogate: Surrogate, + /// The row's storage key, matching the `vector_doc_map` keying used by the + /// put and delete paths. + pub storage_key: StorageKey, /// The freshly-written stored body (Binary Tuple for strict, MessagePack /// for schemaless). Decoded storage-mode-aware to extract the new vectors. pub new_body: &'a [u8], @@ -69,7 +68,7 @@ impl CoreLoop { // entries) before re-inserting: `insert_with_surrogate` appends a new // node rather than replacing, so skipping this would leave the stale // embedding searchable alongside the new one. - self.remove_document_vector_indexes(p.database_id, p.tid, p.collection, p.row_key); + self.remove_document_vector_indexes(p.database_id, p.tid, p.collection, p.storage_key); // Re-extract vectors from the new body via the exact put-time path. // Vector extraction reads MessagePack; strict bodies are stored as @@ -107,8 +106,8 @@ impl CoreLoop { database_id: p.database_id, tid: p.tid, collection: p.collection, - document_id: p.row_key, - surrogate: p.surrogate, + storage_key: p.storage_key, + surrogate: p.storage_key.surrogate(), value: mp, wal_lsn: 0, })?; diff --git a/nodedb/src/data/executor/handlers/spatial_sync.rs b/nodedb/src/data/executor/handlers/spatial_sync.rs index dd8cd7bb0..72a4a56e6 100644 --- a/nodedb/src/data/executor/handlers/spatial_sync.rs +++ b/nodedb/src/data/executor/handlers/spatial_sync.rs @@ -13,8 +13,8 @@ //! This matches the direct `INSERT INTO ... VALUES (ST_GeomFromText(...))` //! path so cross-engine prefilter (roaring-bitmap intersect against the //! surrogate space) just works — `spatial_doc_map` stores the same -//! 8-char hex string that `surrogate_to_doc_id(surrogate)` produces, which -//! is what the scan path parses via `u32::from_str_radix(doc_id, 16)`. +//! 8-char hex string that `StorageKey::for_surrogate(surrogate)` produces, +//! which is what the scan path parses via `u32::from_str_radix(doc_id, 16)`. //! //! ## Document-store parity //! @@ -31,7 +31,7 @@ use crate::bridge::envelope::Response; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::sync_gate::{SyncAdmit, ack_status_from_admit}; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::StorageKey; use crate::engine::spatial::{RTree, RTreeEntry}; use crate::types::TenantId; use crate::util::fnv1a_hash; @@ -80,7 +80,7 @@ impl CoreLoop { geometry, provenance, } = args; - let doc_id = surrogate_to_doc_id(surrogate); + let doc_id = StorageKey::for_surrogate(surrogate).to_string(); debug!( core = self.core_id, @@ -222,7 +222,7 @@ impl CoreLoop { surrogate: Surrogate, provenance: Option<&SyncProvenance>, ) -> Response { - let doc_id = surrogate_to_doc_id(surrogate); + let doc_id = StorageKey::for_surrogate(surrogate).to_string(); debug!( core = self.core_id, diff --git a/nodedb/src/data/executor/handlers/text_search.rs b/nodedb/src/data/executor/handlers/text_search.rs index 74f93dc8b..af1defba8 100644 --- a/nodedb/src/data/executor/handlers/text_search.rs +++ b/nodedb/src/data/executor/handlers/text_search.rs @@ -215,8 +215,8 @@ impl CoreLoop { if rows.len() >= top_k { break; } - let hex_key = crate::engine::document::store::surrogate_to_doc_id(surrogate); let storage_key = nodedb_types::StorageKey::for_surrogate(surrogate); + let hex_key = storage_key.to_string(); let bytes_opt = match self.overlay_or_base_body(txn_id, &coll_key, &storage_key, || { self.sparse.get(database_id, tid, collection, &storage_key) }) { diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/spatial_merge.rs b/nodedb/src/data/executor/handlers/transaction/overlay/spatial_merge.rs index eecb35d96..34c32a810 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/spatial_merge.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/spatial_merge.rs @@ -42,7 +42,7 @@ use crate::data::executor::handlers::spatial_refine::{ apply_predicate, extract_geometry, project_doc, }; use crate::data::executor::handlers::transaction::overlay::Staged; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::StorageKey; use crate::types::{DatabaseId, TenantId, TxnId}; /// Inputs for [`CoreLoop::merge_overlay_into_spatial_scan`]. @@ -187,7 +187,7 @@ impl CoreLoop { return true; } } - let doc_id = surrogate_to_doc_id(Surrogate(raw)); + let doc_id = StorageKey::for_surrogate(Surrogate(raw)).to_string(); *row = project_doc(&doc, &doc_id, projection); true } @@ -215,7 +215,7 @@ impl CoreLoop { if !row_matches(&doc)? { continue; } - let doc_id = surrogate_to_doc_id(Surrogate(surrogate)); + let doc_id = StorageKey::for_surrogate(Surrogate(surrogate)).to_string(); results.push(project_doc(&doc, &doc_id, projection)); seen.insert(surrogate); } diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_spatial.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_spatial.rs index d087cafd6..2d1c6fdb4 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_spatial.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_spatial.rs @@ -33,7 +33,7 @@ use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::spatial_sync::geometry_to_value; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::StorageKey; use crate::types::TxnId; /// Inputs for [`CoreLoop::stage_spatial_insert`]. @@ -66,7 +66,7 @@ impl CoreLoop { geometry, } = params; - let doc_id = surrogate_to_doc_id(surrogate); + let doc_id = StorageKey::for_surrogate(surrogate).to_string(); let mut doc_map = std::collections::HashMap::new(); doc_map.insert(field.to_string(), geometry_to_value(geometry)); diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs index 30ec6213e..e20d9db90 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs @@ -68,8 +68,6 @@ impl CoreLoop { resolved_sum_targets, deferred_sum_targets, } = p; - let row_key = crate::engine::document::store::surrogate_to_doc_id(surrogate); - let row_key = row_key.as_str(); let storage_key = crate::engine::document::store::StorageKey::for_surrogate(surrogate); let database_id = dummy_task.request.database_id.as_u64(); @@ -182,7 +180,7 @@ impl CoreLoop { database_id, tid, collection, - document_id: row_key, + storage_key, surrogate, value: effective_value, index_text: true, diff --git a/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs b/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs index ec6d37cea..f1abba4f0 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs @@ -82,10 +82,6 @@ mod tests { zerompk::to_msgpack_vec(&Value::Object(obj)).unwrap() } - fn row_key() -> String { - crate::engine::document::store::surrogate_to_doc_id(Surrogate::new(1)) - } - /// Autocommit PUT via `apply_point_put` inside a self-owned redb txn (mirrors /// `execute_point_put`). fn autocommit_put(core: &mut CoreLoop) { @@ -98,7 +94,9 @@ mod tests { database_id: DB, tid: TID, collection: COLL, - document_id: &row_key(), + storage_key: crate::engine::document::store::StorageKey::for_surrogate( + Surrogate::new(1), + ), surrogate: Surrogate::new(1), value: &value, index_text: true, diff --git a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs index 7e2203724..c5866efbd 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs @@ -411,7 +411,9 @@ mod tests { database_id: DB, tid: TID, collection: COLL, - document_id: &row_key(), + storage_key: crate::engine::document::store::StorageKey::for_surrogate( + Surrogate::new(1), + ), surrogate: Surrogate::new(1), value: &value, index_text: true, diff --git a/nodedb/src/data/executor/handlers/truncate.rs b/nodedb/src/data/executor/handlers/truncate.rs index 065ff4d93..5b90269d3 100644 --- a/nodedb/src/data/executor/handlers/truncate.rs +++ b/nodedb/src/data/executor/handlers/truncate.rs @@ -178,7 +178,7 @@ impl CoreLoop { // the leaked vector keeps scoring in KNN search in the same // process (mirrors `execute_bulk_delete`'s vector cascade). if has_vectors { - self.remove_document_vector_indexes(database_id, tid, collection, &doc_id); + self.remove_document_vector_indexes(database_id, tid, collection, *storage_key); write_set.push(WriteSetEntry { surrogate: surrogate.as_u32(), is_delete: true, diff --git a/nodedb/src/data/executor/handlers/update_from_join_write.rs b/nodedb/src/data/executor/handlers/update_from_join_write.rs index 2dc76d8f3..f5ee441a4 100644 --- a/nodedb/src/data/executor/handlers/update_from_join_write.rs +++ b/nodedb/src/data/executor/handlers/update_from_join_write.rs @@ -225,8 +225,7 @@ impl CoreLoop { database_id, tid, collection: target_collection, - row_key: &doc_id, - surrogate, + storage_key, new_body: &updated_bytes, is_strict, has_vectors, diff --git a/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs b/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs index 16ff2f867..3728b029a 100644 --- a/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs +++ b/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs @@ -12,7 +12,6 @@ use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::enforcement::write_hook::HookCtx; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::surrogate_to_doc_id; use nodedb_physical::physical_plan::ResolvedSumTarget; use nodedb_types::Surrogate; @@ -68,8 +67,6 @@ impl CoreLoop { rls_filters, resolved_sum_targets, } = params; - let row_key = surrogate_to_doc_id(surrogate); - let row_key = row_key.as_str(); debug!( core = self.core_id, %collection, @@ -130,7 +127,6 @@ impl CoreLoop { collection, document_id, surrogate, - row_key, value, on_conflict_updates, rls_write_check, @@ -150,7 +146,6 @@ impl CoreLoop { collection, document_id, surrogate, - row_key, value, rls_write_check, returning, diff --git a/nodedb/src/data/executor/handlers/upsert/exec/insert.rs b/nodedb/src/data/executor/handlers/upsert/exec/insert.rs index e5974c0d9..fe24da059 100644 --- a/nodedb/src/data/executor/handlers/upsert/exec/insert.rs +++ b/nodedb/src/data/executor/handlers/upsert/exec/insert.rs @@ -21,7 +21,6 @@ pub(super) struct InsertCtx<'a> { pub collection: &'a str, pub document_id: &'a str, pub surrogate: Surrogate, - pub row_key: &'a str, pub value: &'a [u8], pub rls_write_check: &'a nodedb_types::RlsWriteCheck, pub returning: Option<&'a nodedb_physical::physical_plan::ReturningSpec>, @@ -46,7 +45,6 @@ impl CoreLoop { collection, document_id, surrogate, - row_key, value, rls_write_check, returning, @@ -106,7 +104,7 @@ impl CoreLoop { database_id, tid, collection, - document_id: row_key, + storage_key, surrogate, value: effective_value, index_text: true, diff --git a/nodedb/src/data/executor/handlers/upsert/exec/overwrite.rs b/nodedb/src/data/executor/handlers/upsert/exec/overwrite.rs index b4633d3dd..0fd220e71 100644 --- a/nodedb/src/data/executor/handlers/upsert/exec/overwrite.rs +++ b/nodedb/src/data/executor/handlers/upsert/exec/overwrite.rs @@ -22,7 +22,6 @@ pub(super) struct OverwriteCtx<'a> { pub collection: &'a str, pub document_id: &'a str, pub surrogate: Surrogate, - pub row_key: &'a str, pub value: &'a [u8], pub on_conflict_updates: &'a [(String, nodedb_physical::physical_plan::UpdateValue)], pub rls_write_check: &'a nodedb_types::RlsWriteCheck, @@ -51,7 +50,6 @@ impl CoreLoop { collection, document_id, surrogate, - row_key, value, on_conflict_updates, rls_write_check, @@ -172,7 +170,7 @@ impl CoreLoop { // the write below puts the new one in — otherwise KNN keeps // scoring both. No-op when `has_vectors` is false. if has_vectors { - self.remove_document_vector_indexes(database_id, tid, collection, row_key); + self.remove_document_vector_indexes(database_id, tid, collection, storage_key); } // One transaction for the body, every index that describes it, @@ -199,7 +197,7 @@ impl CoreLoop { database_id, tid, collection, - document_id: row_key, + storage_key, surrogate, value: &merged_body, index_text: true, diff --git a/nodedb/src/data/executor/handlers/write_batch.rs b/nodedb/src/data/executor/handlers/write_batch.rs index 0f95638cb..e38588be0 100644 --- a/nodedb/src/data/executor/handlers/write_batch.rs +++ b/nodedb/src/data/executor/handlers/write_batch.rs @@ -103,7 +103,7 @@ impl CoreLoop { }; let tid = task.request.tenant_id.as_u64(); let db_id = task.request.database_id.as_u64(); - let row_key = crate::engine::document::store::surrogate_to_doc_id(*surrogate); + let storage_key = crate::engine::document::store::StorageKey::for_surrogate(*surrogate); results.push( self.apply_point_put( &txn, @@ -111,7 +111,7 @@ impl CoreLoop { database_id: db_id, tid, collection: collection.as_str(), - document_id: &row_key, + storage_key, surrogate: *surrogate, value, index_text: true, diff --git a/nodedb/src/data/executor/wal_replay_document_vector.rs b/nodedb/src/data/executor/wal_replay_document_vector.rs index 8495c9edb..46e82f16f 100644 --- a/nodedb/src/data/executor/wal_replay_document_vector.rs +++ b/nodedb/src/data/executor/wal_replay_document_vector.rs @@ -36,7 +36,7 @@ use nodedb_wal::record::RecordType; use super::core_loop::CoreLoop; use crate::data::executor::core_loop::write_index::KeyRepr; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::StorageKey; impl CoreLoop { /// Replay document `Put` records to rebuild secondary vector indexes, @@ -101,8 +101,13 @@ impl CoreLoop { continue; } let database_id = record.header.database_id; - let row_key = surrogate_to_doc_id(Surrogate::new(surrogate_u32)); - self.remove_document_vector_indexes(database_id, tenant_id, &collection, &row_key); + let storage_key = StorageKey::for_surrogate(Surrogate::new(surrogate_u32)); + self.remove_document_vector_indexes( + database_id, + tenant_id, + &collection, + storage_key, + ); let record_lsn = record.header.lsn; self.note_replay_write_lsn( database_id, @@ -145,9 +150,10 @@ impl CoreLoop { let database_id = record.header.database_id; // Live inserts key the vector reverse-map on the hex surrogate row - // key (`surrogate_to_doc_id`), not the user PK; reproduce that here - // so a later delete can still find and soft-delete the node. - let row_key = surrogate_to_doc_id(surrogate); + // key (`StorageKey::for_surrogate`), not the user PK; reproduce + // that here so a later delete can still find and soft-delete the + // node. + let storage_key = StorageKey::for_surrogate(surrogate); // The forward path rejects a width mismatch before the write is // acknowledged, so one can only appear here for a record journalled // before that check existed. It is already durable — refusing to @@ -158,7 +164,7 @@ impl CoreLoop { database_id, tid: tenant_id, collection: &collection, - document_id: &row_key, + storage_key, surrogate, value: &value, wal_lsn: record_lsn, diff --git a/nodedb/src/data/executor/wal_replay_redo_document.rs b/nodedb/src/data/executor/wal_replay_redo_document.rs index 841db3660..fde569635 100644 --- a/nodedb/src/data/executor/wal_replay_redo_document.rs +++ b/nodedb/src/data/executor/wal_replay_redo_document.rs @@ -24,8 +24,8 @@ //! * DELETE — `(collection, document_id, Option, surrogate)`. //! The autocommit delete shape `(collection, document_id, prov)` omits the //! surrogate; replay needs it (the redb storage key is -//! `surrogate_to_doc_id(surrogate)`, and the delete cascade keys on it), so -//! the redo shape appends it as a fourth element. +//! `StorageKey::for_surrogate(surrogate)`, and the delete cascade keys on +//! it), so the redo shape appends it as a fourth element. //! //! ## Idempotency //! @@ -76,7 +76,7 @@ use super::handlers::point::apply_delete::PointDeleteParams; use super::handlers::point::apply_put::PointPutParams; use super::handlers::transaction::overlay::BitemporalStamp; use crate::data::executor::core_loop::write_index::KeyRepr; -use crate::engine::document::store::surrogate_to_doc_id; +use crate::engine::document::store::StorageKey; impl CoreLoop { /// Replay reconstituted document `Put` / `Delete` redo sub-records. @@ -239,7 +239,7 @@ impl CoreLoop { record_lsn: u64, ) -> bool { let surrogate = Surrogate::new(surrogate_u32); - let row_key = surrogate_to_doc_id(surrogate); + let storage_key = StorageKey::for_surrogate(surrogate); let txn = match self.sparse.begin_write() { Ok(t) => t, Err(e) => { @@ -258,7 +258,7 @@ impl CoreLoop { database_id, tid: tenant_id, collection, - document_id: row_key.as_str(), + storage_key, surrogate, value, index_text: true, @@ -308,7 +308,8 @@ impl CoreLoop { surrogate_u32: u32, ) -> bool { let surrogate = Surrogate::new(surrogate_u32); - let row_key = surrogate_to_doc_id(surrogate); + let storage_key = StorageKey::for_surrogate(surrogate); + let row_key = storage_key.to_string(); let txn = match self.sparse.begin_write() { Ok(t) => t, Err(e) => { From 588964abcac35a0ba9ec462531a7e843bc98cc54 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 12 Sep 2026 15:28:12 +0800 Subject: [PATCH 14/17] refactor(identity): key CDC events and RRF fusion by RowIdentity Change events, hybrid-search fusion, and WAL replay dispatch carried row identity as raw String/&str throughout, so a batch write's "*" sentinel, a headless vector hit, and a real primary key were all the same type and could be mixed up at call sites. - Generalize nodedb_query::fusion's RankedResult/FusedResult over a typed key `K: Clone + Eq + Hash + Ord` instead of hardcoding String, and factor the shared score/sort/truncate logic into finish_fusion. - Add HybridFusionKey (Bound(StorageKey) or Headless(u32)) as the one key space the vector and text legs of a hybrid search fuse on, so a headless vector hit can never be misread as a bound surrogate. - Route change events, extract_write_metadata, and WAL CRDT replay through nodedb_types::RowIdentity instead of String/&str, rendering to text only at wire and response boundaries. - Update GraphRAG, graph expansion, and text-search-hybrid/triple handlers and CDC/WS/pgwire tests to the typed key. --- nodedb-query/src/fusion.rs | 166 +++++++++-------- .../src/control/change_stream/stream/bus.rs | 18 +- .../src/control/change_stream/stream/types.rs | 6 +- .../dispatch_utils/change_events/extract.rs | 169 +++++++++--------- .../dispatch_utils/change_events/publish.rs | 11 +- nodedb/src/control/server/http/routes/cdc.rs | 2 +- .../control/server/http/routes/subscribe.rs | 6 +- .../server/http/routes/ws_rpc/format.rs | 4 +- .../server/shared/ddl/neutral/show_changes.rs | 11 +- .../src/control/server/shared/session/live.rs | 19 +- .../data/executor/handlers/graph_expansion.rs | 18 +- .../src/data/executor/handlers/graph_rag.rs | 60 +++---- .../executor/handlers/graph_rag_triple.rs | 14 +- .../src/data/executor/handlers/hybrid_key.rs | 78 ++++++++ .../data/executor/handlers/hybrid_overlay.rs | 65 ++++--- nodedb/src/data/executor/handlers/mod.rs | 1 + .../executor/handlers/text_search_hybrid.rs | 52 +++--- .../executor/handlers/text_search_triple.rs | 105 ++++++++--- .../data/executor/handlers/vector_search.rs | 67 +++---- .../src/data/executor/wal_replay/crdt_doc.rs | 39 ++-- .../src/data/executor/wal_replay/crdt_list.rs | 49 +++-- .../executor/wal_replay_document_vector.rs | 4 + .../data/executor/wal_replay_redo_document.rs | 8 +- .../test_tenant_isolation_cdc.rs | 19 +- .../test_tenant_isolation_cdc_negative.rs | 18 +- nodedb/tests/inproc/cases/http_cdc.rs | 23 +-- nodedb/tests/inproc/cases/http_ws.rs | 9 +- .../inproc/cases/http_ws_authorization.rs | 3 +- .../inproc/cases/pgwire_cdc_change_events.rs | 2 +- 29 files changed, 615 insertions(+), 431 deletions(-) create mode 100644 nodedb/src/data/executor/handlers/hybrid_key.rs diff --git a/nodedb-query/src/fusion.rs b/nodedb-query/src/fusion.rs index 6ef260630..02f29d8fa 100644 --- a/nodedb-query/src/fusion.rs +++ b/nodedb-query/src/fusion.rs @@ -1,5 +1,8 @@ // SPDX-License-Identifier: Apache-2.0 +use std::collections::HashMap; +use std::hash::Hash; + /// Reciprocal Rank Fusion (RRF) for combining ranked results from multiple engines. /// /// RRF is used when a query hits multiple engines (e.g., vector similarity + @@ -12,10 +15,13 @@ pub const DEFAULT_RRF_K: f64 = 60.0; /// A scored result from a single engine. +/// +/// `K` is the key the engine ranks under. It defaults to `String` for +/// callers whose key is opaque text. #[derive(Debug, Clone)] -pub struct RankedResult { +pub struct RankedResult { /// Document identifier (engine-specific). - pub document_id: String, + pub document_id: K, /// Rank within the engine's result list (0-based). pub rank: usize, /// Original score from the engine (for diagnostics). @@ -26,38 +32,23 @@ pub struct RankedResult { /// A fused result after RRF combination. #[derive(Debug, Clone)] -pub struct FusedResult { - pub document_id: String, +pub struct FusedResult { + pub document_id: K, pub rrf_score: f64, /// Per-engine contributions for explainability. pub contributions: Vec<(&'static str, f64)>, } -/// Fuse multiple ranked result lists using Reciprocal Rank Fusion. +/// Sum each key's contributions, sort by RRF score, and cut to `top_k`. /// -/// Each inner Vec is a ranked list from one engine (ordered by relevance). -/// Returns the top_k fused results sorted by RRF score (descending). -pub fn reciprocal_rank_fusion( - ranked_lists: &[Vec], - k: Option, +/// The tie-break on `document_id` is what makes the output deterministic: +/// RRF produces many equal scores and the score map iterates in arbitrary +/// order, so a unique secondary key gives a total order. +fn finish_fusion( + scores: HashMap>, top_k: usize, -) -> Vec { - let k = k.unwrap_or(DEFAULT_RRF_K); - - let mut scores: std::collections::HashMap> = - std::collections::HashMap::new(); - - for list in ranked_lists { - for result in list { - let contribution = 1.0 / (k + result.rank as f64 + 1.0); - scores - .entry(result.document_id.clone()) - .or_default() - .push((result.source, contribution)); - } - } - - let mut fused: Vec = scores +) -> Vec> { + let mut fused: Vec> = scores .into_iter() .map(|(doc_id, contributions)| { let rrf_score = contributions.iter().map(|(_, s)| s).sum(); @@ -73,16 +64,38 @@ pub fn reciprocal_rank_fusion( b.rrf_score .partial_cmp(&a.rrf_score) .unwrap_or(std::cmp::Ordering::Equal) - // Deterministic tie-break: RRF produces many equal scores, and the - // score map iterates in nondeterministic order, so without a stable - // secondary key the output ranking varies run-to-run. document_id - // is unique, giving a total deterministic order. .then_with(|| a.document_id.cmp(&b.document_id)) }); fused.truncate(top_k); fused } +/// Fuse multiple ranked result lists using Reciprocal Rank Fusion. +/// +/// Each inner Vec is a ranked list from one engine (ordered by relevance). +/// Returns the top_k fused results sorted by RRF score (descending). +pub fn reciprocal_rank_fusion( + ranked_lists: &[Vec>], + k: Option, + top_k: usize, +) -> Vec> { + let k = k.unwrap_or(DEFAULT_RRF_K); + + let mut scores: HashMap> = HashMap::new(); + + for list in ranked_lists { + for result in list { + let contribution = 1.0 / (k + result.rank as f64 + 1.0); + scores + .entry(result.document_id.clone()) + .or_default() + .push((result.source, contribution)); + } + } + + finish_fusion(scores, top_k) +} + /// Fuse ranked lists with per-list **linear weights**. /// /// Each list's reciprocal-rank contribution is scaled by its weight, so a @@ -97,12 +110,12 @@ pub fn reciprocal_rank_fusion( /// # Panics /// /// Panics if `weights.len() != ranked_lists.len()`. -pub fn reciprocal_rank_fusion_linear( - ranked_lists: &[Vec], +pub fn reciprocal_rank_fusion_linear( + ranked_lists: &[Vec>], k: Option, weights: &[f64], top_k: usize, -) -> Vec { +) -> Vec> { assert_eq!( ranked_lists.len(), weights.len(), @@ -110,8 +123,7 @@ pub fn reciprocal_rank_fusion_linear( ); let k = k.unwrap_or(DEFAULT_RRF_K); - let mut scores: std::collections::HashMap> = - std::collections::HashMap::new(); + let mut scores: HashMap> = HashMap::new(); for (list_idx, list) in ranked_lists.iter().enumerate() { let w = weights[list_idx]; @@ -124,27 +136,7 @@ pub fn reciprocal_rank_fusion_linear( } } - let mut fused: Vec = scores - .into_iter() - .map(|(doc_id, contributions)| { - let rrf_score = contributions.iter().map(|(_, s)| s).sum(); - FusedResult { - document_id: doc_id, - rrf_score, - contributions, - } - }) - .collect(); - - fused.sort_unstable_by(|a, b| { - b.rrf_score - .partial_cmp(&a.rrf_score) - .unwrap_or(std::cmp::Ordering::Equal) - // Deterministic tie-break by unique document_id (see note above). - .then_with(|| a.document_id.cmp(&b.document_id)) - }); - fused.truncate(top_k); - fused + finish_fusion(scores, top_k) } /// Fuse ranked lists with per-list k-constants for weighted influence. @@ -155,19 +147,18 @@ pub fn reciprocal_rank_fusion_linear( /// # Panics /// /// Panics if `k_per_list.len() != ranked_lists.len()`. -pub fn reciprocal_rank_fusion_weighted( - ranked_lists: &[Vec], +pub fn reciprocal_rank_fusion_weighted( + ranked_lists: &[Vec>], k_per_list: &[f64], top_k: usize, -) -> Vec { +) -> Vec> { assert_eq!( ranked_lists.len(), k_per_list.len(), "k_per_list length must match ranked_lists length" ); - let mut scores: std::collections::HashMap> = - std::collections::HashMap::new(); + let mut scores: HashMap> = HashMap::new(); for (list_idx, list) in ranked_lists.iter().enumerate() { let k = k_per_list[list_idx]; @@ -180,27 +171,7 @@ pub fn reciprocal_rank_fusion_weighted( } } - let mut fused: Vec = scores - .into_iter() - .map(|(doc_id, contributions)| { - let rrf_score = contributions.iter().map(|(_, s)| s).sum(); - FusedResult { - document_id: doc_id, - rrf_score, - contributions, - } - }) - .collect(); - - fused.sort_unstable_by(|a, b| { - b.rrf_score - .partial_cmp(&a.rrf_score) - .unwrap_or(std::cmp::Ordering::Equal) - // Deterministic tie-break by unique document_id (see note above). - .then_with(|| a.document_id.cmp(&b.document_id)) - }); - fused.truncate(top_k); - fused + finish_fusion(scores, top_k) } #[cfg(test)] @@ -238,6 +209,29 @@ mod tests { assert!(top2_ids.contains(&"d2")); } + /// A non-`String` key fuses the same way: the key type is the caller's. + #[test] + fn typed_key_fuses_and_tie_breaks_on_ord() { + let make = |ids: &[u32], source: &'static str| -> Vec> { + ids.iter() + .enumerate() + .map(|(rank, &id)| RankedResult { + document_id: id, + rank, + score: 0.0, + source, + }) + .collect() + }; + let a = make(&[7, 3], "a"); + let b = make(&[3, 9], "b"); + let fused = reciprocal_rank_fusion(&[a, b], None, 10); + assert_eq!(fused[0].document_id, 3); + // 7 and 9 each rank #1 in one list, so they tie; `Ord` breaks it. + assert_eq!(fused[1].document_id, 7); + assert_eq!(fused[2].document_id, 9); + } + #[test] fn weighted_rrf() { let list_a = make_ranked(&["a1", "a2"], "vector"); @@ -283,8 +277,8 @@ mod tests { #[test] fn empty() { - assert!(reciprocal_rank_fusion(&[], None, 10).is_empty()); - assert!(reciprocal_rank_fusion_linear(&[], None, &[], 10).is_empty()); - assert!(reciprocal_rank_fusion_weighted(&[], &[], 10).is_empty()); + assert!(reciprocal_rank_fusion::(&[], None, 10).is_empty()); + assert!(reciprocal_rank_fusion_linear::(&[], None, &[], 10).is_empty()); + assert!(reciprocal_rank_fusion_weighted::(&[], &[], 10).is_empty()); } } diff --git a/nodedb/src/control/change_stream/stream/bus.rs b/nodedb/src/control/change_stream/stream/bus.rs index 72e739c8a..1a2bbb620 100644 --- a/nodedb/src/control/change_stream/stream/bus.rs +++ b/nodedb/src/control/change_stream/stream/bus.rs @@ -4,6 +4,7 @@ use std::collections::VecDeque; use std::sync::Arc; use std::sync::atomic::{AtomicU64, Ordering}; +use nodedb_types::RowIdentity; use tracing::{debug, trace, warn}; use crate::types::{DatabaseId, Lsn, TenantId}; @@ -237,7 +238,8 @@ impl ChangeStream { lsn: Lsn::new(msg.lsn), tenant_id: TenantId::new(msg.tenant_id), collection: msg.collection.clone(), - document_id: msg.document_id.clone(), + // The wire carries the identity as text; wrap it verbatim. + document_id: RowIdentity::from_user_key(msg.document_id.clone()), operation, timestamp_ms: msg.timestamp_ms, after: None, @@ -288,7 +290,7 @@ pub fn broadcast_notify_to_cluster( tenant_id: event.tenant_id.as_u64(), database_id: database_id.as_u64(), collection: event.collection.clone(), - document_id: event.document_id.clone(), + document_id: event.document_id.to_string(), operation: event.operation.as_str().to_string(), timestamp_ms: event.timestamp_ms, lsn: event.lsn.as_u64(), @@ -340,7 +342,7 @@ mod tests { lsn: Lsn::new(lsn), tenant_id: TenantId::new(tenant), collection: "orders".into(), - document_id: document.into(), + document_id: RowIdentity::from_user_key(document), operation: ChangeOperation::Insert, timestamp_ms: 1, after: None, @@ -362,7 +364,7 @@ mod tests { 8, ) .unwrap_or_else(|_| panic!()); - assert_eq!(next.events[0].document_id, "second"); + assert_eq!(next.events[0].document_id.as_str(), "second"); } #[test] fn duplicate_lsn_events_paginate() { @@ -380,7 +382,7 @@ mod tests { 1, ) .unwrap_or_else(|_| panic!()); - assert_eq!(next.events[0].document_id, "b"); + assert_eq!(next.events[0].document_id.as_str(), "b"); } #[test] fn evicted_and_wrong_epoch_cursors_expire() { @@ -415,7 +417,7 @@ mod tests { let result = stream .query_changes(TenantId::new(1), None, ReplayStart::Timestamp(0), 1) .unwrap_or_else(|_| panic!()); - assert_eq!(result.events[0].document_id, "mine"); + assert_eq!(result.events[0].document_id.as_str(), "mine"); } #[tokio::test] @@ -436,9 +438,9 @@ mod tests { 1, ) .unwrap_or_else(|_| panic!()); - assert_eq!(result.events[0].document_id, "database-a"); + assert_eq!(result.events[0].document_id.as_str(), "database-a"); let received = subscription.recv_sequenced().await.unwrap(); - assert_eq!(received.document_id, "database-a"); + assert_eq!(received.document_id.as_str(), "database-a"); } #[test] diff --git a/nodedb/src/control/change_stream/stream/types.rs b/nodedb/src/control/change_stream/stream/types.rs index 8ec731dc5..4453f0c78 100644 --- a/nodedb/src/control/change_stream/stream/types.rs +++ b/nodedb/src/control/change_stream/stream/types.rs @@ -2,6 +2,8 @@ use std::ops::Deref; +use nodedb_types::RowIdentity; + use crate::types::{DatabaseId, Lsn, TenantId}; use super::ChangeCursor; @@ -12,7 +14,9 @@ pub struct ChangeEvent { pub lsn: Lsn, pub tenant_id: TenantId, pub collection: String, - pub document_id: String, + /// The identity a subscriber addresses the changed row by. A batch or + /// predicate write carries `"*"`: every row in the collection. + pub document_id: RowIdentity, pub operation: ChangeOperation, pub timestamp_ms: u64, pub after: Option, diff --git a/nodedb/src/control/server/dispatch_utils/change_events/extract.rs b/nodedb/src/control/server/dispatch_utils/change_events/extract.rs index 81272e7d7..b99493b8a 100644 --- a/nodedb/src/control/server/dispatch_utils/change_events/extract.rs +++ b/nodedb/src/control/server/dispatch_utils/change_events/extract.rs @@ -5,21 +5,37 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::change_stream::ChangeOperation; -use crate::engine::document::store::surrogate_to_doc_id; use crate::types::TenantId; use nodedb_physical::physical_plan::{ ArrayOp, ClusterArrayOp, ColumnarOp, CrdtOp, DocumentOp, DocumentResolvedMutation, KvOp, KvResolvedMutation, MetaOp, TimeseriesOp, VectorOp, }; +use nodedb_types::{RowIdentity, StorageKey}; + +/// One row change a plan yields: `(collection, row identity, op)`. +pub(super) type WriteChangeMeta = (String, RowIdentity, ChangeOperation); + +/// The identity a batch or predicate write reports: every row in the +/// collection, not one addressable row. A subscriber sees `"*"`. +fn every_row() -> RowIdentity { + RowIdentity::from_user_key("*") +} + +/// A KV row's identity is its key bytes, rendered as text for the subscriber. +fn kv_identity(key: &[u8]) -> RowIdentity { + RowIdentity::from_user_key(String::from_utf8_lossy(key)) +} /// Extract write metadata from a physical plan for change event publishing. /// -/// One `(collection, document_id, op)` tuple per row change; empty for reads/DDL. +/// One `(collection, identity, op)` tuple per row change; empty for reads/DDL. +/// The identity is the one a CDC subscriber addresses the row by: the +/// user-facing primary key, the KV key bytes, or the decimal surrogate. /// Exhaustive over [`PhysicalPlan`] — no catch-all — so a new variant is a compile error. pub(super) fn extract_write_metadata( plan: &PhysicalPlan, _tenant_id: TenantId, -) -> Vec<(String, String, ChangeOperation)> { +) -> Vec { match plan { PhysicalPlan::Document(DocumentOp::PointPut { collection, @@ -27,7 +43,7 @@ pub(super) fn extract_write_metadata( .. }) => vec![( collection.to_string(), - document_id.clone(), + RowIdentity::from_user_key(document_id.clone()), ChangeOperation::Insert, )], PhysicalPlan::Document(DocumentOp::PointDelete { @@ -36,7 +52,7 @@ pub(super) fn extract_write_metadata( .. }) => vec![( collection.to_string(), - document_id.clone(), + RowIdentity::from_user_key(document_id.clone()), ChangeOperation::Delete, )], PhysicalPlan::Document(DocumentOp::PointUpdate { @@ -45,7 +61,7 @@ pub(super) fn extract_write_metadata( .. }) => vec![( collection.to_string(), - document_id.clone(), + RowIdentity::from_user_key(document_id.clone()), ChangeOperation::Update, )], // `PointInsert` is plain SQL INSERT; distinct from `PointPut` (unconditional @@ -56,7 +72,7 @@ pub(super) fn extract_write_metadata( .. }) => vec![( collection.to_string(), - document_id.clone(), + RowIdentity::from_user_key(document_id.clone()), ChangeOperation::Insert, )], PhysicalPlan::Document(DocumentOp::Upsert { @@ -65,33 +81,33 @@ pub(super) fn extract_write_metadata( .. }) => vec![( collection.to_string(), - document_id.clone(), + RowIdentity::from_user_key(document_id.clone()), ChangeOperation::Insert, )], PhysicalPlan::Document(DocumentOp::BatchInsert { collection, .. }) => { - vec![(collection.to_string(), "*".into(), ChangeOperation::Insert)] + vec![(collection.to_string(), every_row(), ChangeOperation::Insert)] } PhysicalPlan::Document(DocumentOp::InsertSelect { target_collection, .. }) => vec![( target_collection.to_string(), - "*".into(), + every_row(), ChangeOperation::Insert, )], PhysicalPlan::Document(DocumentOp::BulkUpdate { collection, .. }) => { - vec![(collection.to_string(), "*".into(), ChangeOperation::Update)] + vec![(collection.to_string(), every_row(), ChangeOperation::Update)] } PhysicalPlan::Document(DocumentOp::BulkDelete { collection, .. }) => { - vec![(collection.to_string(), "*".into(), ChangeOperation::Delete)] + vec![(collection.to_string(), every_row(), ChangeOperation::Delete)] } PhysicalPlan::Document(DocumentOp::Truncate { collection, .. }) => { - vec![(collection.to_string(), "*".into(), ChangeOperation::Delete)] + vec![(collection.to_string(), every_row(), ChangeOperation::Delete)] } PhysicalPlan::Document(DocumentOp::UpdateFromJoin { target_collection, .. }) => vec![( target_collection.to_string(), - "*".into(), + every_row(), ChangeOperation::Update, )], // MERGE mixes INSERT/UPDATE/DELETE per arm; not individually addressable, so @@ -100,7 +116,7 @@ pub(super) fn extract_write_metadata( target_collection, .. }) => vec![( target_collection.to_string(), - "*".into(), + every_row(), ChangeOperation::Update, )], // Reports one event per mutation, naming every row touched — never collapses to "*". @@ -117,7 +133,7 @@ pub(super) fn extract_write_metadata( }; ( mutation.collection().to_string(), - mutation.document_id().to_string(), + RowIdentity::from_user_key(mutation.document_id().to_string()), operation, ) }) @@ -128,7 +144,7 @@ pub(super) fn extract_write_metadata( // Batch write; document_id="*" indicates a batch. High-cardinality metrics // would flood the bus otherwise — subscribe via collection_filter. PhysicalPlan::Timeseries(TimeseriesOp::Ingest { collection, .. }) => { - vec![(collection.to_string(), "*".into(), ChangeOperation::Insert)] + vec![(collection.to_string(), every_row(), ChangeOperation::Insert)] } // TimeseriesOp::Scan is a read — no row changed. PhysicalPlan::Timeseries(_) => Vec::new(), @@ -147,16 +163,16 @@ pub(super) fn extract_write_metadata( collection, key, .. }) => vec![( collection.to_string(), - String::from_utf8_lossy(key).into_owned(), + kv_identity(key), ChangeOperation::Insert, )], PhysicalPlan::Kv(KvOp::Delete { collection, .. }) // Predicate keys are decided in the Data Plane, so this reports one event with "*". | PhysicalPlan::Kv(KvOp::PredicateDelete { collection, .. }) => { - vec![(collection.to_string(), "*".into(), ChangeOperation::Delete)] + vec![(collection.to_string(), every_row(), ChangeOperation::Delete)] } PhysicalPlan::Kv(KvOp::PredicateUpdate { collection, .. }) => { - vec![(collection.to_string(), "*".into(), ChangeOperation::Update)] + vec![(collection.to_string(), every_row(), ChangeOperation::Update)] } PhysicalPlan::Kv(KvOp::FieldSet { collection, key, .. @@ -180,19 +196,19 @@ pub(super) fn extract_write_metadata( collection, key, .. }) => vec![( collection.to_string(), - String::from_utf8_lossy(key).into_owned(), + kv_identity(key), ChangeOperation::Update, )], PhysicalPlan::Kv(KvOp::BatchPut { collection, .. }) => { - vec![(collection.to_string(), "*".into(), ChangeOperation::Insert)] + vec![(collection.to_string(), every_row(), ChangeOperation::Insert)] } PhysicalPlan::Kv(KvOp::Truncate { collection }) => { - vec![(collection.to_string(), "*".into(), ChangeOperation::Delete)] + vec![(collection.to_string(), every_row(), ChangeOperation::Delete)] } // Debits + credits two keys in the same collection; not individually addressable, // so reported as one event with document_id="*" like other batch ops. PhysicalPlan::Kv(KvOp::Transfer { collection, .. }) => { - vec![(collection.to_string(), "*".into(), ChangeOperation::Update)] + vec![(collection.to_string(), every_row(), ChangeOperation::Update)] } // Spans two collections (delete source, insert dest) — the only write here // that can't be a single tuple, so it reports two. @@ -205,12 +221,12 @@ pub(super) fn extract_write_metadata( }) => vec![ ( source_collection.to_string(), - String::from_utf8_lossy(item_key).into_owned(), + kv_identity(item_key), ChangeOperation::Delete, ), ( dest_collection.to_string(), - String::from_utf8_lossy(dest_key).into_owned(), + kv_identity(dest_key), ChangeOperation::Insert, ), ], @@ -232,7 +248,7 @@ pub(super) fn extract_write_metadata( }; ( mutation.collection().to_string(), - String::from_utf8_lossy(mutation.key()).into_owned(), + kv_identity(mutation.key()), operation, ) }) @@ -243,20 +259,20 @@ pub(super) fn extract_write_metadata( // `spatial` rows are stored via the same `ColumnarOp` path as `columnar`. PhysicalPlan::Columnar(ColumnarOp::Insert { collection, .. }) => { - vec![(collection.to_string(), "*".into(), ChangeOperation::Insert)] + vec![(collection.to_string(), every_row(), ChangeOperation::Insert)] } PhysicalPlan::Columnar(ColumnarOp::Update { collection, .. }) => { - vec![(collection.to_string(), "*".into(), ChangeOperation::Update)] + vec![(collection.to_string(), every_row(), ChangeOperation::Update)] } PhysicalPlan::Columnar(ColumnarOp::Delete { collection, .. }) => { - vec![(collection.to_string(), "*".into(), ChangeOperation::Delete)] + vec![(collection.to_string(), every_row(), ChangeOperation::Delete)] } // Resolved-row-set form of the same UPDATE/DELETE — same CDC event as above. PhysicalPlan::Columnar(ColumnarOp::ResolvedUpdate { collection, .. }) => { - vec![(collection.to_string(), "*".into(), ChangeOperation::Update)] + vec![(collection.to_string(), every_row(), ChangeOperation::Update)] } PhysicalPlan::Columnar(ColumnarOp::ResolvedDelete { collection, .. }) => { - vec![(collection.to_string(), "*".into(), ChangeOperation::Delete)] + vec![(collection.to_string(), every_row(), ChangeOperation::Delete)] } // Scan / MaterializeScan are reads — no row changed. PhysicalPlan::Columnar(_) => Vec::new(), @@ -264,10 +280,10 @@ pub(super) fn extract_write_metadata( // Array cells are data-bearing rows, not an index — need CDC. // `array_id.name` is the user-visible collection name. PhysicalPlan::Array(ArrayOp::Put { array_id, .. }) => { - vec![(array_id.name.clone(), "*".into(), ChangeOperation::Insert)] + vec![(array_id.name.clone(), every_row(), ChangeOperation::Insert)] } PhysicalPlan::Array(ArrayOp::Delete { array_id, .. }) => { - vec![(array_id.name.clone(), "*".into(), ChangeOperation::Delete)] + vec![(array_id.name.clone(), every_row(), ChangeOperation::Delete)] } // Remaining ArrayOp variants are reads or maintenance — no user-data row changed. PhysicalPlan::Array(_) => Vec::new(), @@ -278,13 +294,14 @@ pub(super) fn extract_write_metadata( // Vector is normally a Document secondary index — publishing here would duplicate. // `DirectUpsert` is the exception: the sole write for a vector-primary collection. + // The row carries no user key, so its identity is the decimal surrogate. PhysicalPlan::Vector(VectorOp::DirectUpsert { collection, surrogate, .. }) => vec![( collection.to_string(), - surrogate_to_doc_id(*surrogate), + StorageKey::for_surrogate(*surrogate).to_identity(), ChangeOperation::Insert, )], PhysicalPlan::Vector(_) => Vec::new(), @@ -303,7 +320,7 @@ pub(super) fn extract_write_metadata( .. }) => vec![( collection.to_string(), - document_id.clone(), + RowIdentity::from_user_key(document_id.clone()), ChangeOperation::Insert, )], PhysicalPlan::Crdt(CrdtOp::ListInsert { @@ -312,7 +329,7 @@ pub(super) fn extract_write_metadata( .. }) => vec![( collection.to_string(), - document_id.clone(), + RowIdentity::from_user_key(document_id.clone()), ChangeOperation::Insert, )], PhysicalPlan::Crdt(CrdtOp::ListDelete { @@ -321,7 +338,7 @@ pub(super) fn extract_write_metadata( .. }) => vec![( collection.to_string(), - document_id.clone(), + RowIdentity::from_user_key(document_id.clone()), ChangeOperation::Delete, )], PhysicalPlan::Crdt(CrdtOp::ListMove { @@ -335,12 +352,12 @@ pub(super) fn extract_write_metadata( .. }) => vec![( collection.to_string(), - document_id.clone(), + RowIdentity::from_user_key(document_id.clone()), ChangeOperation::Update, )], // Collection-wide snapshot import: no single document identity. PhysicalPlan::Crdt(CrdtOp::ImportSnapshot { collection, .. }) => { - vec![(collection.to_string(), "*".into(), ChangeOperation::Update)] + vec![(collection.to_string(), every_row(), ChangeOperation::Update)] } // Full replace = Insert, partial update = Update. PhysicalPlan::Crdt(CrdtOp::DocUpsert { @@ -350,7 +367,7 @@ pub(super) fn extract_write_metadata( .. }) => vec![( collection.to_string(), - document_id.clone(), + RowIdentity::from_user_key(document_id.clone()), if *partial { ChangeOperation::Update } else { @@ -363,7 +380,7 @@ pub(super) fn extract_write_metadata( .. }) => vec![( collection.to_string(), - document_id.clone(), + RowIdentity::from_user_key(document_id.clone()), ChangeOperation::Delete, )], // Remaining CrdtOp variants are reads, history maintenance, or config/DDL. @@ -389,15 +406,13 @@ pub(super) fn extract_write_metadata( /// Map a `ClusterArrayOp` to its CDC change metadata. Shared by the /// `PhysicalPlan::ClusterArray` arm and `publish_cluster_array_change_events`, /// which holds the op by reference to avoid cloning the write batch. -pub(crate) fn cluster_array_change_meta( - op: &ClusterArrayOp, -) -> Vec<(String, String, ChangeOperation)> { +pub(crate) fn cluster_array_change_meta(op: &ClusterArrayOp) -> Vec { match op { ClusterArrayOp::Put { array_id, .. } => { - vec![(array_id.name.clone(), "*".into(), ChangeOperation::Insert)] + vec![(array_id.name.clone(), every_row(), ChangeOperation::Insert)] } ClusterArrayOp::Delete { array_id, .. } => { - vec![(array_id.name.clone(), "*".into(), ChangeOperation::Delete)] + vec![(array_id.name.clone(), every_row(), ChangeOperation::Delete)] } // Slice/Agg are reads — no row changed. ClusterArrayOp::Slice { .. } | ClusterArrayOp::Agg { .. } => Vec::new(), @@ -446,8 +461,16 @@ mod tests { assert_eq!( meta, vec![ - ("users".into(), "u1".into(), ChangeOperation::Insert), - ("users".into(), "u2".into(), ChangeOperation::Delete), + ( + "users".into(), + RowIdentity::from_user_key("u1"), + ChangeOperation::Insert + ), + ( + "users".into(), + RowIdentity::from_user_key("u2"), + ChangeOperation::Delete + ), ] ); } @@ -471,11 +494,7 @@ mod tests { let meta = extract_write_metadata(&plan, TenantId::new(1)); assert_eq!( meta, - vec![( - "metrics".to_string(), - "*".to_string(), - ChangeOperation::Insert - )] + vec![("metrics".to_string(), every_row(), ChangeOperation::Insert)] ); } @@ -489,11 +508,7 @@ mod tests { let meta = extract_write_metadata(&plan, TenantId::new(1)); assert_eq!( meta, - vec![( - "metrics".to_string(), - "*".to_string(), - ChangeOperation::Delete - )] + vec![("metrics".to_string(), every_row(), ChangeOperation::Delete)] ); } @@ -508,11 +523,7 @@ mod tests { let meta = extract_write_metadata(&plan, TenantId::new(1)); assert_eq!( meta, - vec![( - "genome".to_string(), - "*".to_string(), - ChangeOperation::Insert - )] + vec![("genome".to_string(), every_row(), ChangeOperation::Insert)] ); } @@ -527,11 +538,7 @@ mod tests { let meta = extract_write_metadata(&plan, TenantId::new(1)); assert_eq!( meta, - vec![( - "genome".to_string(), - "*".to_string(), - ChangeOperation::Delete - )] + vec![("genome".to_string(), every_row(), ChangeOperation::Delete)] ); } @@ -547,11 +554,7 @@ mod tests { let meta = extract_write_metadata(&plan, TenantId::new(1)); assert_eq!( meta, - vec![( - "genome".to_string(), - "*".to_string(), - ChangeOperation::Insert - )] + vec![("genome".to_string(), every_row(), ChangeOperation::Insert)] ); } @@ -567,11 +570,7 @@ mod tests { let meta = extract_write_metadata(&plan, TenantId::new(1)); assert_eq!( meta, - vec![( - "genome".to_string(), - "*".to_string(), - ChangeOperation::Delete - )] + vec![("genome".to_string(), every_row(), ChangeOperation::Delete)] ); } @@ -634,13 +633,13 @@ mod tests { rls_filters: Vec::new(), }); let meta = extract_write_metadata(&plan, TenantId::new(1)); - // The row id is the document storage key, so a consumer can address - // the row the event describes. + // The row id is the client identity of the row, the decimal + // surrogate, so a consumer can address the row the event describes. assert_eq!( meta, vec![( "embeddings".to_string(), - "0000002a".to_string(), + RowIdentity::from_user_key("42"), ChangeOperation::Insert )] ); @@ -675,7 +674,7 @@ mod tests { meta, vec![( "users".to_string(), - "u1".to_string(), + RowIdentity::from_user_key("u1"), ChangeOperation::Insert )] ); @@ -698,12 +697,12 @@ mod tests { vec![ ( "inventory_a".to_string(), - "sword".to_string(), + RowIdentity::from_user_key("sword"), ChangeOperation::Delete ), ( "inventory_b".to_string(), - "sword".to_string(), + RowIdentity::from_user_key("sword"), ChangeOperation::Insert ), ] diff --git a/nodedb/src/control/server/dispatch_utils/change_events/publish.rs b/nodedb/src/control/server/dispatch_utils/change_events/publish.rs index 8e855e04c..0a32f0dcb 100644 --- a/nodedb/src/control/server/dispatch_utils/change_events/publish.rs +++ b/nodedb/src/control/server/dispatch_utils/change_events/publish.rs @@ -5,12 +5,11 @@ //! change stream plus the cluster-wide NOTIFY fan-out. use crate::bridge::envelope::{PhysicalPlan, Response}; -use crate::control::change_stream::ChangeOperation; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId}; use nodedb_physical::physical_plan::ClusterArrayOp; -use super::extract::{cluster_array_change_meta, extract_write_metadata}; +use super::extract::{WriteChangeMeta, cluster_array_change_meta, extract_write_metadata}; /// Current wall-clock time as milliseconds since Unix epoch. /// @@ -58,10 +57,10 @@ fn publish_change_event( shared: &SharedState, tenant_id: TenantId, database_id: DatabaseId, - change_meta: (String, String, ChangeOperation), + change_meta: WriteChangeMeta, lsn: nodedb_types::Lsn, ) { - let (collection, doc_id, op) = change_meta; + let (collection, document_id, op) = change_meta; if !is_timeseries_cdc_enabled(shared, database_id, tenant_id, &collection) { return; } @@ -71,7 +70,7 @@ fn publish_change_event( lsn, tenant_id, collection, - document_id: doc_id, + document_id, operation: op, timestamp_ms: current_timestamp_ms(), after: None, @@ -104,7 +103,7 @@ fn publish_change_event( /// uses [`publish_origin_change_events`] and never names this type. pub(crate) struct WriteChangeSet { /// One tuple per logical row change — see `extract_write_metadata`. - metas: Vec<(String, String, ChangeOperation)>, + metas: Vec, } /// Derive a plan's change events. Pure: it matches over the plan and clones out diff --git a/nodedb/src/control/server/http/routes/cdc.rs b/nodedb/src/control/server/http/routes/cdc.rs index af3b2340e..dda094363 100644 --- a/nodedb/src/control/server/http/routes/cdc.rs +++ b/nodedb/src/control/server/http/routes/cdc.rs @@ -227,7 +227,7 @@ fn reset_error(_: ReplayError) -> ApiError { } fn change_json(event: &SequencedChangeEvent) -> serde_json::Value { - serde_json::json!({ "operation": event.operation.as_str(), "document_id": event.document_id, "timestamp_ms": event.timestamp_ms, "lsn": event.lsn.as_u64(), "collection": event.collection, "cursor": event.cursor().to_string() }) + serde_json::json!({ "operation": event.operation.as_str(), "document_id": event.document_id.as_str(), "timestamp_ms": event.timestamp_ms, "lsn": event.lsn.as_u64(), "collection": event.collection, "cursor": event.cursor().to_string() }) } fn format_sse_event(event: &SequencedChangeEvent) -> Event { diff --git a/nodedb/src/control/server/http/routes/subscribe.rs b/nodedb/src/control/server/http/routes/subscribe.rs index 072b2d2cf..e827d4ae1 100644 --- a/nodedb/src/control/server/http/routes/subscribe.rs +++ b/nodedb/src/control/server/http/routes/subscribe.rs @@ -52,7 +52,7 @@ impl From<&ChangeEvent> for ChangeNotification { fn from(e: &ChangeEvent) -> Self { Self { event: e.operation.as_str().to_string(), - doc_id: e.document_id.clone(), + doc_id: e.document_id.to_string(), collection: e.collection.clone(), lsn: e.lsn.as_u64(), timestamp_ms: e.timestamp_ms, @@ -189,7 +189,7 @@ mod tests { lsn: Lsn::new(1), tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: "o1".into(), + document_id: nodedb_types::RowIdentity::from_user_key("o1"), operation: ChangeOperation::Insert, timestamp_ms: 0, after: None, @@ -206,7 +206,7 @@ mod tests { lsn: Lsn::new(42), tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: "o1".into(), + document_id: nodedb_types::RowIdentity::from_user_key("o1"), operation: ChangeOperation::Update, timestamp_ms: 12345, after: None, diff --git a/nodedb/src/control/server/http/routes/ws_rpc/format.rs b/nodedb/src/control/server/http/routes/ws_rpc/format.rs index a1b6553c5..0b66f3974 100644 --- a/nodedb/src/control/server/http/routes/ws_rpc/format.rs +++ b/nodedb/src/control/server/http/routes/ws_rpc/format.rs @@ -18,7 +18,7 @@ pub fn format_sequenced_live_notification(sub_id: u64, event: &SequencedChangeEv "database_id": event.database_id().as_u64(), "collection": event.collection, "operation": event.operation.as_str(), - "document_id": event.document_id, + "document_id": event.document_id.as_str(), "timestamp_ms": event.timestamp_ms, } }) @@ -30,7 +30,7 @@ pub fn format_resume_notification(event: &SequencedChangeEvent) -> String { serde_json::json!({"method": "change", "params": { "cursor": event.cursor().to_string(), "wal_lsn": event.lsn.as_u64(), "database_id": event.database_id().as_u64(), "collection": event.collection, "operation": event.operation.as_str(), - "document_id": event.document_id, "timestamp_ms": event.timestamp_ms, + "document_id": event.document_id.as_str(), "timestamp_ms": event.timestamp_ms, }}) .to_string() } diff --git a/nodedb/src/control/server/shared/ddl/neutral/show_changes.rs b/nodedb/src/control/server/shared/ddl/neutral/show_changes.rs index fd97473b1..d57ab15f2 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/show_changes.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/show_changes.rs @@ -116,7 +116,7 @@ pub fn show_changes( ); row.insert( "document_id".to_string(), - JsonValue::String(change.document_id.clone()), + JsonValue::String(change.document_id.to_string()), ); row.insert( "timestamp_ms".to_string(), @@ -153,6 +153,7 @@ mod tests { use crate::control::security::identity::{AuthMethod, DatabaseSet, Role}; use crate::types::{Lsn, TenantId}; use crate::wal::WalManager; + use nodedb_types::RowIdentity; fn test_state() -> (tempfile::TempDir, Arc) { let dir = tempfile::tempdir().expect("create test directory"); @@ -184,7 +185,7 @@ mod tests { lsn: Lsn::new(1), tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: "hidden-order".into(), + document_id: RowIdentity::from_user_key("hidden-order"), operation: ChangeOperation::Insert, timestamp_ms: 1, after: None, @@ -216,7 +217,7 @@ mod tests { lsn, tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: document_id.into(), + document_id: RowIdentity::from_user_key(document_id), operation: ChangeOperation::Insert, timestamp_ms: 1, after: None, @@ -263,7 +264,7 @@ mod tests { lsn: Lsn::new(1), tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: "tenant-1-order".into(), + document_id: RowIdentity::from_user_key("tenant-1-order"), operation: ChangeOperation::Insert, timestamp_ms: 1, after: None, @@ -272,7 +273,7 @@ mod tests { lsn: Lsn::new(2), tenant_id: TenantId::new(2), collection: "orders".into(), - document_id: "tenant-2-order".into(), + document_id: RowIdentity::from_user_key("tenant-2-order"), operation: ChangeOperation::Insert, timestamp_ms: 1, after: None, diff --git a/nodedb/src/control/server/shared/session/live.rs b/nodedb/src/control/server/shared/session/live.rs index 79a350c24..2be00d6fe 100644 --- a/nodedb/src/control/server/shared/session/live.rs +++ b/nodedb/src/control/server/shared/session/live.rs @@ -164,6 +164,7 @@ mod tests { use super::*; use crate::control::change_stream::{ChangeEvent, ChangeOperation, ChangeStream}; use crate::types::{DatabaseId, Lsn, TenantId}; + use nodedb_types::RowIdentity; #[test] fn live_subscription_store_and_check() { @@ -210,7 +211,7 @@ mod tests { lsn: Lsn::new(1), tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: "o42".into(), + document_id: RowIdentity::from_user_key("o42"), operation: ChangeOperation::Insert, timestamp_ms: 0, after: None, @@ -244,7 +245,7 @@ mod tests { lsn: Lsn::new(1), tenant_id: TenantId::new(1), collection: "users".into(), - document_id: "u1".into(), + document_id: RowIdentity::from_user_key("u1"), operation: ChangeOperation::Update, timestamp_ms: 0, after: None, @@ -290,7 +291,7 @@ mod tests { lsn, tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: document_id.into(), + document_id: RowIdentity::from_user_key(document_id), operation: ChangeOperation::Insert, timestamp_ms: 1, after: None, @@ -338,7 +339,7 @@ mod tests { lsn, tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: document_id.into(), + document_id: RowIdentity::from_user_key(document_id), operation: ChangeOperation::Insert, timestamp_ms: 1, after: None, @@ -356,7 +357,7 @@ mod tests { lsn: Lsn::new(3), tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: "later-order".into(), + document_id: RowIdentity::from_user_key("later-order"), operation: ChangeOperation::Insert, timestamp_ms: 1, after: None, @@ -389,7 +390,7 @@ mod tests { lsn, tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: "lagged-order".into(), + document_id: RowIdentity::from_user_key("lagged-order"), operation: ChangeOperation::Insert, timestamp_ms: 1, after: None, @@ -399,7 +400,7 @@ mod tests { lsn: Lsn::new(3), tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: "healthy-order".into(), + document_id: RowIdentity::from_user_key("healthy-order"), operation: ChangeOperation::Insert, timestamp_ms: 1, after: None, @@ -444,7 +445,7 @@ mod tests { lsn: Lsn::new(1), tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: "order".into(), + document_id: RowIdentity::from_user_key("order"), operation: ChangeOperation::Insert, timestamp_ms: 1, after: None, @@ -496,7 +497,7 @@ mod tests { lsn: Lsn::new(3), tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: "old-database-order".into(), + document_id: RowIdentity::from_user_key("old-database-order"), operation: ChangeOperation::Insert, timestamp_ms: 1, after: None, diff --git a/nodedb/src/data/executor/handlers/graph_expansion.rs b/nodedb/src/data/executor/handlers/graph_expansion.rs index 6a2f90d64..42b5df82e 100644 --- a/nodedb/src/data/executor/handlers/graph_expansion.rs +++ b/nodedb/src/data/executor/handlers/graph_expansion.rs @@ -45,12 +45,16 @@ pub(in crate::data::executor) struct GraphExpansionParams<'a> { /// Reached nodes, in both currencies. /// -/// `reached` is the surrogate set — the form that intersects with another -/// engine's candidates. `names` / `distances` are the same nodes resolved for -/// ranking and for the response, produced by a single pass at the end. +/// `reached` is the surrogate-bound subset with hop distances — the form that +/// fuses with another engine's candidates on the storage key. `names` / +/// `distances` are every reached node resolved by name, for ranking by name +/// and for the response. Both come from a single pass at the end. pub(in crate::data::executor) struct GraphExpansion { pub names: Vec, pub distances: HashMap, + /// `(surrogate, hop distance)` for each reached node that carries a + /// surrogate. Nodes counted in `unaddressable` are absent here. + pub reached: Vec<(Surrogate, usize)>, pub truncated: bool, /// Reached nodes that carry no surrogate, so they are traversed *through* /// but can never intersect another engine's candidates. Carried to the @@ -89,6 +93,7 @@ impl CoreLoop { return GraphExpansion { names: Vec::new(), distances: HashMap::new(), + reached: Vec::new(), truncated: false, unaddressable: 0, }; @@ -153,6 +158,7 @@ impl CoreLoop { ) -> GraphExpansion { let mut names = Vec::with_capacity(hops.distances.len()); let mut distances = HashMap::with_capacity(hops.distances.len()); + let mut reached = Vec::with_capacity(hops.reached.len() as usize); for &(local, depth) in &hops.distances { // Local ids come from this partition's own walk, so the name lookup // is total; skipping rather than unwrapping keeps a torn index from @@ -161,10 +167,16 @@ impl CoreLoop { names.push(name.to_string()); distances.insert(name.to_string(), depth); } + // `node_surrogate_raw` yields the ZERO sentinel for an unbound node. + let raw = partition.node_surrogate_raw(local); + if raw != 0 { + reached.push((Surrogate::new(raw), depth)); + } } GraphExpansion { names, distances, + reached, truncated: hops.truncated, unaddressable: hops.unaddressable, } diff --git a/nodedb/src/data/executor/handlers/graph_rag.rs b/nodedb/src/data/executor/handlers/graph_rag.rs index 8f1aa0d31..437ac7742 100644 --- a/nodedb/src/data/executor/handlers/graph_rag.rs +++ b/nodedb/src/data/executor/handlers/graph_rag.rs @@ -16,7 +16,7 @@ use std::collections::HashMap; -use nodedb_types::Surrogate; +use nodedb_types::{RowIdentity, Surrogate}; use nodedb_vector::SearchResult; use tracing::{debug, warn}; @@ -34,19 +34,20 @@ use crate::query::fusion::{FusedResult, RankedResult, reciprocal_rank_fusion_wei /// /// - `Vec` is the raw HNSW output, for reporting candidate counts. /// - `HashMap` maps each hit's *reporting key* to `(rank, distance)`. That key -/// is what RRF fuses on and what the response returns, so it has to be a name. +/// is what RRF fuses on and what the response returns: the client-visible +/// identity of the row. /// - `Vec` is the same hits in the identity currency, ready to seed /// graph expansion with no translation. type VectorNodeScores = ( Vec, - HashMap, + HashMap, Vec, ); /// Parameters for `build_rag_response`. pub(in crate::data::executor) struct RagResponseParams<'a> { - pub fused: &'a [FusedResult], - pub vector_scores: &'a HashMap, + pub fused: &'a [FusedResult], + pub vector_scores: &'a HashMap, pub hop_distances: &'a HashMap, pub vector_candidate_count: usize, pub graph_expanded_count: usize, @@ -131,7 +132,7 @@ impl CoreLoop { let (vector_k, graph_k) = rrf_k; - let vector_list: Vec = vector_scores + let vector_list: Vec> = vector_scores .iter() .map(|(node_id, (rank, dist))| RankedResult { document_id: node_id.clone(), @@ -141,7 +142,8 @@ impl CoreLoop { }) .collect(); - let graph_list = graph_nodes_to_ranked_results(&expanded_nodes, &hop_distances); + let graph_expanded_count = expanded_nodes.len(); + let graph_list = graph_nodes_to_ranked_results(expanded_nodes, &hop_distances); let fused = reciprocal_rank_fusion_weighted( &[vector_list, graph_list], @@ -156,7 +158,7 @@ impl CoreLoop { vector_scores: &vector_scores, hop_distances: &hop_distances, vector_candidate_count: vector_results.len(), - graph_expanded_count: expanded_nodes.len(), + graph_expanded_count, bfs_truncated, graph_unaddressable: unaddressable, op_name: "graph rag fusion", @@ -207,7 +209,7 @@ impl CoreLoop { // sentinel could match nothing and leaked an internal index id into // the response's `node_id`. let csr = self.csr_partition(database_id, tenant_id); - let mut vector_scores: HashMap = HashMap::new(); + let mut vector_scores: HashMap = HashMap::new(); let mut seeds: Vec = Vec::with_capacity(vector_results.len()); for (rank, result) in vector_results.iter().enumerate() { let surrogate = index.get_surrogate(result.id); @@ -217,17 +219,13 @@ impl CoreLoop { let key = match surrogate { Some(s) => csr .and_then(|c| c.node_id_for_surrogate(s)) - .map(str::to_string) - .unwrap_or_else(|| { - crate::engine::document::store::RowIdentity::for_surrogate(s) - .as_str() - .to_string() - }), + .map(RowIdentity::from_user_key) + .unwrap_or_else(|| RowIdentity::for_surrogate(s)), // No surrogate at all: the vector entry predates surrogate // plumbing, so it has no cross-engine identity. It still ranks // in the vector leg under a key that deliberately matches // nothing else. - None => format!("__unbound_{}", result.id), + None => RowIdentity::from_user_key(format!("__unbound_{}", result.id)), }; vector_scores.insert(key, (rank, result.distance)); } @@ -250,12 +248,13 @@ impl CoreLoop { .map(|f| { let (vector_rank, vector_distance) = p .vector_scores - .get(f.document_id.as_str()) + .get(&f.document_id) .map(|(rank, dist)| (Some(*rank), Some(*dist))) .unwrap_or((None, None)); let hop_distance = p.hop_distances.get(f.document_id.as_str()).copied(); GraphRagResult { - node_id: f.document_id.clone(), + // The identity is rendered here, at the response envelope. + node_id: f.document_id.clone().into_string(), rrf_score: f.rrf_score, vector_rank, vector_distance, @@ -292,29 +291,28 @@ impl CoreLoop { /// Sort expanded graph nodes by hop distance and convert to `RankedResult` list. /// -/// Used by 2-source GraphRAG, 3-source GraphRAG triple, and 3-source hybrid -/// text search to avoid duplicating the sort-and-rank pattern. +/// A graph node name is the row's client identity, so the list keys on +/// `RowIdentity`. Used by 2-source GraphRAG and 3-source GraphRAG triple. pub(super) fn graph_nodes_to_ranked_results( - expanded_nodes: &[String], + expanded_nodes: Vec, hop_distances: &HashMap, -) -> Vec { - let mut sorted: Vec<(&str, usize)> = expanded_nodes - .iter() +) -> Vec> { + // Takes `expanded_nodes` by value so the name moves straight into + // `RowIdentity` below instead of being copied from a borrow. + let mut sorted: Vec<(String, usize)> = expanded_nodes + .into_iter() .map(|node| { - let dist = hop_distances - .get(node.as_str()) - .copied() - .unwrap_or(usize::MAX); - (node.as_str(), dist) + let dist = hop_distances.get(&node).copied().unwrap_or(usize::MAX); + (node, dist) }) .collect(); - sorted.sort_by(|a, b| a.1.cmp(&b.1).then_with(|| a.0.cmp(b.0))); + sorted.sort_by(|a, b| a.1.cmp(&b.1).then_with(|| a.0.cmp(&b.0))); sorted .into_iter() .enumerate() .map(|(rank, (node_id, hop_dist))| RankedResult { - document_id: node_id.to_string(), + document_id: RowIdentity::from_user_key(node_id), rank, score: hop_dist as f32, source: "graph", diff --git a/nodedb/src/data/executor/handlers/graph_rag_triple.rs b/nodedb/src/data/executor/handlers/graph_rag_triple.rs index a87cd358c..11b0799b2 100644 --- a/nodedb/src/data/executor/handlers/graph_rag_triple.rs +++ b/nodedb/src/data/executor/handlers/graph_rag_triple.rs @@ -12,6 +12,7 @@ use nodedb_fts::FtsSearchParams; use nodedb_fts::posting::QueryMode; +use nodedb_types::RowIdentity; use tracing::debug; use crate::bridge::envelope::Response; @@ -125,7 +126,7 @@ impl CoreLoop { let (vector_k, text_k, graph_k) = rrf_k; - let vector_list: Vec = vector_scores + let vector_list: Vec> = vector_scores .iter() .map(|(node_id, (rank, dist))| RankedResult { document_id: node_id.clone(), @@ -135,20 +136,19 @@ impl CoreLoop { }) .collect(); - let text_list: Vec = text_results + let text_list: Vec> = text_results .iter() .enumerate() .map(|(rank, r)| RankedResult { - document_id: crate::engine::document::store::RowIdentity::for_surrogate(r.doc_id) - .as_str() - .to_string(), + document_id: RowIdentity::for_surrogate(r.doc_id), rank, score: r.score, source: "text", }) .collect(); - let graph_list = graph_nodes_to_ranked_results(&expanded_nodes, &hop_distances); + let graph_expanded_count = expanded_nodes.len(); + let graph_list = graph_nodes_to_ranked_results(expanded_nodes, &hop_distances); let fused = reciprocal_rank_fusion_weighted( &[vector_list, text_list, graph_list], @@ -163,7 +163,7 @@ impl CoreLoop { vector_scores: &vector_scores, hop_distances: &hop_distances, vector_candidate_count: vector_results.len(), - graph_expanded_count: expanded_nodes.len(), + graph_expanded_count, bfs_truncated, graph_unaddressable: unaddressable, op_name: "graph rag fusion triple", diff --git a/nodedb/src/data/executor/handlers/hybrid_key.rs b/nodedb/src/data/executor/handlers/hybrid_key.rs new file mode 100644 index 000000000..cdaff8ea4 --- /dev/null +++ b/nodedb/src/data/executor/handlers/hybrid_key.rs @@ -0,0 +1,78 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The key the vector, text, and graph legs of a hybrid search fuse on. +//! +//! RRF fuses on key equality, so every leg must rank under one key type. The +//! text and graph legs always carry a surrogate. A vector hit whose HNSW node +//! has no surrogate binding carries none, and must never be misread as one. + +use std::fmt; + +use nodedb_types::{StorageKey, Surrogate}; + +/// Prefix of the rendered key for a vector hit with no surrogate binding. +/// The Control Plane hybrid translator passes such a `doc_id` through untouched. +pub(in crate::data::executor) const HEADLESS_SENTINEL_PREFIX: &str = "__local_"; + +/// One hybrid-search leg hit, in the key space RRF fuses on. +/// +/// `Bound` is a row with a surrogate: it fuses across legs and renders as the +/// storage key the response envelope carries. `Headless` is a vector hit with +/// no surrogate binding: it ranks in the vector leg only, fuses with nothing, +/// and renders as the `__local_` sentinel. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)] +pub(in crate::data::executor) enum HybridFusionKey { + Bound(StorageKey), + Headless(u32), +} + +impl HybridFusionKey { + /// The key of a row that carries `surrogate`. + pub fn for_surrogate(surrogate: Surrogate) -> Self { + Self::Bound(StorageKey::for_surrogate(surrogate)) + } + + /// The storage key of a bound row. `None` for a headless hit. + pub fn storage_key(&self) -> Option { + match self { + Self::Bound(key) => Some(*key), + Self::Headless(_) => None, + } + } +} + +impl fmt::Display for HybridFusionKey { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::Bound(key) => write!(f, "{key}"), + Self::Headless(local_id) => write!(f, "{HEADLESS_SENTINEL_PREFIX}{local_id}"), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn bound_renders_storage_key_and_headless_renders_sentinel() { + let bound = HybridFusionKey::for_surrogate(Surrogate::new(42)); + assert_eq!(bound.to_string(), "0000002a"); + assert_eq!( + bound.storage_key(), + Some(StorageKey::for_surrogate(Surrogate::new(42))) + ); + + let headless = HybridFusionKey::Headless(7); + assert_eq!(headless.to_string(), "__local_7"); + assert_eq!(headless.storage_key(), None); + } + + #[test] + fn headless_never_equals_a_bound_key_with_the_same_number() { + assert_ne!( + HybridFusionKey::Headless(7), + HybridFusionKey::for_surrogate(Surrogate::new(7)) + ); + } +} diff --git a/nodedb/src/data/executor/handlers/hybrid_overlay.rs b/nodedb/src/data/executor/handlers/hybrid_overlay.rs index ddfb88307..c9c053616 100644 --- a/nodedb/src/data/executor/handlers/hybrid_overlay.rs +++ b/nodedb/src/data/executor/handlers/hybrid_overlay.rs @@ -26,18 +26,26 @@ //! The graph leg's RYOW is a separate concern and is deliberately not touched //! here — the triple handler still reads committed graph state. +use std::collections::HashMap; + use nodedb_fts::posting::TextSearchResult; use nodedb_types::{Surrogate, SurrogateBitmap}; +use super::hybrid_key::HybridFusionKey; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::transaction::overlay::{FtsMergeParams, VectorMergeParams}; -use crate::engine::document::store::surrogate_to_doc_id; use crate::engine::vector::DistanceMetric; use crate::engine::vector::SearchResult; use crate::engine::vector::collection::VectorCollection; use crate::query::fusion::RankedResult; use crate::types::{DatabaseId, TenantId, TxnId}; +/// The two RRF-ready legs of a hybrid search, keyed on one fusion key space. +pub(in crate::data::executor) type HybridRankedLegs = ( + Vec>, + Vec>, +); + /// Scope for one hybrid overlay splice: the active transaction, its /// `(database, tenant, collection)` target, the vector and text query inputs, /// the per-leg over-fetch bound, and any surrogate prefilter applied to the @@ -62,33 +70,35 @@ impl CoreLoop { /// `text_results` (base BM25 hits). This reuses the exact single-source /// overlay merges to re-score the staged puts and drop the staged /// tombstones, then emits `(vector_ranked, text_ranked)` keyed by - /// surrogate-hex doc-id — the shared RRF key space the caller fuses on. + /// [`HybridFusionKey`], the shared RRF key space the caller fuses on. pub(in crate::data::executor) fn hybrid_ranked_with_overlay( &self, params: HybridOverlayParams<'_>, vector_results: &[SearchResult], vector_collection: Option<&VectorCollection>, text_results: &[TextSearchResult], - ) -> crate::Result<(Vec, Vec)> { + ) -> crate::Result { // Base committed legs, in the shape each single-source overlay merge // consumes: vector hits carry the surrogate-resolved id, text scores // carry the FTS surrogate + score + fuzzy flag. let mut vector_hits: Vec<_> = vector_results .iter() - .map(|r| { - let mut hit = - super::vector_search::build_search_hit(vector_collection, r.id, r.distance); - // Pin the committed-path fusion doc_id now (surrogate hex, or - // the `__local_{id}` sentinel for a headless row) so a headless - // base hit's raw local id is never later misread as a global - // surrogate. Staged hits added by the merge carry `doc_id: - // None` and fall back to their (real) surrogate when the ranked - // list is built below — matching the autocommit branch exactly. - hit.doc_id = Some(super::vector_search::vector_leg_doc_id( - vector_collection, - r.id, - )); - hit + .map(|r| super::vector_search::build_search_hit(vector_collection, r.id, r.distance)) + .collect(); + // Pin each base hit's fusion key by its `id` before the merge runs. A + // headless base hit carries a raw local id in `id`, so without this + // pin the ranked list below would misread it as a global surrogate. + // The merge updates a same-id hit in place, so the pinned key still + // names the row after a staged put; staged hits the merge adds carry + // no pin and resolve to their (real) surrogate. + let base_keys: HashMap = vector_results + .iter() + .zip(vector_hits.iter()) + .map(|(r, hit)| { + ( + hit.id, + super::vector_search::vector_leg_key(vector_collection, r.id), + ) }) .collect(); let mut text_scored: Vec<(Surrogate, f32, bool)> = text_results @@ -151,20 +161,19 @@ impl CoreLoop { &mut text_scored, )?; - // Rebuild the RRF-ready ranked lists from the merged legs. Both keys are - // surrogate-hex doc-ids so the vector and text legs fuse on one key - // space (matching the committed-only construction in the handlers). + // Rebuild the RRF-ready ranked lists from the merged legs. Both legs + // key on `HybridFusionKey`, matching the committed-only construction + // in the handlers. let vector_ranked = vector_hits .iter() .enumerate() .map(|(rank, hit)| RankedResult { - // Base hits carry their committed-path doc_id (surrogate hex or - // `__local_` sentinel); staged hits added by the merge have - // `doc_id: None` and resolve to their real surrogate. - document_id: hit - .doc_id - .clone() - .unwrap_or_else(|| surrogate_to_doc_id(Surrogate::new(hit.id))), + // A base hit keeps its pinned key; a staged hit the merge added + // carries a real surrogate in `id`. + document_id: base_keys + .get(&hit.id) + .copied() + .unwrap_or_else(|| HybridFusionKey::for_surrogate(Surrogate::new(hit.id))), rank, score: hit.distance, source: "vector", @@ -174,7 +183,7 @@ impl CoreLoop { .iter() .enumerate() .map(|(rank, (surrogate, score, _fuzzy))| RankedResult { - document_id: surrogate_to_doc_id(*surrogate), + document_id: HybridFusionKey::for_surrogate(*surrogate), rank, score: *score, source: "text", diff --git a/nodedb/src/data/executor/handlers/mod.rs b/nodedb/src/data/executor/handlers/mod.rs index f7abad62d..40935615e 100644 --- a/nodedb/src/data/executor/handlers/mod.rs +++ b/nodedb/src/data/executor/handlers/mod.rs @@ -31,6 +31,7 @@ pub mod graph_stats; pub mod graph_temporal; pub mod graph_wcc; pub(super) mod grouping_sets_exec; +pub mod hybrid_key; pub mod hybrid_overlay; pub mod join; pub mod kv; diff --git a/nodedb/src/data/executor/handlers/text_search_hybrid.rs b/nodedb/src/data/executor/handlers/text_search_hybrid.rs index 8702953cd..b73fe1d9c 100644 --- a/nodedb/src/data/executor/handlers/text_search_hybrid.rs +++ b/nodedb/src/data/executor/handlers/text_search_hybrid.rs @@ -131,6 +131,7 @@ impl CoreLoop { // 3. Build ranked lists for weighted RRF. // Higher weight → lower k → steeper rank discount → more influence. + use super::hybrid_key::HybridFusionKey; use crate::query::fusion::{RankedResult, reciprocal_rank_fusion_weighted}; let base_k = 60.0_f64; @@ -145,17 +146,16 @@ impl CoreLoop { base_k * 100.0 }; - // Translate vector local-hnsw IDs to surrogate-hex doc_ids so the - // vector and text legs share the same RRF key space. Headless rows - // (no surrogate binding) fall back to a non-fusable sentinel — - // they cannot match any FTS doc_id, which is the correct behavior. + // Both legs key on `HybridFusionKey` so they fuse on one key space. + // A headless vector row (no surrogate binding) is `Headless` and + // cannot match any FTS hit, which is the correct behavior. // // Inside a transaction, read-your-own-writes: the vector and text legs // must also observe this transaction's staged document writes, folded // in via the shared overlay splice (which reuses the single-source // vector/FTS overlay merges). Outside a transaction the committed-only // construction below runs unchanged. - let (vector_ranked, text_ranked): (Vec, Vec) = + let (vector_ranked, text_ranked): super::hybrid_overlay::HybridRankedLegs = if let Some(txn_id) = task.request.txn_id { match self.hybrid_ranked_with_overlay( super::hybrid_overlay::HybridOverlayParams { @@ -176,25 +176,22 @@ impl CoreLoop { Err(e) => return self.response_error(task, e), } } else { - let vector_ranked: Vec = vector_results + let vector_ranked: Vec> = vector_results .iter() .enumerate() .map(|(rank, r)| RankedResult { - document_id: super::vector_search::vector_leg_doc_id( - vector_collection, - r.id, - ), + document_id: super::vector_search::vector_leg_key(vector_collection, r.id), rank, score: r.distance, source: "vector", }) .collect(); - let text_ranked: Vec = text_results + let text_ranked: Vec> = text_results .iter() .enumerate() .map(|(rank, r)| RankedResult { - document_id: crate::engine::document::store::surrogate_to_doc_id(r.doc_id), + document_id: HybridFusionKey::for_surrogate(r.doc_id), rank, score: r.score, source: "text", @@ -220,16 +217,17 @@ impl CoreLoop { // the collection's registered kind — a tagged map and a plain document // map share the same map header, so the bytes cannot answer it. let body_format = self.sparse_body_format(task.request.database_id, tenant_id, collection); - let results: Vec<_> = fused + // The fused key is rendered once per row here, at the response + // envelope; it is the only place the key becomes text. + let rendered: Vec<(String, &crate::query::fusion::FusedResult)> = fused .iter() .filter(|f| { if rls_filters.is_empty() { return true; } - // `document_id` is a fused-result string several hops from any - // scan; a shape that fails to parse as a storage key is - // treated the same as a row the lookup below could not find. - let Some(key) = nodedb_types::StorageKey::parse(&f.document_id) else { + // A headless hit has no stored row to check the policy against, + // so it is treated the same as a row the lookup cannot find. + let Some(key) = f.document_id.storage_key() else { return false; }; match self @@ -244,20 +242,20 @@ impl CoreLoop { _ => false, } }) - .map(|f| { + .map(|f| (f.document_id.to_string(), f)) + .collect(); + let results: Vec<_> = rendered + .iter() + .map(|(doc_id, f)| { let vector_rank = vector_results.iter().position(|r| { - let doc_id = vector_collection - .and_then(|c| c.get_surrogate(r.id)) - .map(crate::engine::document::store::surrogate_to_doc_id) - .unwrap_or_else(|| format!("__local_{}", r.id)); - doc_id == f.document_id - }); - let text_rank = text_results.iter().position(|r| { - crate::engine::document::store::surrogate_to_doc_id(r.doc_id) == f.document_id + super::vector_search::vector_leg_key(vector_collection, r.id) == f.document_id }); + let text_rank = text_results + .iter() + .position(|r| HybridFusionKey::for_surrogate(r.doc_id) == f.document_id); super::super::response_codec::HybridSearchHit { - doc_id: &f.document_id, + doc_id, score_field: score_alias.unwrap_or("rrf_score"), rrf_score: f.rrf_score, vector_rank, diff --git a/nodedb/src/data/executor/handlers/text_search_triple.rs b/nodedb/src/data/executor/handlers/text_search_triple.rs index b3b96365a..3cfbc73fb 100644 --- a/nodedb/src/data/executor/handlers/text_search_triple.rs +++ b/nodedb/src/data/executor/handlers/text_search_triple.rs @@ -7,21 +7,25 @@ //! 2. BM25 full-text search from the inverted index — top-K by score. //! 3. Graph BFS from `graph_seed_id` up to `graph_depth` hops — scored by hop distance. //! 4. All three ranked lists are fused via `reciprocal_rank_fusion_weighted` with -//! per-source k-constants `(vector_k, text_k, graph_k)`. +//! per-source k-constants `(vector_k, text_k, graph_k)`. Every leg keys on +//! [`HybridFusionKey`], so a graph node fuses with the vector and text hits +//! for the same row through its surrogate. //! 5. Final top-K fused results are materialised with per-source rank diagnostics. use tracing::debug; use nodedb_fts::FtsSearchParams; use nodedb_fts::posting::QueryMode; +use nodedb_types::Surrogate; +use super::hybrid_key::HybridFusionKey; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::graph_expansion::{GraphExpansionParams, GraphSeeds}; -use crate::data::executor::handlers::graph_rag::graph_nodes_to_ranked_results; use crate::data::executor::scan_normalize::sparse_body_to_msgpack; use crate::data::executor::task::ExecutionTask; use crate::engine::graph::edge_store::Direction; +use crate::query::fusion::{FusedResult, RankedResult, reciprocal_rank_fusion_weighted}; /// Parameters for [`CoreLoop::execute_hybrid_search_triple`]. pub(in crate::data::executor) struct HybridSearchTripleParams<'a> { @@ -133,7 +137,6 @@ impl CoreLoop { .unwrap_or_default(); // 3. Graph BFS from seed node. - let edge_label_owned = graph_edge_label.map(str::to_string); // The seed is named by the query itself, so it resolves to a surrogate // once; the walk then runs in the same identity currency as the vector // and text legs it will be fused with. @@ -148,11 +151,7 @@ impl CoreLoop { / self.query_tuning.bfs_bytes_per_node, collection, }); - let (graph_expanded, hop_distances) = (expansion.names, expansion.distances); - // 4. Build ranked lists. - use crate::query::fusion::{RankedResult, reciprocal_rank_fusion_weighted}; - let _ = edge_label_owned; // consumed above // Inside a transaction, read-your-own-writes: the vector and text legs // must also observe this transaction's staged document writes, folded @@ -160,7 +159,7 @@ impl CoreLoop { // vector/FTS overlay merges). The graph leg's RYOW is a separate // concern and is not folded in here. Outside a transaction the // committed-only construction below runs unchanged. - let (vector_ranked, text_ranked): (Vec, Vec) = + let (vector_ranked, text_ranked): super::hybrid_overlay::HybridRankedLegs = if let Some(txn_id) = task.request.txn_id { match self.hybrid_ranked_with_overlay( super::hybrid_overlay::HybridOverlayParams { @@ -181,25 +180,22 @@ impl CoreLoop { Err(e) => return self.response_error(task, e), } } else { - let vector_ranked: Vec = vector_results + let vector_ranked: Vec> = vector_results .iter() .enumerate() .map(|(rank, r)| RankedResult { - document_id: super::vector_search::vector_leg_doc_id( - vector_collection, - r.id, - ), + document_id: super::vector_search::vector_leg_key(vector_collection, r.id), rank, score: r.distance, source: "vector", }) .collect(); - let text_ranked: Vec = text_results + let text_ranked: Vec> = text_results .iter() .enumerate() .map(|(rank, r)| RankedResult { - document_id: crate::engine::document::store::surrogate_to_doc_id(r.doc_id), + document_id: HybridFusionKey::for_surrogate(r.doc_id), rank, score: r.score, source: "text", @@ -208,7 +204,7 @@ impl CoreLoop { (vector_ranked, text_ranked) }; - let graph_ranked = graph_nodes_to_ranked_results(&graph_expanded, &hop_distances); + let graph_ranked = graph_reached_to_ranked_keys(&expansion.reached); let (k_vector, k_text, k_graph) = rrf_k; let fused = reciprocal_rank_fusion_weighted( @@ -228,16 +224,17 @@ impl CoreLoop { // plain document map share the same map header, so the bytes cannot // answer it. let body_format = self.sparse_body_format(task.request.database_id, tenant_id, collection); - let results: Vec<_> = fused + // The fused key is rendered once per row here, at the response + // envelope; it is the only place the key becomes text. + let rendered: Vec<(String, &FusedResult)> = fused .iter() .filter(|f| { if rls_filters.is_empty() { return true; } - // `document_id` is a fused-result string several hops from any - // scan; a shape that fails to parse as a storage key is - // treated the same as a row the lookup below could not find. - let Some(key) = nodedb_types::StorageKey::parse(&f.document_id) else { + // A headless hit has no stored row to check the policy against, + // so it is treated the same as a row the lookup cannot find. + let Some(key) = f.document_id.storage_key() else { return false; }; match self @@ -252,17 +249,20 @@ impl CoreLoop { _ => false, } }) - .map(|f| { + .map(|f| (f.document_id.to_string(), f)) + .collect(); + let results: Vec<_> = rendered + .iter() + .map(|(doc_id, f)| { let vector_rank = vector_results.iter().position(|r| { - super::vector_search::vector_leg_doc_id(vector_collection, r.id) - == f.document_id - }); - let text_rank = text_results.iter().position(|r| { - crate::engine::document::store::surrogate_to_doc_id(r.doc_id) == f.document_id + super::vector_search::vector_leg_key(vector_collection, r.id) == f.document_id }); + let text_rank = text_results + .iter() + .position(|r| HybridFusionKey::for_surrogate(r.doc_id) == f.document_id); super::super::response_codec::HybridSearchHit { - doc_id: &f.document_id, + doc_id, score_field: score_alias.unwrap_or("rrf_score"), rrf_score: f.rrf_score, vector_rank, @@ -285,3 +285,52 @@ impl CoreLoop { } } } + +/// Rank the surrogate-bound nodes an expansion reached by hop distance, keyed +/// on [`HybridFusionKey`] so they fuse with the vector and text legs. +/// +/// Ties on hop distance break on the surrogate, so the rank order is +/// deterministic. Nodes without a surrogate never enter the list: they have +/// no cross-engine identity to fuse on. +fn graph_reached_to_ranked_keys( + reached: &[(Surrogate, usize)], +) -> Vec> { + let mut sorted: Vec<(Surrogate, usize)> = reached.to_vec(); + sorted.sort_by(|a, b| a.1.cmp(&b.1).then_with(|| a.0.cmp(&b.0))); + sorted + .into_iter() + .enumerate() + .map(|(rank, (surrogate, hop_dist))| RankedResult { + document_id: HybridFusionKey::for_surrogate(surrogate), + rank, + score: hop_dist as f32, + source: "graph", + }) + .collect() +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn graph_leg_ranks_by_hop_then_surrogate() { + let reached = [ + (Surrogate::new(9), 2), + (Surrogate::new(4), 1), + (Surrogate::new(2), 1), + ]; + let ranked = graph_reached_to_ranked_keys(&reached); + let keys: Vec = ranked.iter().map(|r| r.document_id).collect(); + assert_eq!( + keys, + vec![ + HybridFusionKey::for_surrogate(Surrogate::new(2)), + HybridFusionKey::for_surrogate(Surrogate::new(4)), + HybridFusionKey::for_surrogate(Surrogate::new(9)), + ] + ); + assert_eq!(ranked[0].rank, 0); + assert_eq!(ranked[2].score, 2.0); + } +} diff --git a/nodedb/src/data/executor/handlers/vector_search.rs b/nodedb/src/data/executor/handlers/vector_search.rs index a8ac6490d..60a401303 100644 --- a/nodedb/src/data/executor/handlers/vector_search.rs +++ b/nodedb/src/data/executor/handlers/vector_search.rs @@ -36,21 +36,22 @@ pub(super) fn build_search_hit( } } -/// Derive the RRF-fusion `document_id` for one vector-search leg hit, exactly -/// as the committed hybrid path does: a hit whose local HNSW id resolves to a -/// bound surrogate becomes that surrogate's hex doc_id (the shared key space -/// the text leg also uses), while a headless hit (no surrogate binding) -/// becomes the `__local_{id}` sentinel — a value that cannot fuse with any -/// real FTS doc_id, the correct behavior for a row that carries no -/// cross-engine identity. Shared by the autocommit hybrid / triple handlers -/// and the transaction-overlay splice so both paths produce identical, -/// non-spurious fusion keys (a headless local id is never misread as a global -/// surrogate). -pub(super) fn vector_leg_doc_id(collection: Option<&VectorCollection>, local_id: u32) -> String { +/// The RRF-fusion key of one vector-search leg hit. +/// +/// A hit whose local HNSW id resolves to a bound surrogate fuses under that +/// surrogate's storage key, the key space the text and graph legs use. A +/// headless hit (no surrogate binding) is `HybridFusionKey::Headless`: it +/// fuses with nothing, so a local id is never misread as a global surrogate. +/// Shared by the autocommit hybrid / triple handlers and the +/// transaction-overlay splice so both paths derive identical keys. +pub(super) fn vector_leg_key( + collection: Option<&VectorCollection>, + local_id: u32, +) -> super::hybrid_key::HybridFusionKey { collection .and_then(|c| c.get_surrogate(local_id)) - .map(crate::engine::document::store::surrogate_to_doc_id) - .unwrap_or_else(|| format!("__local_{local_id}")) + .map(super::hybrid_key::HybridFusionKey::for_surrogate) + .unwrap_or(super::hybrid_key::HybridFusionKey::Headless(local_id)) } /// Translate a `SurrogateBitmap` (keyed by global surrogate IDs) into a @@ -160,7 +161,8 @@ mod tests { use crate::engine::vector::collection::VectorCollection; use crate::engine::vector::hnsw::HnswParams; - use super::{surrogate_bitmap_to_global_ids, vector_leg_doc_id}; + use super::super::hybrid_key::HybridFusionKey; + use super::{surrogate_bitmap_to_global_ids, vector_leg_key}; /// Build a `VectorCollection` with `n` vectors of dimension 1. /// Vector `i` is `[i as f32]` and is bound to `Surrogate(i as u32 + 1)` @@ -266,32 +268,31 @@ mod tests { } } - /// The hybrid-fusion doc_id derivation must emit the `__local_` sentinel - /// for a headless vector hit (no surrogate binding) and the surrogate hex - /// for a bound hit — so a headless row's raw local id is never misread as a - /// global surrogate and can never spuriously fuse with a real FTS doc_id. - /// This is the single shared helper both the committed hybrid path and the - /// transaction-overlay splice use, so parity is guaranteed by construction. + /// The hybrid-fusion key derivation must yield `Headless` for a vector hit + /// with no surrogate binding and `Bound` for a bound hit. A headless row's + /// raw local id is never misread as a global surrogate, so it can never + /// fuse with a real FTS hit. This is the single shared helper both the + /// committed hybrid path and the transaction-overlay splice use. #[test] - fn vector_leg_doc_id_headless_emits_local_sentinel_bound_emits_surrogate() { - // No collection at all → always the sentinel. - assert_eq!(vector_leg_doc_id(None, 7), "__local_7"); + fn vector_leg_key_headless_emits_sentinel_bound_emits_surrogate() { + // No collection at all → always headless. + assert_eq!(vector_leg_key(None, 7), HybridFusionKey::Headless(7)); let coll = make_collection_with_surrogates(3); - // Local id 0 is bound to Surrogate(1): resolves to a real hex doc_id - // (never the sentinel). - let bound = vector_leg_doc_id(Some(&coll), 0); + // Local id 0 is bound to Surrogate(1): resolves to that surrogate's + // storage key (never the sentinel). + let bound = vector_leg_key(Some(&coll), 0); assert_eq!( bound, - crate::engine::document::store::surrogate_to_doc_id(Surrogate(1)), - "a bound hit must resolve to its surrogate's hex doc_id" + HybridFusionKey::for_surrogate(Surrogate(1)), + "a bound hit must resolve to its surrogate's storage key" ); - assert!( - !bound.starts_with("__local_"), - "a bound hit must never emit the headless sentinel: {bound}" + assert_eq!(bound.to_string(), "00000001"); + // A local id with no surrogate binding in this collection → headless. + assert_eq!( + vector_leg_key(Some(&coll), 999), + HybridFusionKey::Headless(999) ); - // A local id with no surrogate binding in this collection → sentinel. - assert_eq!(vector_leg_doc_id(Some(&coll), 999), "__local_999"); } #[test] diff --git a/nodedb/src/data/executor/wal_replay/crdt_doc.rs b/nodedb/src/data/executor/wal_replay/crdt_doc.rs index 649f284ff..3c0d76ca0 100644 --- a/nodedb/src/data/executor/wal_replay/crdt_doc.rs +++ b/nodedb/src/data/executor/wal_replay/crdt_doc.rs @@ -19,7 +19,7 @@ use crate::data::executor::core_loop::CoreLoop; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; use crate::wal::CrdtDocOpWalRecord; use nodedb_physical::physical_plan::CrdtOp; -use nodedb_types::Surrogate; +use nodedb_types::{RowIdentity, Surrogate}; impl CoreLoop { /// Try to decode `record` as a `RecordType::CrdtDocOp` record and, if it is @@ -78,7 +78,12 @@ impl CoreLoop { // The task carries the real intent so a handler that starts reading the // plan cannot silently degrade to a no-op (mirrors `try_replay_crdt_list`). - let (collection, document_id, response) = match &payload { + // + // The record's `document_id` is the CRDT document's client key and its + // `surrogate` a raw `u32`; both become typed here, at the decode + // boundary. `CrdtOp` and the handler params still carry the key as + // text, so the identity is rendered at each dispatch call. + let (collection, document_id, response) = match payload { CrdtDocOpWalRecord::Upsert { collection, document_id, @@ -86,12 +91,14 @@ impl CoreLoop { fields_json, partial, } => { + let document_id = RowIdentity::from_user_key(document_id); + let surrogate = Surrogate::new(surrogate); let plan = PhysicalPlan::Crdt(CrdtOp::DocUpsert { collection: nodedb_types::QualifiedCollection::from_stored(collection.clone()), - document_id: document_id.clone(), + document_id: document_id.clone().into_string(), fields_json: fields_json.clone(), - surrogate: Surrogate::new(*surrogate), - partial: *partial, + surrogate, + partial, returning: None, rls_filters: Vec::new(), }); @@ -100,11 +107,11 @@ impl CoreLoop { let response = self.execute_crdt_doc_upsert( &task, crate::data::executor::handlers::control::crdt_doc::CrdtDocUpsert { - collection, - document_id, - fields_json, - surrogate: Surrogate::new(*surrogate), - partial: *partial, + collection: &collection, + document_id: document_id.as_str(), + fields_json: &fields_json, + surrogate, + partial, returning: None, rls_filters: &[], }, @@ -116,10 +123,12 @@ impl CoreLoop { document_id, surrogate, } => { + let document_id = RowIdentity::from_user_key(document_id); + let surrogate = Surrogate::new(surrogate); let plan = PhysicalPlan::Crdt(CrdtOp::DocDelete { collection: nodedb_types::QualifiedCollection::from_stored(collection.clone()), - document_id: document_id.clone(), - surrogate: Surrogate::new(*surrogate), + document_id: document_id.clone().into_string(), + surrogate, returning: None, rls_filters: Vec::new(), }); @@ -128,9 +137,9 @@ impl CoreLoop { let response = self.execute_crdt_doc_delete( &task, crate::data::executor::handlers::control::crdt_doc::CrdtDocDelete { - collection, - document_id, - surrogate: Surrogate::new(*surrogate), + collection: &collection, + document_id: document_id.as_str(), + surrogate, returning: None, rls_filters: &[], }, diff --git a/nodedb/src/data/executor/wal_replay/crdt_list.rs b/nodedb/src/data/executor/wal_replay/crdt_list.rs index cdda95f12..ced7b1297 100644 --- a/nodedb/src/data/executor/wal_replay/crdt_list.rs +++ b/nodedb/src/data/executor/wal_replay/crdt_list.rs @@ -19,7 +19,7 @@ use crate::data::executor::core_loop::CoreLoop; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; use crate::wal::CrdtListOpWalRecord; use nodedb_physical::physical_plan::CrdtOp; -use nodedb_types::Surrogate; +use nodedb_types::{RowIdentity, Surrogate}; /// Narrow a WAL-logged `u64` list position to the `usize` the live /// `execute_crdt_list_*` handlers take. Returns `None` (with the record @@ -114,7 +114,12 @@ impl CoreLoop { // variant — no `Option` + `unwrap_or(0)` fallback. A record // whose position doesn't fit `usize` is refused (`Some(0)`, logged), // never silently replayed at position 0. - let (collection, document_id, list_path, response) = match &payload { + // + // The record's `document_id` is the CRDT document's client key and + // becomes a `RowIdentity` here, at the decode boundary. `CrdtOp` and + // the list handlers still take the key as text, so the identity is + // rendered at each dispatch call. + let (collection, document_id, list_path, response) = match payload { CrdtListOpWalRecord::Insert { collection, document_id, @@ -122,12 +127,13 @@ impl CoreLoop { index, fields_json, } => { - let Some(index) = wal_list_index(core_id, record_lsn, "index", *index) else { + let Some(index) = wal_list_index(core_id, record_lsn, "index", index) else { return Some(0); }; + let document_id = RowIdentity::from_user_key(document_id); let plan = PhysicalPlan::Crdt(CrdtOp::ListInsert { collection: nodedb_types::QualifiedCollection::from_stored(collection.clone()), - document_id: document_id.clone(), + document_id: document_id.clone().into_string(), list_path: list_path.clone(), index, fields_json: fields_json.clone(), @@ -137,11 +143,11 @@ impl CoreLoop { Self::replay_task(tid, database_id, vshard, plan, Some(Lsn::new(record_lsn))); let response = self.execute_crdt_list_insert( &task, - collection, - document_id, - list_path, + &collection, + document_id.as_str(), + &list_path, index, - fields_json, + &fields_json, ); (collection, document_id, list_path, response) } @@ -151,20 +157,26 @@ impl CoreLoop { list_path, index, } => { - let Some(index) = wal_list_index(core_id, record_lsn, "index", *index) else { + let Some(index) = wal_list_index(core_id, record_lsn, "index", index) else { return Some(0); }; + let document_id = RowIdentity::from_user_key(document_id); let plan = PhysicalPlan::Crdt(CrdtOp::ListDelete { collection: nodedb_types::QualifiedCollection::from_stored(collection.clone()), - document_id: document_id.clone(), + document_id: document_id.clone().into_string(), list_path: list_path.clone(), index, surrogate: Surrogate::ZERO, }); let task = Self::replay_task(tid, database_id, vshard, plan, Some(Lsn::new(record_lsn))); - let response = - self.execute_crdt_list_delete(&task, collection, document_id, list_path, index); + let response = self.execute_crdt_list_delete( + &task, + &collection, + document_id.as_str(), + &list_path, + index, + ); (collection, document_id, list_path, response) } CrdtListOpWalRecord::Move { @@ -175,17 +187,18 @@ impl CoreLoop { to_index, } => { let Some(from_index) = - wal_list_index(core_id, record_lsn, "from_index", *from_index) + wal_list_index(core_id, record_lsn, "from_index", from_index) else { return Some(0); }; - let Some(to_index) = wal_list_index(core_id, record_lsn, "to_index", *to_index) + let Some(to_index) = wal_list_index(core_id, record_lsn, "to_index", to_index) else { return Some(0); }; + let document_id = RowIdentity::from_user_key(document_id); let plan = PhysicalPlan::Crdt(CrdtOp::ListMove { collection: nodedb_types::QualifiedCollection::from_stored(collection.clone()), - document_id: document_id.clone(), + document_id: document_id.clone().into_string(), list_path: list_path.clone(), from_index, to_index, @@ -195,9 +208,9 @@ impl CoreLoop { Self::replay_task(tid, database_id, vshard, plan, Some(Lsn::new(record_lsn))); let response = self.execute_crdt_list_move( &task, - collection, - document_id, - list_path, + &collection, + document_id.as_str(), + &list_path, from_index, to_index, ); diff --git a/nodedb/src/data/executor/wal_replay_document_vector.rs b/nodedb/src/data/executor/wal_replay_document_vector.rs index 46e82f16f..bb39a2235 100644 --- a/nodedb/src/data/executor/wal_replay_document_vector.rs +++ b/nodedb/src/data/executor/wal_replay_document_vector.rs @@ -86,6 +86,8 @@ impl CoreLoop { // earlier `Put` rebuilt. KV / other-engine deletes decode to a // different shape and are skipped by the strict tuple decode. if is_delete { + // Replay keys the row by its surrogate; the record's text + // `document_id` is the client key and stays unread. let Ok((collection, _document_id, _prov, surrogate_u32)) = zerompk::from_msgpack::<(String, String, Option, u32)>(payload) else { @@ -265,6 +267,8 @@ fn is_kv_put_record(payload: &[u8]) -> bool { /// `String` where the document value's `Vec` is) fail both decodes and /// return `None`. fn decode_document_put(payload: &[u8]) -> Option<(String, Vec, Surrogate)> { + // Replay keys the row by its surrogate; the record's text `document_id` + // is the client key and stays unread. if let Ok((collection, _document_id, value, _prov, surrogate_u32)) = zerompk::from_msgpack::<(String, String, Vec, Option, u32)>(payload) { diff --git a/nodedb/src/data/executor/wal_replay_redo_document.rs b/nodedb/src/data/executor/wal_replay_redo_document.rs index fde569635..64caa0de1 100644 --- a/nodedb/src/data/executor/wal_replay_redo_document.rs +++ b/nodedb/src/data/executor/wal_replay_redo_document.rs @@ -133,9 +133,11 @@ impl CoreLoop { i64, ); type PlainPut = (String, String, Vec, Option, u32); + // Replay keys the row by its surrogate; the record's text + // `document_id` is the client key and stays unread. let decoded = zerompk::from_msgpack::(&record.payload) .map( - |(collection, _doc_id, value, _prov, surrogate, sys, vf, vu)| { + |(collection, _document_id, value, _prov, surrogate, sys, vf, vu)| { ( collection, value, @@ -150,7 +152,7 @@ impl CoreLoop { ) .or_else(|_| { zerompk::from_msgpack::(&record.payload).map( - |(collection, _doc_id, value, _prov, surrogate)| { + |(collection, _document_id, value, _prov, surrogate)| { (collection, value, surrogate, None) }, ) @@ -190,6 +192,8 @@ impl CoreLoop { ); } } else { + // Replay keys the row by its surrogate; the record's text + // `document_id` is the client key and stays unread. let Ok((collection, _document_id, _prov, surrogate_u32)) = zerompk::from_msgpack::<(String, String, Option, u32)>( &record.payload, diff --git a/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_cdc.rs b/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_cdc.rs index 112b50002..ab4ba729d 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_cdc.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_cdc.rs @@ -8,6 +8,7 @@ use super::helpers::{TENANT_A, TENANT_B}; use nodedb::control::change_stream::{ChangeEvent, ChangeOperation, ChangeStream, ReplayStart}; use nodedb::types::{Lsn, TenantId}; +use nodedb_types::RowIdentity; #[test] fn cdc_stream_isolated_between_tenants() { @@ -19,7 +20,7 @@ fn cdc_stream_isolated_between_tenants() { // Publish a change event for Tenant A on "orders". stream.publish(ChangeEvent { collection: "orders".into(), - document_id: "order_1".into(), + document_id: RowIdentity::from_user_key("order_1"), operation: ChangeOperation::Insert, timestamp_ms: 1000, tenant_id: TenantId::new(TENANT_A), @@ -30,7 +31,7 @@ fn cdc_stream_isolated_between_tenants() { // Publish a change event for Tenant B on "orders". stream.publish(ChangeEvent { collection: "orders".into(), - document_id: "order_2".into(), + document_id: RowIdentity::from_user_key("order_2"), operation: ChangeOperation::Insert, timestamp_ms: 2000, tenant_id: TenantId::new(TENANT_B), @@ -60,10 +61,10 @@ fn cdc_stream_isolated_between_tenants() { assert_eq!(a_events.len(), 1); assert_eq!(a_events[0].tenant_id, TenantId::new(TENANT_A)); - assert_eq!(a_events[0].document_id, "order_1"); + assert_eq!(a_events[0].document_id.as_str(), "order_1"); assert_eq!(b_events.len(), 1); assert_eq!(b_events[0].tenant_id, TenantId::new(TENANT_B)); - assert_eq!(b_events[0].document_id, "order_2"); + assert_eq!(b_events[0].document_id.as_str(), "order_2"); } #[test] @@ -74,7 +75,7 @@ fn cdc_opaque_cursor_keeps_same_millisecond_events_pageable() { for (lsn, document_id) in [(1, "first"), (2, "second"), (3, "third")] { stream.publish(ChangeEvent { collection: "orders".into(), - document_id: document_id.into(), + document_id: RowIdentity::from_user_key(document_id), operation: ChangeOperation::Insert, timestamp_ms: 1_000, tenant_id, @@ -87,7 +88,7 @@ fn cdc_opaque_cursor_keeps_same_millisecond_events_pageable() { .query_changes(tenant_id, Some("orders"), ReplayStart::Timestamp(0), 1) .expect("timestamp replay cannot expire"); assert_eq!(first_page.events.len(), 1); - assert_eq!(first_page.events[0].document_id, "first"); + assert_eq!(first_page.events[0].document_id.as_str(), "first"); let second_page = stream .query_changes( @@ -98,7 +99,7 @@ fn cdc_opaque_cursor_keeps_same_millisecond_events_pageable() { ) .expect("fresh cursor must resume"); assert_eq!(second_page.events.len(), 1); - assert_eq!(second_page.events[0].document_id, "second"); + assert_eq!(second_page.events[0].document_id.as_str(), "second"); } #[test] @@ -108,7 +109,7 @@ fn cdc_different_collections_isolated() { // Same tenant, different collections. stream.publish(ChangeEvent { collection: "orders".into(), - document_id: "o1".into(), + document_id: RowIdentity::from_user_key("o1"), operation: ChangeOperation::Insert, timestamp_ms: 1000, tenant_id: TenantId::new(TENANT_A), @@ -117,7 +118,7 @@ fn cdc_different_collections_isolated() { }); stream.publish(ChangeEvent { collection: "users".into(), - document_id: "u1".into(), + document_id: RowIdentity::from_user_key("u1"), operation: ChangeOperation::Insert, timestamp_ms: 2000, tenant_id: TenantId::new(TENANT_A), diff --git a/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_cdc_negative.rs b/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_cdc_negative.rs index 6239ada32..1cccfbde0 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_cdc_negative.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_cdc_negative.rs @@ -11,6 +11,7 @@ use std::time::Duration; use super::helpers::{TENANT_A, TENANT_B}; use nodedb::control::change_stream::{ChangeEvent, ChangeOperation, ChangeStream, ReplayStart}; use nodedb::types::{Lsn, TenantId}; +use nodedb_types::RowIdentity; /// A subscription scoped to Tenant B must never deliver Tenant A's events, /// even when both tenants write to the same collection. @@ -25,7 +26,7 @@ async fn cdc_tenant_b_subscription_rejects_tenant_a_events() { for i in 0..5u64 { stream.publish(ChangeEvent { collection: "orders".into(), - document_id: format!("a_order_{i}"), + document_id: RowIdentity::from_user_key(format!("a_order_{i}")), operation: ChangeOperation::Insert, timestamp_ms: (i + 1) * 1000, tenant_id: TenantId::new(TENANT_A), @@ -37,7 +38,7 @@ async fn cdc_tenant_b_subscription_rejects_tenant_a_events() { // Publish one event for Tenant B. stream.publish(ChangeEvent { collection: "orders".into(), - document_id: "b_order_1".into(), + document_id: RowIdentity::from_user_key("b_order_1"), operation: ChangeOperation::Insert, timestamp_ms: 10_000, tenant_id: TenantId::new(TENANT_B), @@ -57,7 +58,8 @@ async fn cdc_tenant_b_subscription_rejects_tenant_a_events() { "First event delivered to Tenant B subscription must belong to Tenant B" ); assert_eq!( - received.document_id, "b_order_1", + received.document_id.as_str(), + "b_order_1", "Received wrong document_id: expected b_order_1, got {}", received.document_id ); @@ -81,7 +83,7 @@ async fn cdc_unfiltered_subscription_receives_all_tenants() { stream.publish(ChangeEvent { collection: "events".into(), - document_id: "e_a".into(), + document_id: RowIdentity::from_user_key("e_a"), operation: ChangeOperation::Insert, timestamp_ms: 1000, tenant_id: TenantId::new(TENANT_A), @@ -90,7 +92,7 @@ async fn cdc_unfiltered_subscription_receives_all_tenants() { }); stream.publish(ChangeEvent { collection: "events".into(), - document_id: "e_b".into(), + document_id: RowIdentity::from_user_key("e_b"), operation: ChangeOperation::Insert, timestamp_ms: 2000, tenant_id: TenantId::new(TENANT_B), @@ -129,7 +131,7 @@ fn cdc_query_changes_scopes_the_requested_tenant_before_limit() { for (lsn, document_id) in [(Lsn::new(1), "l_b_1"), (Lsn::new(2), "l_b_2")] { stream.publish(ChangeEvent { collection: "logs".into(), - document_id: document_id.into(), + document_id: RowIdentity::from_user_key(document_id), operation: ChangeOperation::Insert, timestamp_ms: 1_000, tenant_id: TenantId::new(TENANT_B), @@ -139,7 +141,7 @@ fn cdc_query_changes_scopes_the_requested_tenant_before_limit() { } stream.publish(ChangeEvent { collection: "logs".into(), - document_id: "l_a".into(), + document_id: RowIdentity::from_user_key("l_a"), operation: ChangeOperation::Insert, timestamp_ms: 1_000, tenant_id: TenantId::new(TENANT_A), @@ -163,5 +165,5 @@ fn cdc_query_changes_scopes_the_requested_tenant_before_limit() { "tenant filtering must precede limit so Tenant A receives its matching event" ); assert_eq!(a_events[0].tenant_id, TenantId::new(TENANT_A)); - assert_eq!(a_events[0].document_id, "l_a"); + assert_eq!(a_events[0].document_id.as_str(), "l_a"); } diff --git a/nodedb/tests/inproc/cases/http_cdc.rs b/nodedb/tests/inproc/cases/http_cdc.rs index 9a0d441c8..72056983f 100644 --- a/nodedb/tests/inproc/cases/http_cdc.rs +++ b/nodedb/tests/inproc/cases/http_cdc.rs @@ -24,6 +24,7 @@ use nodedb::control::security::identity::Role; use nodedb::control::state::SharedState; use nodedb::types::{DatabaseId, Lsn, TenantId}; use nodedb::wal::WalManager; +use nodedb_types::RowIdentity; struct TestServer { local_addr: std::net::SocketAddr, @@ -262,7 +263,7 @@ async fn cdc_poll_isolates_events_by_selected_database_and_prefers_header() { lsn, tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: document_id.into(), + document_id: RowIdentity::from_user_key(document_id), operation: ChangeOperation::Insert, timestamp_ms: 9_000, after: None, @@ -338,7 +339,7 @@ async fn cdc_poll_excludes_matching_events_from_other_tenants() { lsn: Lsn::new(101), tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: "tenant-one-order".into(), + document_id: RowIdentity::from_user_key("tenant-one-order"), operation: ChangeOperation::Insert, timestamp_ms, after: None, @@ -347,7 +348,7 @@ async fn cdc_poll_excludes_matching_events_from_other_tenants() { lsn: Lsn::new(102), tenant_id: TenantId::new(2), collection: "orders".into(), - document_id: "tenant-two-order".into(), + document_id: RowIdentity::from_user_key("tenant-two-order"), operation: ChangeOperation::Insert, timestamp_ms, after: None, @@ -396,7 +397,7 @@ async fn cdc_poll_paginates_same_timestamp_with_opaque_cursor() { lsn, tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: document_id.into(), + document_id: RowIdentity::from_user_key(document_id), operation: ChangeOperation::Insert, timestamp_ms, after: None, @@ -476,7 +477,7 @@ async fn cdc_poll_opaque_cursor_preserves_publish_order_across_out_of_order_lsns lsn: Lsn::new(602), tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: "published-first-lsn-602".into(), + document_id: RowIdentity::from_user_key("published-first-lsn-602"), operation: ChangeOperation::Insert, timestamp_ms, after: None, @@ -504,7 +505,7 @@ async fn cdc_poll_opaque_cursor_preserves_publish_order_across_out_of_order_lsns lsn: Lsn::new(601), tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: "delayed-lsn-601".into(), + document_id: RowIdentity::from_user_key("delayed-lsn-601"), operation: ChangeOperation::Insert, timestamp_ms, after: None, @@ -559,7 +560,7 @@ async fn cdc_poll_opaque_cursor_paginates_events_sharing_one_lsn() { lsn: Lsn::new(603), tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: document_id.into(), + document_id: RowIdentity::from_user_key(document_id), operation: ChangeOperation::Insert, timestamp_ms, after: None, @@ -661,7 +662,7 @@ async fn cdc_poll_limit_zero_clamps_to_a_nonempty_page() { lsn: Lsn::new(301), tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: "zero-limit-order".into(), + document_id: RowIdentity::from_user_key("zero-limit-order"), operation: ChangeOperation::Insert, timestamp_ms: 3_000, after: None, @@ -704,7 +705,7 @@ async fn cdc_sse_last_event_id_replays_only_events_after_the_cursor() { lsn, tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: document_id.into(), + document_id: RowIdentity::from_user_key(document_id), operation: ChangeOperation::Insert, timestamp_ms, after: None, @@ -770,7 +771,7 @@ async fn cdc_poll_rejects_custom_role_without_collection_grant() { lsn: Lsn::new(103), tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: "ungranted-poll-order".into(), + document_id: RowIdentity::from_user_key("ungranted-poll-order"), operation: ChangeOperation::Insert, timestamp_ms: 1_000, after: None, @@ -801,7 +802,7 @@ async fn cdc_sse_rejects_custom_role_without_collection_grant_before_streaming() lsn: Lsn::new(104), tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: "ungranted-sse-order".into(), + document_id: RowIdentity::from_user_key("ungranted-sse-order"), operation: ChangeOperation::Insert, timestamp_ms: 1_000, after: None, diff --git a/nodedb/tests/inproc/cases/http_ws.rs b/nodedb/tests/inproc/cases/http_ws.rs index cfaf6bf90..312b9711e 100644 --- a/nodedb/tests/inproc/cases/http_ws.rs +++ b/nodedb/tests/inproc/cases/http_ws.rs @@ -23,6 +23,7 @@ use nodedb::control::security::catalog::DatabaseDescriptor; use nodedb::control::state::SharedState; use nodedb::types::{DatabaseId, Lsn, TenantId}; use nodedb::wal::WalManager; +use nodedb_types::RowIdentity; use tokio_tungstenite::tungstenite::client::IntoClientRequest; use tokio_tungstenite::tungstenite::{Message, http}; @@ -161,7 +162,7 @@ fn publish_change(srv: &TestServer, lsn: u64, document_id: &str) { lsn: Lsn::new(lsn), tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: document_id.into(), + document_id: RowIdentity::from_user_key(document_id), operation: ChangeOperation::Insert, timestamp_ms: 1_000, after: None, @@ -378,7 +379,7 @@ async fn ws_auth_replay_isolates_events_by_selected_database() { lsn, tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: document_id.into(), + document_id: RowIdentity::from_user_key(document_id), operation: ChangeOperation::Insert, timestamp_ms: 1_000, after: None, @@ -446,7 +447,9 @@ async fn ws_live_lag_never_silently_skips_events() { lsn: Lsn::new(10_000 + sequence), tenant_id: TenantId::new(1), collection: "orders".into(), - document_id: format!("lagged-order-{sequence}-{document_suffix}"), + document_id: RowIdentity::from_user_key(format!( + "lagged-order-{sequence}-{document_suffix}" + )), operation: ChangeOperation::Insert, timestamp_ms: 10_000, after: None, diff --git a/nodedb/tests/inproc/cases/http_ws_authorization.rs b/nodedb/tests/inproc/cases/http_ws_authorization.rs index 88460017f..44e98ffa4 100644 --- a/nodedb/tests/inproc/cases/http_ws_authorization.rs +++ b/nodedb/tests/inproc/cases/http_ws_authorization.rs @@ -13,6 +13,7 @@ use nodedb::control::security::identity::Role; use nodedb::control::state::SharedState; use nodedb::types::{DatabaseId, Lsn, TenantId}; use nodedb_test_support::pgwire_harness::TestServer; +use nodedb_types::RowIdentity; use tokio_tungstenite::tungstenite::client::IntoClientRequest; use tokio_tungstenite::tungstenite::{Message, http}; @@ -217,7 +218,7 @@ async fn ws_resume_session_id_is_not_shared_across_authenticated_identities() { lsn, tenant_id, collection: "orders".into(), - document_id: document_id.into(), + document_id: RowIdentity::from_user_key(document_id), operation: ChangeOperation::Insert, timestamp_ms: 1_000, after: None, diff --git a/nodedb/tests/inproc/cases/pgwire_cdc_change_events.rs b/nodedb/tests/inproc/cases/pgwire_cdc_change_events.rs index fc40ff6b5..605d3d702 100644 --- a/nodedb/tests/inproc/cases/pgwire_cdc_change_events.rs +++ b/nodedb/tests/inproc/cases/pgwire_cdc_change_events.rs @@ -89,7 +89,7 @@ async fn pgwire_sql_dml_publishes_change_events() { "{what}: change event carries the wrong operation kind" ); assert!( - !event.document_id.is_empty(), + !event.document_id.as_str().is_empty(), "{what}: change event carries no document identity" ); } From a347fdbc4891967c4cc512480de7bbd8bd961789 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 12 Sep 2026 17:38:17 +0800 Subject: [PATCH 15/17] refactor(executor): key vector/hybrid search hits by storage key, not raw surrogate HybridFusionKey now carries its own MessagePack wire encoding (the hex storage key, or the __local_ headless sentinel) and VectorSearchHit's id field is typed as HybridFusionKey instead of a bare u32 surrogate. Dense, sparse, and multi-vector search, the transaction overlay merge, and the response-codec flatteners all read and filter through this key so a headless (surrogate-less) hit is represented distinctly rather than aliased onto surrogate 0. The Control Plane's surrogate-or-headless decode is consolidated into a shared parse_surrogate_hex helper (response_translate/hit_key.rs) used by the vector, text-hybrid, and MERGE/UPDATE...FROM exchange translators, replacing three separate ad hoc parses of the sentinel prefix. Spatial R-tree entry ids are now a dedicated SpatialEntryId newtype (hashed from either a document's storage key or a columnar row's user id) instead of raw fnv1a_hash calls at each call site, so the put and remove paths cannot hash a row's identity differently. vector_doc_map, VectorIndexDelta, and the vector undo-log entries are rekeyed from String doc ids to typed StorageKey / Option, and ResolvedUpdateRow / ResolvedUpdateRowWire drop the Option in favor of carrying the resolved StorageKey (or a surrogate that is now always present, since every matched row is storage-keyed) directly. --- nodedb-types/src/lib.rs | 4 +- nodedb-types/src/row_identity.rs | 5 ++ .../merge_orchestrator/expand_staged_merge.rs | 6 ++- .../merge_orchestrator/resolve_arms.rs | 12 +++-- .../resolve/exchange/post_process_arm.rs | 22 ++------ .../server/response_translate/hit_key.rs | 20 +++++++ .../control/server/response_translate/mod.rs | 1 + .../server/response_translate/text_hybrid.rs | 17 +----- .../server/response_translate/vector.rs | 46 ++++++++++------ .../expand_staged_update_from_join.rs | 24 ++++----- nodedb/src/data/executor/core_loop/state.rs | 18 +++++-- .../core_loop/vector_index_rebuild.rs | 1 - .../executor/dispatch/array/surrogate_scan.rs | 8 ++- .../data/executor/handlers/bulk_dml/update.rs | 10 +--- .../handlers/bulk_dml/update_persist.rs | 2 +- .../handlers/bulk_dml/update_project.rs | 10 ++-- .../handlers/columnar_write/geometry_index.rs | 6 ++- .../handlers/document/resolve/bulk.rs | 8 +-- .../src/data/executor/handlers/hybrid_key.rs | 50 ++++++++++++++++-- .../data/executor/handlers/hybrid_overlay.rs | 25 +-------- .../merge_orchestrated/apply/update_rows.rs | 2 +- .../merge_orchestrated/apply_support.rs | 2 +- .../handlers/merge_orchestrated/resolve.rs | 7 +-- .../executor/handlers/point/apply_delete.rs | 20 +++---- .../executor/handlers/point/apply_put/core.rs | 1 - .../handlers/point/apply_put/index.rs | 51 ++++++++++++++---- .../executor/handlers/point/apply_put/mod.rs | 1 + .../handlers/point/apply_put/vector/put.rs | 27 +++------- .../handlers/point/apply_put/vector/remove.rs | 20 ++++--- .../handlers/point/apply_put/vector/types.rs | 9 ++-- .../executor/handlers/point/update/exec.rs | 3 -- .../executor/handlers/point/update/persist.rs | 8 +-- .../executor/handlers/point/update_reindex.rs | 2 - .../handlers/point/update_reindex_vector.rs | 1 - .../transaction/overlay/vector_merge.rs | 16 +++--- .../transaction/sub_plan_doc/delete.rs | 4 +- .../handlers/transaction/sub_plan_doc/put.rs | 4 +- .../handlers/transaction/sub_plan_write.rs | 12 ++--- .../handlers/transaction/undo/apply.rs | 22 +++++--- .../handlers/transaction/undo/entry.rs | 12 +++-- .../executor/handlers/update_from_join.rs | 4 +- .../handlers/update_from_join_collect.rs | 3 +- .../handlers/update_from_join_types.rs | 14 +++-- .../handlers/update_from_join_write.rs | 30 ++--------- .../data/executor/handlers/vector_multi.rs | 2 +- .../data/executor/handlers/vector_search.rs | 15 +++--- .../executor/handlers/vector_search_exec.rs | 17 ++++-- .../data/executor/handlers/vector_sparse.rs | 20 +++---- .../src/data/executor/response_codec/hits.rs | 8 +-- .../src/data/executor/response_codec/raw.rs | 52 +++++++++++++------ .../executor/wal_replay_document_vector.rs | 1 - nodedb/src/query/resolved_update_row.rs | 5 +- .../cases/cross_engine_bitmap_currency.rs | 5 +- .../cross_engine_three_way_fts_vector_doc.rs | 5 +- .../inproc/cases/surrogate_round_trip.rs | 3 +- 55 files changed, 383 insertions(+), 320 deletions(-) create mode 100644 nodedb/src/control/server/response_translate/hit_key.rs diff --git a/nodedb-types/src/lib.rs b/nodedb-types/src/lib.rs index 8e71fca06..a7b0b15b4 100644 --- a/nodedb-types/src/lib.rs +++ b/nodedb-types/src/lib.rs @@ -120,8 +120,8 @@ pub use quota::{ pub use result::{QueryResult, SearchResult, SubGraph}; pub use rls_write_check::{RlsWriteCheck, WriteGateDecision}; pub use row_identity::{ - DEFAULT_IDENTITY_COLUMN, RowIdentity, StorageKey, doc_id_to_surrogate, extract_pk_value, - identity_of, surrogate_to_doc_id, value_to_pk_string, + DEFAULT_IDENTITY_COLUMN, HEADLESS_SENTINEL_PREFIX, RowIdentity, StorageKey, + doc_id_to_surrogate, extract_pk_value, identity_of, surrogate_to_doc_id, value_to_pk_string, }; pub use sparse_vector::{SparseVector, SparseVectorError}; pub use sql_quote::{quote_ident, quote_literal}; diff --git a/nodedb-types/src/row_identity.rs b/nodedb-types/src/row_identity.rs index 5e24c23d1..2ec4dbcc2 100644 --- a/nodedb-types/src/row_identity.rs +++ b/nodedb-types/src/row_identity.rs @@ -34,6 +34,11 @@ use crate::{Surrogate, Value}; /// `PRIMARY KEY`. INSERT and every stored-row identity derivation use it. pub const DEFAULT_IDENTITY_COLUMN: &str = "id"; +/// Prefix of a rendered key for a row with no surrogate binding, used by both +/// planes: the Data Plane's `HybridFusionKey::Headless` and the Control +/// Plane's hybrid response decoders that must recognize the same sentinel. +pub const HEADLESS_SENTINEL_PREFIX: &str = "__local_"; + /// The redb key a document row is stored under. /// /// Fixed-width lowercase hex, so lexicographic order matches surrogate order diff --git a/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs b/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs index 925ae4faf..46b0502f1 100644 --- a/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs +++ b/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs @@ -232,8 +232,10 @@ fn emit_arms( )); } - for (doc_id, surrogate_u32, body, _old_body) in arms.updates { - let surrogate = require_surrogate(surrogate_u32, &doc_id, "MERGE")?; + for (_doc_id, surrogate_u32, body, _old_body) in arms.updates { + // The wire surrogate is never absent: every matched MERGE UPDATE row + // is a storage-keyed row. + let surrogate = nodedb_types::Surrogate::new(surrogate_u32); let document_id = derive_document_id(target_pk, &body, surrogate); let pk_bytes = document_id.clone().into_bytes(); out.push(point_task( diff --git a/nodedb/src/control/merge_orchestrator/resolve_arms.rs b/nodedb/src/control/merge_orchestrator/resolve_arms.rs index d14a8e228..9e12d9754 100644 --- a/nodedb/src/control/merge_orchestrator/resolve_arms.rs +++ b/nodedb/src/control/merge_orchestrator/resolve_arms.rs @@ -10,10 +10,12 @@ /// The three resolved arms of a MERGE, decoded from the Data-Plane RESOLVE /// pass. `updates` / `deletes` carry the EXISTING target row's storage key -/// (`doc_id`), its registered `surrogate` (`None` only for a legacy -/// non-surrogate-keyed row — unreachable for any surrogate-keyed collection), -/// and the arm's resolved body (post-image for updates, the deleted row for -/// deletes so its PK can be extracted). `inserts` carry `(join_key, body)`. +/// (`doc_id`) and its registered `surrogate` — always present on `updates` +/// (every matched row is storage-keyed); `None` on `deletes` only for a +/// legacy non-surrogate-keyed row, unreachable for any surrogate-keyed +/// collection — and the arm's resolved body (post-image for updates, the +/// deleted row for deletes so its PK can be extracted). `inserts` carry +/// `(join_key, body)`. /// /// An UPDATE arm additionally carries the target row's PRE-image as a fourth /// element. A materialized-sum delta is the DIFFERENCE between the two images @@ -34,7 +36,7 @@ pub(crate) fn decode_resolve(payload: &[u8]) -> crate::Result return Ok(ResolvedMergeArms::default()); } type Wire = ( - Vec<(String, Option, Vec, Vec)>, + Vec, Vec<(String, Option, Vec)>, Vec<(String, Vec)>, ); diff --git a/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs b/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs index 444ffb90b..9e4920724 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs @@ -12,6 +12,7 @@ use crate::control::server::exchange::gather::{ GatherOutcome, finalize_aggregate, gather_all_vshards, }; use crate::control::server::exchange::resolve::capture::DistributedReadCapture; +use crate::control::server::response_translate::hit_key::parse_surrogate_hex; use crate::control::server::response_translate::vector::resolve_surrogate_pk; use crate::control::state::SharedState; use crate::data::executor::response_codec::{ @@ -210,23 +211,10 @@ pub(super) async fn resolve_post_process( nodedb_types::Surrogate::new(surrogate), ) }), - HitShape::Hybrid => { - flatten_hybrid_hits_to_relational_rows(&merged, |hex| { - // `__local_` is the headless-vector-leg sentinel; it - // is not a real surrogate and must not be parsed as hex. - if hex.starts_with("__local_") { - return None; - } - let surrogate = u32::from_str_radix(hex, 16).ok()?; - resolve_surrogate_pk( - state, - database_id, - tenant_id, - &coll, - nodedb_types::Surrogate::new(surrogate), - ) - }) - } + HitShape::Hybrid => flatten_hybrid_hits_to_relational_rows(&merged, |key| { + let surrogate = parse_surrogate_hex(key)?; + resolve_surrogate_pk(state, database_id, tenant_id, &coll, surrogate) + }), HitShape::None => flatten_to_relational_rows(&merged), }; Ok(Resolved::Plan(Box::new(PhysicalPlan::Query( diff --git a/nodedb/src/control/server/response_translate/hit_key.rs b/nodedb/src/control/server/response_translate/hit_key.rs new file mode 100644 index 000000000..68ccc503c --- /dev/null +++ b/nodedb/src/control/server/response_translate/hit_key.rs @@ -0,0 +1,20 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Shared decode for the surrogate-or-headless key the vector, full-text, +//! and hybrid response translators all read off a Data-Plane hit. +//! +//! The Data Plane's `HybridFusionKey` (`nodedb/src/data/executor/handlers/ +//! hybrid_key.rs`) renders a bound row as its hex storage key and a +//! headless vector hit (no surrogate binding) as the `__local_` +//! sentinel. [`parse_surrogate_hex`] is the one place that string gets +//! reinterpreted back into a [`Surrogate`] on the Control Plane. + +use nodedb_types::{StorageKey, Surrogate}; + +/// Decode a hit's identity-key text into a surrogate. `None` means the row +/// carries no surrogate binding — the `__local_` headless sentinel, or a +/// value that is not a storage key — and callers leave the row's identifier +/// untouched rather than fabricate one. +pub(crate) fn parse_surrogate_hex(candidate: &str) -> Option { + StorageKey::parse(candidate).map(|key| key.surrogate()) +} diff --git a/nodedb/src/control/server/response_translate/mod.rs b/nodedb/src/control/server/response_translate/mod.rs index 42a721cf9..df235a662 100644 --- a/nodedb/src/control/server/response_translate/mod.rs +++ b/nodedb/src/control/server/response_translate/mod.rs @@ -4,6 +4,7 @@ //! payloads with catalog-resolved fields (e.g. surrogate → user PK). pub mod dispatch; +pub mod hit_key; pub mod text_hybrid; pub mod vector; diff --git a/nodedb/src/control/server/response_translate/text_hybrid.rs b/nodedb/src/control/server/response_translate/text_hybrid.rs index e736b948c..6812ed02b 100644 --- a/nodedb/src/control/server/response_translate/text_hybrid.rs +++ b/nodedb/src/control/server/response_translate/text_hybrid.rs @@ -21,28 +21,15 @@ //! the same [`super::vector::resolve_surrogate_pk`] catalog call the vector //! translator uses. -use nodedb_types::{DatabaseId, Surrogate, TenantId}; +use nodedb_types::{DatabaseId, TenantId}; use serde_json::Value as JsonValue; use crate::control::state::SharedState; use crate::data::executor::response_codec::decode_payload_to_json; +use super::hit_key::parse_surrogate_hex; use super::vector::resolve_surrogate_pk; -/// A `__local_` doc_id is the vector leg's sentinel for a hit with no -/// surrogate binding (see `vector_leg_doc_id`) — it never corresponds to a -/// real surrogate and must not be parsed as hex. -const HEADLESS_SENTINEL_PREFIX: &str = "__local_"; - -/// Decode a `doc_id` candidate string into a surrogate, rejecting the -/// headless sentinel and any non-hex value. -fn parse_surrogate_hex(candidate: &str) -> Option { - if candidate.starts_with(HEADLESS_SENTINEL_PREFIX) { - return None; - } - u32::from_str_radix(candidate, 16).ok().map(Surrogate::new) -} - /// Decode the DP-side JSON/msgpack array of `TextOp::Search` / /// `PhraseSearch`-shaped hits (`{id: , data: {...}}`), and for /// any row whose `data` object has no `id` field of its own, resolve the diff --git a/nodedb/src/control/server/response_translate/vector.rs b/nodedb/src/control/server/response_translate/vector.rs index e85dd7038..10eac7e97 100644 --- a/nodedb/src/control/server/response_translate/vector.rs +++ b/nodedb/src/control/server/response_translate/vector.rs @@ -2,11 +2,13 @@ //! Surrogate → user-PK translation for vector search responses. //! -//! The Data Plane emits each hit's `id` as the bound `Surrogate.as_u32()` -//! (or the local node id for headless rows) and leaves `doc_id` as -//! `None`. The Control Plane runs this translator at the response -//! boundary so pgwire / HTTP / native clients still see human-readable -//! document identifiers without the engine ever consulting the catalog. +//! The Data Plane emits each hit's `id` as its `HybridFusionKey` wire +//! string: the hex storage key of a bound row's surrogate, or the +//! `__local_` sentinel for a headless row (no surrogate binding). It +//! leaves `doc_id` as `None`. The Control Plane runs this translator at the +//! response boundary so pgwire / HTTP / native clients still see +//! human-readable document identifiers without the engine ever consulting +//! the catalog. //! //! Behaviour: //! - non-msgpack payloads (already JSON, empty, or non-array) round- @@ -23,13 +25,16 @@ use nodedb_types::Surrogate; use nodedb_types::TenantId; use serde::{Deserialize, Serialize}; +use super::hit_key::parse_surrogate_hex; use crate::bridge::scan_filter::ScanFilter; use crate::control::state::SharedState; #[derive(Serialize, Deserialize, zerompk::ToMessagePack, zerompk::FromMessagePack)] #[msgpack(map)] struct Hit { - id: u32, + /// The `HybridFusionKey` wire string: a hex storage key, or the + /// `__local_` sentinel for a headless (surrogate-less) row. + id: String, distance: f32, #[serde(skip_serializing_if = "Option::is_none")] doc_id: Option, @@ -129,13 +134,13 @@ pub fn translate_vector_search_payload( if hit.doc_id.is_some() { continue; } - if let Some(pk) = resolve_surrogate_pk( - state, - database_id, - tenant_id, - collection, - Surrogate::new(hit.id), - ) { + // A headless hit (no surrogate binding) has no PK to resolve; leave + // it untouched exactly as `text_hybrid.rs` does for the same sentinel. + let Some(surrogate) = parse_surrogate_hex(&hit.id) else { + continue; + }; + if let Some(pk) = resolve_surrogate_pk(state, database_id, tenant_id, collection, surrogate) + { hit.doc_id = Some(pk); } } @@ -170,17 +175,24 @@ pub fn translate_vector_search_payload( obj.insert(k, v); } } - // Fall back to catalog-resolved doc_id or raw surrogate when - // the body didn't provide an "id" field (e.g. skip_payload_fetch). + // A storage key never reaches a client: a bound hit's client + // identity is its surrogate's decimal string; a headless hit has + // no surrogate, so the sentinel is the only identity it carries. + let identity = match parse_surrogate_hex(&h.id) { + Some(surrogate) => surrogate.as_u32().to_string(), + None => h.id.clone(), + }; + // Fall back to catalog-resolved doc_id or the decimal identity + // when the body didn't provide an "id" field (e.g. skip_payload_fetch). if !obj.contains_key("id") { if let Some(ref doc) = h.doc_id { obj.insert("id".into(), serde_json::json!(doc)); } else { - obj.insert("id".into(), serde_json::json!(h.id)); + obj.insert("id".into(), serde_json::json!(identity)); } } // Always expose internal surrogate for debugging / join use. - obj.insert("_surrogate".into(), serde_json::json!(h.id)); + obj.insert("_surrogate".into(), serde_json::json!(identity)); obj }) .collect(); diff --git a/nodedb/src/control/update_from_join_orchestrator/expand_staged_update_from_join.rs b/nodedb/src/control/update_from_join_orchestrator/expand_staged_update_from_join.rs index 9324ffaee..28c82eccb 100644 --- a/nodedb/src/control/update_from_join_orchestrator/expand_staged_update_from_join.rs +++ b/nodedb/src/control/update_from_join_orchestrator/expand_staged_update_from_join.rs @@ -11,17 +11,13 @@ use crate::bridge::envelope::{PhysicalPlan, Status}; use crate::control::maintenance::clone_materializer::{dispatch_local, read_all_source_rows}; use crate::control::state::SharedState; use crate::control::target_identity::{ - bare_collection_name, derive_document_id, require_surrogate, resolve_target_pk, + bare_collection_name, derive_document_id, resolve_target_pk, }; +use crate::query::ResolvedUpdateRowWire; use crate::types::VShardId; use nodedb_physical::physical_plan::DocumentOp; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; -/// One resolved UPDATE row: `(doc_id, surrogate — None only for legacy rows, -/// post-image body, pre-image body)`. Pre-image lets the Control Plane resolve -/// both sides of a materialized-sum join-key rewrite. -pub(crate) type ResolvedUpdateArm = (String, Option, Vec, Vec); - /// Resolve one in-transaction `UpdateFromJoin` task into the concrete, /// surrogate-carrying `PointPut` tasks its matched target rows expand to. /// `task.txn_id` must be the active transaction. Caller stages + buffers each op. @@ -92,8 +88,10 @@ pub(crate) async fn resolve_and_emit_update_from_join_ops( .await?; let mut out: Vec = Vec::with_capacity(resolved.len()); - for (doc_id, surrogate_u32, body, _old_body) in resolved { - let surrogate = require_surrogate(surrogate_u32, &doc_id, "UPDATE ... FROM")?; + for (_doc_id, surrogate_u32, body, _old_body) in resolved { + // The wire surrogate is never absent: every matched `UPDATE ... FROM` + // row is a storage-keyed row. + let surrogate = nodedb_types::Surrogate::new(surrogate_u32); let document_id = derive_document_id(&target_pk, &body, surrogate); let pk_bytes = document_id.clone().into_bytes(); out.push(PhysicalTask { @@ -125,7 +123,7 @@ async fn resolve_update_rows( state: &SharedState, tenant_id: TenantId, task: &PhysicalTask, -) -> crate::Result> { +) -> crate::Result> { let PhysicalPlan::Document(DocumentOp::UpdateFromJoin { target_collection, source_collection, @@ -198,9 +196,11 @@ async fn resolve_update_rows( decode_resolved_update_rows(&resolve_resp.payload) } -/// Decode the RESOLVE pass payload (a msgpack `Vec<(doc_id, Option, -/// post_image_body)>`; see `encode_resolved_update_rows`). -pub(crate) fn decode_resolved_update_rows(payload: &[u8]) -> crate::Result> { +/// Decode the RESOLVE pass payload (a msgpack `Vec`; +/// see `encode_resolved_update_rows`). +pub(crate) fn decode_resolved_update_rows( + payload: &[u8], +) -> crate::Result> { if payload.is_empty() { return Ok(Vec::new()); } diff --git a/nodedb/src/data/executor/core_loop/state.rs b/nodedb/src/data/executor/core_loop/state.rs index ad1e96491..74d036b9f 100644 --- a/nodedb/src/data/executor/core_loop/state.rs +++ b/nodedb/src/data/executor/core_loop/state.rs @@ -136,13 +136,21 @@ pub struct CoreLoop { std::collections::HashMap<(DatabaseId, TenantId, String, String, u64), String>, /// Reverse map from an indexed document to the HNSW vector ID it produced, - /// keyed by (DatabaseId, TenantId, collection, field, doc_id). `doc_id` is - /// the hex-encoded surrogate row key (matching the key `apply_point_put` - /// indexes under). Populated on every vector index insert; consulted by + /// keyed by (DatabaseId, TenantId, collection, field, storage key). The + /// storage key matches the key `apply_point_put` indexes the row under. + /// Populated on every vector index insert; consulted by /// `apply_point_delete` to soft-delete the orphaned vector when its owning /// document is removed. - pub(in crate::data::executor) vector_doc_map: - std::collections::HashMap<(DatabaseId, TenantId, String, String, String), u32>, + pub(in crate::data::executor) vector_doc_map: std::collections::HashMap< + ( + DatabaseId, + TenantId, + String, + String, + nodedb_types::StorageKey, + ), + u32, + >, /// Base data directory for this core (used for sort spill temp files). pub(in crate::data::executor) data_dir: std::path::PathBuf, diff --git a/nodedb/src/data/executor/core_loop/vector_index_rebuild.rs b/nodedb/src/data/executor/core_loop/vector_index_rebuild.rs index 0d1efbbbe..129c9d99d 100644 --- a/nodedb/src/data/executor/core_loop/vector_index_rebuild.rs +++ b/nodedb/src/data/executor/core_loop/vector_index_rebuild.rs @@ -109,7 +109,6 @@ impl CoreLoop { tid: tenant_id, collection: &collection, storage_key, - surrogate, value: &value, wal_lsn: 0, }) { diff --git a/nodedb/src/data/executor/dispatch/array/surrogate_scan.rs b/nodedb/src/data/executor/dispatch/array/surrogate_scan.rs index 98572f31b..94b8dcdd0 100644 --- a/nodedb/src/data/executor/dispatch/array/surrogate_scan.rs +++ b/nodedb/src/data/executor/dispatch/array/surrogate_scan.rs @@ -339,13 +339,17 @@ mod tests { })); assert_eq!(r.status, Status::Ok, "vector+prefilter failed: {r:?}"); - // Result hits MUST all carry surrogate ids in 1..=5. + // Result hits MUST all carry storage keys whose surrogate is in 1..=5. let json = nodedb_types::msgpack_to_json_string(r.payload.as_ref()).expect("hits msgpack→json"); let hits: Vec = serde_json::from_str(&json).expect("hits json parse"); assert!(!hits.is_empty(), "expected at least one hit, got none"); for hit in &hits { - let id = hit["id"].as_u64().expect("hit.id present") as u32; + let key = hit["id"].as_str().expect("hit.id present"); + let id = nodedb_types::StorageKey::parse(key) + .expect("hit.id is a storage key") + .surrogate() + .as_u32(); assert!( (1..=5).contains(&id), "hit surrogate {id} outside slice prefilter range 1..=5" diff --git a/nodedb/src/data/executor/handlers/bulk_dml/update.rs b/nodedb/src/data/executor/handlers/bulk_dml/update.rs index e4eb9c11b..3c4c3c8e7 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/update.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/update.rs @@ -226,14 +226,12 @@ impl CoreLoop { } for row in projected { let ProjectedUpdateRow { - doc_id, - storage_key, + key: storage_key, current_bytes, old_doc: old_doc_json, mut doc, updated_bytes, } = row; - let doc_id = doc_id.as_str(); // Period lock, both images — matching `execute_point_update`: a // closed period must reject an edit to a row it already holds, // and must reject an edit that assigns the period column into it. @@ -284,7 +282,6 @@ impl CoreLoop { database_id, tid, collection, - doc_id, storage_key: &storage_key, new_body: &updated_bytes, index_paths: &index_paths, @@ -368,10 +365,7 @@ impl CoreLoop { // Event Plane's WAL-replay bulk variants are aggregate // metadata reconstructed only when the live per-row events // were lost — the live path always emits per row. - // `doc_id` is the surrogate hex storage key. A value that fails - // to parse as a minted key is a legacy or user key, taken - // verbatim. - let row_identity = crate::engine::document::store::identity_of(doc_id); + let row_identity = storage_key.to_identity(); // `row_identity` is read again below for `RETURNING`'s `id` field, // so the event-emit boundary gets a clone rather than the move. self.emit_put_event( diff --git a/nodedb/src/data/executor/handlers/bulk_dml/update_persist.rs b/nodedb/src/data/executor/handlers/bulk_dml/update_persist.rs index 42569d95d..3b58e16cf 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/update_persist.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/update_persist.rs @@ -52,7 +52,7 @@ impl CoreLoop { // Hoisted before the transaction opens: the images below borrow them, // and a collection that declares no image-folding enforcement must not // pay for the fold at all. - let doc_id = p.doc_id.to_string(); + let doc_id = *p.storage_key; // Copies of the borrows, not of the documents: they outlive `p`'s move // into the reindex below because they borrow the caller's images, not // the params struct. diff --git a/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs b/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs index 8fd71ab39..833476975 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs @@ -17,11 +17,8 @@ use crate::types::{DatabaseId, TenantId}; /// One matched row and everything the apply loop needs to land it. pub(in crate::data::executor) struct ProjectedUpdateRow { - /// Storage key (the surrogate hex). - pub(in crate::data::executor) doc_id: String, - /// The same storage key, typed — parsed once here so consumers never - /// re-interpret `doc_id`'s shape. - pub(in crate::data::executor) storage_key: crate::engine::document::store::StorageKey, + /// The row's storage key. + pub(in crate::data::executor) key: crate::engine::document::store::StorageKey, /// The row as stored before the update — the `old_value` of the emitted /// event and the old side of the secondary-index diff. pub(in crate::data::executor) current_bytes: Vec, @@ -178,8 +175,7 @@ impl CoreLoop { }; projected.push(ProjectedUpdateRow { - doc_id: doc_id_owned, - storage_key: *key, + key: *key, current_bytes, old_doc, doc, diff --git a/nodedb/src/data/executor/handlers/columnar_write/geometry_index.rs b/nodedb/src/data/executor/handlers/columnar_write/geometry_index.rs index 1cc001e1c..48015054c 100644 --- a/nodedb/src/data/executor/handlers/columnar_write/geometry_index.rs +++ b/nodedb/src/data/executor/handlers/columnar_write/geometry_index.rs @@ -7,6 +7,7 @@ use nodedb_types::columnar::{ColumnType, ColumnarSchema}; use nodedb_types::value::Value; use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::point::apply_put::SpatialEntryId; use crate::data::executor::task::ExecutionTask; impl CoreLoop { @@ -60,11 +61,12 @@ impl CoreLoop { // document twice. Mirrors `apply_point_put_spatial`'s use of // the same helper; the removed tuples aren't needed here // since this insert path has no transactional undo to feed. + let spatial_entry_id = SpatialEntryId::from_user_id(&doc_id); let _ = self.remove_document_spatial_indexes( db_id.as_u64(), tid.as_u64(), collection, - &doc_id, + spatial_entry_id, ); for &col_idx in &geom_cols { let col_def = &schema.columns[col_idx]; @@ -86,7 +88,7 @@ impl CoreLoop { }; let bbox = nodedb_types::bbox::geometry_bbox(&geom); let index_key = (db_id, tid, collection.to_string(), col_def.name.clone()); - let entry_id = crate::util::fnv1a_hash(doc_id.as_bytes()); + let entry_id = spatial_entry_id.as_u64(); let memory = nodedb_mem::ScopedMemory::new( self.governor.clone(), db_id, diff --git a/nodedb/src/data/executor/handlers/document/resolve/bulk.rs b/nodedb/src/data/executor/handlers/document/resolve/bulk.rs index afe95b99c..e513ed705 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/bulk.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/bulk.rs @@ -106,8 +106,7 @@ impl CoreLoop { let mut returned_docs: Vec = Vec::new(); for row in projected { let ProjectedUpdateRow { - doc_id, - storage_key, + key: storage_key, current_bytes, old_doc: _, doc, @@ -117,9 +116,10 @@ impl CoreLoop { // `execute_bulk_update` decides it — on the same JSON document. rls_write_gate::admit_row(rls_write_check, &doc, tid, collection) .map_err(ErrorCode::from)?; - // The projection stage already parsed `doc_id` as a storage key, - // so its surrogate identity is always available here. let surrogate = storage_key.surrogate(); + // Rendered once here — `ResolvedPut::document_id` is the one `&str` + // API downstream that still needs the storage key as text. + let doc_id = storage_key.to_string(); mutations.push(put_mutation(ResolvedPut { collection, document_id: &doc_id, diff --git a/nodedb/src/data/executor/handlers/hybrid_key.rs b/nodedb/src/data/executor/handlers/hybrid_key.rs index cdaff8ea4..83766f6cc 100644 --- a/nodedb/src/data/executor/handlers/hybrid_key.rs +++ b/nodedb/src/data/executor/handlers/hybrid_key.rs @@ -8,11 +8,7 @@ use std::fmt; -use nodedb_types::{StorageKey, Surrogate}; - -/// Prefix of the rendered key for a vector hit with no surrogate binding. -/// The Control Plane hybrid translator passes such a `doc_id` through untouched. -pub(in crate::data::executor) const HEADLESS_SENTINEL_PREFIX: &str = "__local_"; +use nodedb_types::{HEADLESS_SENTINEL_PREFIX, StorageKey, Surrogate}; /// One hybrid-search leg hit, in the key space RRF fuses on. /// @@ -20,6 +16,11 @@ pub(in crate::data::executor) const HEADLESS_SENTINEL_PREFIX: &str = "__local_"; /// storage key the response envelope carries. `Headless` is a vector hit with /// no surrogate binding: it ranks in the vector leg only, fuses with nothing, /// and renders as the `__local_` sentinel. +/// +/// Wire encoding: `ToMessagePack`/`FromMessagePack` write and read this type +/// as its `Display` string — the hex storage key for `Bound`, or +/// `__local_` for `Headless`. The Control Plane response translator +/// decodes the same two shapes; see `response_translate/hit_key.rs`. #[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)] pub(in crate::data::executor) enum HybridFusionKey { Bound(StorageKey), @@ -50,6 +51,33 @@ impl fmt::Display for HybridFusionKey { } } +impl serde::Serialize for HybridFusionKey { + fn serialize(&self, serializer: S) -> Result { + serializer.collect_str(self) + } +} + +impl zerompk::ToMessagePack for HybridFusionKey { + fn write(&self, writer: &mut W) -> zerompk::Result<()> { + writer.write_string(&self.to_string()) + } +} + +impl<'de> zerompk::FromMessagePack<'de> for HybridFusionKey { + fn read>(reader: &mut R) -> zerompk::Result { + let text = reader.read_string()?; + if let Some(local_id) = text.strip_prefix(HEADLESS_SENTINEL_PREFIX) { + let local_id = local_id + .parse::() + .map_err(|_| zerompk::Error::InvalidMarker(0))?; + return Ok(Self::Headless(local_id)); + } + StorageKey::parse(&text) + .map(Self::Bound) + .ok_or(zerompk::Error::InvalidMarker(0)) + } +} + #[cfg(test)] mod tests { use super::*; @@ -75,4 +103,16 @@ mod tests { HybridFusionKey::for_surrogate(Surrogate::new(7)) ); } + + #[test] + fn wire_round_trips_bound_and_headless() { + for key in [ + HybridFusionKey::for_surrogate(Surrogate::new(42)), + HybridFusionKey::Headless(7), + ] { + let bytes = zerompk::to_msgpack_vec(&key).expect("encode"); + let decoded: HybridFusionKey = zerompk::from_msgpack(&bytes).expect("decode"); + assert_eq!(decoded, key); + } + } } diff --git a/nodedb/src/data/executor/handlers/hybrid_overlay.rs b/nodedb/src/data/executor/handlers/hybrid_overlay.rs index c9c053616..2515f9656 100644 --- a/nodedb/src/data/executor/handlers/hybrid_overlay.rs +++ b/nodedb/src/data/executor/handlers/hybrid_overlay.rs @@ -26,8 +26,6 @@ //! The graph leg's RYOW is a separate concern and is deliberately not touched //! here — the triple handler still reads committed graph state. -use std::collections::HashMap; - use nodedb_fts::posting::TextSearchResult; use nodedb_types::{Surrogate, SurrogateBitmap}; @@ -85,22 +83,6 @@ impl CoreLoop { .iter() .map(|r| super::vector_search::build_search_hit(vector_collection, r.id, r.distance)) .collect(); - // Pin each base hit's fusion key by its `id` before the merge runs. A - // headless base hit carries a raw local id in `id`, so without this - // pin the ranked list below would misread it as a global surrogate. - // The merge updates a same-id hit in place, so the pinned key still - // names the row after a staged put; staged hits the merge adds carry - // no pin and resolve to their (real) surrogate. - let base_keys: HashMap = vector_results - .iter() - .zip(vector_hits.iter()) - .map(|(r, hit)| { - ( - hit.id, - super::vector_search::vector_leg_key(vector_collection, r.id), - ) - }) - .collect(); let mut text_scored: Vec<(Surrogate, f32, bool)> = text_results .iter() .map(|r| (r.doc_id, r.score, r.fuzzy)) @@ -168,12 +150,7 @@ impl CoreLoop { .iter() .enumerate() .map(|(rank, hit)| RankedResult { - // A base hit keeps its pinned key; a staged hit the merge added - // carries a real surrogate in `id`. - document_id: base_keys - .get(&hit.id) - .copied() - .unwrap_or_else(|| HybridFusionKey::for_surrogate(Surrogate::new(hit.id))), + document_id: hit.id, rank, score: hit.distance, source: "vector", diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs index e6694db60..a7c2b0e0e 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs @@ -92,7 +92,7 @@ impl CoreLoop { vector_id: d.vector_id, collection: d.collection, field: d.field, - doc_id: d.doc_id, + doc_id: Some(d.doc_id), }); } } diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/apply_support.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/apply_support.rs index 168e0c99e..3845f4109 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/apply_support.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/apply_support.rs @@ -30,7 +30,7 @@ pub(super) fn record_put_index_undo(undo_log: &mut Vec, outcome: &mut vector_id: d.vector_id, collection: d.collection, field: d.field, - doc_id: d.doc_id, + doc_id: Some(d.doc_id), }); } for (key, entry_id) in std::mem::take(&mut outcome.spatial_inserts) { diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/resolve.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/resolve.rs index 1d12054ac..17923d450 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/resolve.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/resolve.rs @@ -13,9 +13,10 @@ impl CoreLoop { /// RESOLVE pass: return ALL resolved arms without writing. /// /// Response payload is a msgpack 3-tuple `(updates, deletes, inserts)`: - /// - `updates`: `Vec<(doc_id, Option, body_msgpack, + /// - `updates`: `Vec<(doc_id, surrogate_u32, body_msgpack, /// old_body_msgpack)>` — the existing target row's storage key, its - /// registered surrogate, the post-update body, and the row's PRE-image + /// registered surrogate (always present — every matched row is + /// storage-keyed), the post-update body, and the row's PRE-image /// (which is what lets the Control Plane resolve BOTH sides of a /// materialized-sum join-key rewrite). /// - `deletes`: `Vec<(doc_id, Option, body_msgpack)>` — the @@ -47,7 +48,7 @@ impl CoreLoop { .map(|u| { ( u.key.to_string(), - Some(u.key.surrogate().as_u32()), + u.key.surrogate().as_u32(), u.body, u.old_body, ) diff --git a/nodedb/src/data/executor/handlers/point/apply_delete.rs b/nodedb/src/data/executor/handlers/point/apply_delete.rs index d377b67b6..8e9d1931e 100644 --- a/nodedb/src/data/executor/handlers/point/apply_delete.rs +++ b/nodedb/src/data/executor/handlers/point/apply_delete.rs @@ -16,6 +16,7 @@ use crate::data::executor::enforcement::{append_only, period_lock, retention}; use nodedb_physical::physical_plan::ResolvedSumTarget; use nodedb_types::Surrogate; +use crate::data::executor::handlers::point::apply_put::SpatialEntryId; use crate::data::executor::handlers::point::apply_put::VectorIndexDelta; use crate::data::executor::handlers::point::apply_put::map_enforcement_error; use crate::data::executor::spatial_key::SpatialIndexKey; @@ -145,8 +146,6 @@ impl CoreLoop { let _ = user_roles; let storage_key = crate::engine::document::store::StorageKey::for_surrogate(surrogate); - let row_key = storage_key.to_string(); - let row_key = row_key.as_str(); let bitemporal = self.is_bitemporal(database_id, tid, collection); let config_key = ( crate::types::DatabaseId::new(database_id), @@ -372,13 +371,16 @@ impl CoreLoop { // the spatial cascade below are captured so a transactional caller can // reverse them. let mut mark_node_deleted_capture: Option = None; - // The put path hashes the hex-surrogate storage key (== `row_key`) as - // the R-tree entry id, so the shared removal hashes the same key to - // find and drop every per-field entry + reverse-map pair for this - // document. Captures each removed `(skey, entry_id, bbox, doc)` for - // reversible undo. - let spatial_deletes = - self.remove_document_spatial_indexes(database_id, tid, collection, row_key); + // The put path hashes the same storage key via `SpatialEntryId`, so + // the shared removal hashes the same key to find and drop every + // per-field entry + reverse-map pair for this document. Captures + // each removed `(skey, entry_id, bbox, doc)` for reversible undo. + let spatial_deletes = self.remove_document_spatial_indexes( + database_id, + tid, + collection, + SpatialEntryId::from_storage_key(storage_key), + ); // Record deletion for edge referential integrity. Capture the id // for undo ONLY when this call newly marked it — un-marking a node diff --git a/nodedb/src/data/executor/handlers/point/apply_put/core.rs b/nodedb/src/data/executor/handlers/point/apply_put/core.rs index 102602234..cddfeb803 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/core.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/core.rs @@ -303,7 +303,6 @@ impl CoreLoop { tid, collection, storage_key, - surrogate, value, wal_lsn: wal_lsn.map(|l| l.as_u64()).unwrap_or(0), }, diff --git a/nodedb/src/data/executor/handlers/point/apply_put/index.rs b/nodedb/src/data/executor/handlers/point/apply_put/index.rs index 0f4804ad1..feae9eaa4 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/index.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/index.rs @@ -11,6 +11,37 @@ use crate::data::executor::doc_format; use crate::data::executor::spatial_key::SpatialIndexKey; use crate::engine::document::store::StorageKey; +/// Entry id of one row in a spatial R-tree: `fnv1a_hash` of the row's identity text. +/// +/// A document row hashes its rendered storage key; a columnar row hashes its +/// `id` column value. The put side and the remove side both construct this +/// through the matching constructor, so they can never hash the row's +/// identity differently and silently miss each other's entry. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub(in crate::data::executor) struct SpatialEntryId(u64); + +impl SpatialEntryId { + /// A document row: the rendered storage key. + pub fn from_storage_key(key: StorageKey) -> Self { + Self::from_rendered(&key.to_string()) + } + + /// A document row whose storage key is already rendered — avoids a + /// second `to_string()` when the caller already holds the text. + pub fn from_rendered(rendered_key: &str) -> Self { + Self(crate::util::fnv1a_hash(rendered_key.as_bytes())) + } + + /// A columnar row: its `id` column value. + pub fn from_user_id(id: &str) -> Self { + Self(crate::util::fnv1a_hash(id.as_bytes())) + } + + pub fn as_u64(self) -> u64 { + self.0 + } +} + impl CoreLoop { /// Spatial R-tree + columnar ingest side-effect: parse geometry fields, /// insert into the per-field R-tree, maintain the reverse entry→doc map, @@ -40,6 +71,7 @@ impl CoreLoop { // Rendered once here; every reverse-map / hash use below shares it. let document_id = storage_key.to_string(); let document_id = document_id.as_str(); + let spatial_entry_id = SpatialEntryId::from_rendered(document_id); let mut inserts = Vec::new(); // Re-indexing a document must REPLACE, not append: `RTree::insert` // blindly pushes a fresh entry even when one with this `entry_id` @@ -49,7 +81,8 @@ impl CoreLoop { // this document first (idempotent — a no-op on a genuine first insert). // The removed tuples are discarded here, mirroring the vector put path: // only the new inserts are captured for transactional undo. - let _ = self.remove_document_spatial_indexes(database_id, tid, collection, document_id); + let _ = + self.remove_document_spatial_indexes(database_id, tid, collection, spatial_entry_id); // Spatial index: detect geometry fields and insert into R-tree. // Tries to parse each field as a GeoJSON Geometry — either a native // JSON object (schemaless document writes, e.g. @@ -88,7 +121,7 @@ impl CoreLoop { let db_id = nodedb_types::DatabaseId::new(database_id); let tid_id = crate::types::TenantId::new(tid); let spatial_key = (db_id, tid_id, collection.to_string(), field_name.clone()); - let entry_id = crate::util::fnv1a_hash(document_id.as_bytes()); + let entry_id = spatial_entry_id.as_u64(); let memory = nodedb_mem::ScopedMemory::new( self.governor.clone(), db_id, @@ -127,11 +160,11 @@ impl CoreLoop { /// Remove every R-tree entry (and its paired `spatial_doc_map` reverse /// entry) this document produced across all of the collection's per-field - /// spatial indexes, keyed by `fnv1a_hash(document_id)` — the same hash the - /// insert path uses. Shared by the PointDelete cascade (which orphans the - /// geometry of a removed row) and `apply_point_put_spatial` (which must - /// clear a document's prior geometry before re-inserting, since - /// `RTree::insert` appends rather than replaces). + /// spatial indexes, keyed by `entry_id` — the same [`SpatialEntryId`] the + /// insert path constructs. Shared by the PointDelete cascade (which + /// orphans the geometry of a removed row) and `apply_point_put_spatial` + /// (which must clear a document's prior geometry before re-inserting, + /// since `RTree::insert` appends rather than replaces). /// /// The bbox is read BEFORE the R-tree `delete` (which does not return the /// removed geometry) so a transactional caller can push @@ -144,10 +177,10 @@ impl CoreLoop { database_id: u64, tid: u64, collection: &str, - document_id: &str, + entry_id: SpatialEntryId, ) -> Vec<(SpatialIndexKey, u64, nodedb_types::BoundingBox, String)> { let mut spatial_deletes = Vec::new(); - let entry_id = crate::util::fnv1a_hash(document_id.as_bytes()); + let entry_id = entry_id.as_u64(); let db_id = nodedb_types::DatabaseId::new(database_id); let tid_id = crate::types::TenantId::new(tid); let spatial_fields: Vec = self diff --git a/nodedb/src/data/executor/handlers/point/apply_put/mod.rs b/nodedb/src/data/executor/handlers/point/apply_put/mod.rs index db6113c16..47bd78657 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/mod.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/mod.rs @@ -12,5 +12,6 @@ pub(in crate::data::executor::handlers::point) mod types; pub(in crate::data::executor) mod unique; pub(in crate::data::executor::handlers::point) mod vector; +pub(in crate::data::executor) use index::SpatialEntryId; pub(in crate::data::executor) use types::{PointPutOutcome, PointPutParams, map_enforcement_error}; pub(in crate::data::executor) use vector::{VectorIndexDelta, VectorIndexPutParams}; diff --git a/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs b/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs index 9970821c6..d6e320003 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs @@ -39,14 +39,9 @@ impl CoreLoop { tid, collection, storage_key, - surrogate, value, wal_lsn, } = params; - // Rendered once here — `vector_doc_map` and its undo entries are - // keyed by text. - let document_id = storage_key.to_string(); - let document_id = document_id.as_str(); let mut inserts: Vec = Vec::new(); // Vector index: if the strict schema declares Vector(dim) columns, @@ -103,9 +98,8 @@ impl CoreLoop { index_key, collection, field_name, - document_id, + storage_key, floats, - surrogate, wal_lsn, }) { inserts.push(delta); @@ -186,9 +180,8 @@ impl CoreLoop { index_key: store_key, collection, field_name, - document_id, + storage_key, floats, - surrogate, wal_lsn, }) { inserts.push(delta); @@ -259,9 +252,8 @@ impl CoreLoop { index_key, collection, field_name, - document_id, + storage_key, floats, - surrogate, wal_lsn, } = params; let _ = self.remove_document_vector_index_field( @@ -269,10 +261,10 @@ impl CoreLoop { tid, collection, field_name, - document_id, + storage_key, ); let coll = self.vector_collections.get_mut(&index_key)?; - let vector_id = coll.insert_with_surrogate(floats, surrogate); + let vector_id = coll.insert_with_surrogate(floats, storage_key.surrogate()); coll.note_checkpoint_lsn(wal_lsn); self.vector_doc_map.insert( ( @@ -280,7 +272,7 @@ impl CoreLoop { index_key.1, collection.to_string(), field_name.to_string(), - document_id.to_string(), + storage_key, ), vector_id, ); @@ -289,7 +281,7 @@ impl CoreLoop { vector_id, collection: collection.to_string(), field: field_name.to_string(), - doc_id: document_id.to_string(), + doc_id: storage_key, }) } } @@ -416,7 +408,6 @@ mod tests { tid, collection, storage_key, - surrogate, value: &first, wal_lsn: 0, }) @@ -428,7 +419,6 @@ mod tests { tid, collection, storage_key, - surrogate, value: &second, wal_lsn: 0, }) @@ -475,7 +465,6 @@ mod tests { tid, collection, storage_key, - surrogate, value: &doc, wal_lsn: 0, }) @@ -528,7 +517,6 @@ mod tests { tid, collection, storage_key, - surrogate, value: &body, wal_lsn: 0, }) @@ -569,7 +557,6 @@ mod tests { tid, collection, storage_key, - surrogate, value: &body, wal_lsn: 0, }); diff --git a/nodedb/src/data/executor/handlers/point/apply_put/vector/remove.rs b/nodedb/src/data/executor/handlers/point/apply_put/vector/remove.rs index 07352f78a..57be0fdce 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/vector/remove.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/vector/remove.rs @@ -9,9 +9,9 @@ use super::types::VectorIndexDelta; impl CoreLoop { /// Soft-delete the single HNSW vector node a document produced for one - /// `field`, keyed by its hex-surrogate storage `row_key`, and drop the - /// paired `vector_doc_map` reverse entry. Returns the removed delta, or - /// `None` when the `(db, tid, collection, field, row_key)` key had no prior + /// `field`, keyed by its storage key, and drop the paired + /// `vector_doc_map` reverse entry. Returns the removed delta, or `None` + /// when the `(db, tid, collection, field, storage_key)` key had no prior /// node (a genuine first insert). This is the per-field unit the whole-doc /// `remove_document_vector_indexes` loops over, and the put path calls it /// for the current field only so a sibling field's just-inserted node is @@ -22,7 +22,7 @@ impl CoreLoop { tid: u64, collection: &str, field: &str, - row_key: &str, + storage_key: StorageKey, ) -> Option { let db_id = nodedb_types::DatabaseId::new(database_id); let tid_id = crate::types::TenantId::new(tid); @@ -31,7 +31,7 @@ impl CoreLoop { tid_id, collection.to_string(), field.to_string(), - row_key.to_string(), + storage_key, ); let vector_id = self.vector_doc_map.remove(&doc_key)?; let index_key = Self::vector_index_key(database_id, tid, collection, field); @@ -43,13 +43,13 @@ impl CoreLoop { vector_id, collection: collection.to_string(), field: field.to_string(), - doc_id: row_key.to_string(), + doc_id: storage_key, }) } /// Soft-delete every HNSW vector entry a document produced, keyed by its - /// hex-surrogate storage `row_key`, and drop the paired `vector_doc_map` - /// reverse entries. Shared by the PointDelete cascade (which orphans the + /// storage key, and drop the paired `vector_doc_map` reverse entries. + /// Shared by the PointDelete cascade (which orphans the /// vectors of a removed row) and the PointUpdate re-index (which must clear /// the surrogate's old embedding before inserting the new one, since /// `insert_with_surrogate` appends rather than replaces). @@ -77,15 +77,13 @@ impl CoreLoop { if candidate_fields.is_empty() { return vector_deletes; } - // The vector reverse map keys nodes by the rendered storage key. - let row_key = storage_key.to_string(); for field in candidate_fields { if let Some(delta) = self.remove_document_vector_index_field( database_id, tid, collection, &field, - &row_key, + storage_key, ) { vector_deletes.push(delta); } diff --git a/nodedb/src/data/executor/handlers/point/apply_put/vector/types.rs b/nodedb/src/data/executor/handlers/point/apply_put/vector/types.rs index 8731fc7a8..49af195d9 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/vector/types.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/vector/types.rs @@ -12,7 +12,6 @@ pub(in crate::data::executor) struct VectorIndexPutParams<'a> { pub tid: u64, pub collection: &'a str, pub storage_key: crate::engine::document::store::StorageKey, - pub surrogate: nodedb_types::Surrogate, pub value: &'a [u8], pub wal_lsn: u64, } @@ -28,21 +27,21 @@ pub(in crate::data::executor) struct VectorIndexDelta { pub vector_id: u32, pub collection: String, pub field: String, - pub doc_id: String, + pub doc_id: crate::engine::document::store::StorageKey, } /// Inputs to `remove_then_insert_vector_field`, the shared per-field /// remove-before-insert tail of `apply_point_put_vector_indexes`'s strict and /// schemaless arms, once each has resolved its own `index_key` and extracted -/// `floats` for `field_name`. +/// `floats` for `field_name`. `storage_key` carries the row's surrogate: call +/// `storage_key.surrogate()` rather than threading a second surrogate field. pub(super) struct VectorFieldInsert<'a> { pub(super) database_id: u64, pub(super) tid: u64, pub(super) index_key: (nodedb_types::DatabaseId, crate::types::TenantId, String), pub(super) collection: &'a str, pub(super) field_name: &'a str, - pub(super) document_id: &'a str, + pub(super) storage_key: crate::engine::document::store::StorageKey, pub(super) floats: Vec, - pub(super) surrogate: nodedb_types::Surrogate, pub(super) wal_lsn: u64, } diff --git a/nodedb/src/data/executor/handlers/point/update/exec.rs b/nodedb/src/data/executor/handlers/point/update/exec.rs index dc82ad854..89668c2d6 100644 --- a/nodedb/src/data/executor/handlers/point/update/exec.rs +++ b/nodedb/src/data/executor/handlers/point/update/exec.rs @@ -68,8 +68,6 @@ impl CoreLoop { declared_primary_key, } = params; let storage_key = StorageKey::for_surrogate(surrogate); - let row_key = storage_key.to_string(); - let row_key = row_key.as_str(); let document_identity = RowIdentity::from_user_key(document_id); debug!( core = self.core_id, @@ -230,7 +228,6 @@ impl CoreLoop { database_id, tid, collection, - row_key, storage_key: &storage_key, current_bytes: ¤t_bytes, updated_bytes: &updated_bytes, diff --git a/nodedb/src/data/executor/handlers/point/update/persist.rs b/nodedb/src/data/executor/handlers/point/update/persist.rs index 7aabce0cb..690cecb2d 100644 --- a/nodedb/src/data/executor/handlers/point/update/persist.rs +++ b/nodedb/src/data/executor/handlers/point/update/persist.rs @@ -34,8 +34,6 @@ pub(in crate::data::executor) struct PointUpdatePersist<'a> { pub(in crate::data::executor) database_id: u64, pub(in crate::data::executor) tid: u64, pub(in crate::data::executor) collection: &'a str, - /// Storage key (the surrogate hex), rendered — used in error messages. - pub(in crate::data::executor) row_key: &'a str, /// The same storage key, typed — what every storage call below takes. pub(in crate::data::executor) storage_key: &'a crate::engine::document::store::StorageKey, /// The row as it was before this update — the old side of the index diff, @@ -70,7 +68,6 @@ impl CoreLoop { database_id, tid, collection, - row_key, storage_key, current_bytes, updated_bytes, @@ -142,7 +139,7 @@ impl CoreLoop { engine: "sparse".into(), detail: format!( "bitemporal update: document failed to decode for \ - versioned-index diff (collection {collection}, id {row_key}): {e}" + versioned-index diff (collection {collection}, id {storage_key}): {e}" ), }), None => self @@ -200,7 +197,6 @@ impl CoreLoop { database_id, tid, collection, - doc_id: row_key, storage_key, new_body: updated_bytes, index_paths: &index_paths, @@ -219,7 +215,7 @@ impl CoreLoop { engine: "sparse".into(), detail: format!( "non-bitemporal update: document failed to decode for \ - secondary-index diff (collection {collection}, id {row_key}): {e}" + secondary-index diff (collection {collection}, id {storage_key}): {e}" ), }) } diff --git a/nodedb/src/data/executor/handlers/point/update_reindex.rs b/nodedb/src/data/executor/handlers/point/update_reindex.rs index 3e968ba94..586f8ab18 100644 --- a/nodedb/src/data/executor/handlers/point/update_reindex.rs +++ b/nodedb/src/data/executor/handlers/point/update_reindex.rs @@ -43,8 +43,6 @@ pub(in crate::data::executor) struct NonbitemporalUpdateReindex<'a> { pub database_id: u64, pub tid: u64, pub collection: &'a str, - pub doc_id: &'a str, - /// The same storage key, typed — `put_in_txn` below takes it directly. pub storage_key: &'a crate::engine::document::store::StorageKey, /// New stored bytes for the primary document row. pub new_body: &'a [u8], diff --git a/nodedb/src/data/executor/handlers/point/update_reindex_vector.rs b/nodedb/src/data/executor/handlers/point/update_reindex_vector.rs index cb6b2ca04..d3e6a9c04 100644 --- a/nodedb/src/data/executor/handlers/point/update_reindex_vector.rs +++ b/nodedb/src/data/executor/handlers/point/update_reindex_vector.rs @@ -107,7 +107,6 @@ impl CoreLoop { tid: p.tid, collection: p.collection, storage_key: p.storage_key, - surrogate: p.storage_key.surrogate(), value: mp, wal_lsn: 0, })?; diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/vector_merge.rs b/nodedb/src/data/executor/handlers/transaction/overlay/vector_merge.rs index a41af5fb7..451dcb2f7 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/vector_merge.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/vector_merge.rs @@ -26,6 +26,7 @@ use std::collections::HashMap; use nodedb_types::{PayloadAtom, Surrogate, SurrogateBitmap, value::Value}; use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::hybrid_key::HybridFusionKey; use crate::data::executor::handlers::transaction::overlay::Staged; use crate::data::executor::response_codec::VectorSearchHit; use crate::engine::vector::distance::DistanceMetric; @@ -148,7 +149,7 @@ fn value_as_f64(v: &Value) -> Option { /// After a `Vec::remove(idx)` shifts every later element left by one, shift /// every recorded index greater than `idx` in `seen` to match. -fn reindex_after_removal(seen: &mut HashMap, removed_idx: usize) { +fn reindex_after_removal(seen: &mut HashMap, removed_idx: usize) { for idx in seen.values_mut() { if *idx > removed_idx { *idx -= 1; @@ -199,16 +200,17 @@ impl CoreLoop { // Read-your-own-writes refreshes the lease (see the reaper). self.touch_overlay(txn_id); if let Some(overlay) = self.txn_overlays.get(&txn_id) { - let mut seen: HashMap = hits + let mut seen: HashMap = hits .iter() .enumerate() .map(|(idx, h)| (h.id, idx)) .collect(); for (surrogate, staged) in overlay.iter_for_collection(&coll_key) { + let key = HybridFusionKey::for_surrogate(Surrogate::new(surrogate)); match staged { Staged::Tombstone => { - if let Some(idx) = seen.remove(&surrogate) { + if let Some(idx) = seen.remove(&key) { hits.remove(idx); reindex_after_removal(&mut seen, idx); } @@ -230,7 +232,7 @@ impl CoreLoop { .iter() .all(|atom| payload_atom_matches(atom, &doc)); if !passes_filter_bitmap || !passes_payload { - if let Some(idx) = seen.remove(&surrogate) { + if let Some(idx) = seen.remove(&key) { hits.remove(idx); reindex_after_removal(&mut seen, idx); } @@ -238,15 +240,15 @@ impl CoreLoop { } let dist = nodedb_vector::distance::distance(query_vector, &vector, metric); - match seen.get(&surrogate).copied() { + match seen.get(&key).copied() { Some(idx) => { hits[idx].distance = dist; hits[idx].body = Some(body.clone()); } None => { - seen.insert(surrogate, hits.len()); + seen.insert(key, hits.len()); hits.push(VectorSearchHit { - id: surrogate, + id: key, distance: dist, doc_id: None, body: Some(body.clone()), diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs index 5c86558e5..c4d8984d4 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs @@ -144,7 +144,7 @@ impl CoreLoop { vector_id: delta.vector_id, collection: delta.collection, field: delta.field, - doc_id: delta.doc_id, + doc_id: Some(delta.doc_id), }); } for (key, entry_id) in target.outcome.spatial_inserts { @@ -182,7 +182,7 @@ impl CoreLoop { vector_id: delta.vector_id, collection: delta.collection, field: delta.field, - doc_id: delta.doc_id, + doc_id: Some(delta.doc_id), }); } diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs index e20d9db90..e6c1d3d94 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs @@ -306,7 +306,7 @@ impl CoreLoop { vector_id: delta.vector_id, collection: delta.collection, field: delta.field, - doc_id: delta.doc_id, + doc_id: Some(delta.doc_id), }); } for (key, entry_id) in target.outcome.spatial_inserts { @@ -338,7 +338,7 @@ impl CoreLoop { vector_id: delta.vector_id, collection: delta.collection, field: delta.field, - doc_id: delta.doc_id, + doc_id: Some(delta.doc_id), }); } diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_write.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_write.rs index 8fd390dea..086c11f89 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_write.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_write.rs @@ -100,15 +100,15 @@ impl CoreLoop { } // This is the direct primary-vector write path (VectorOp), not // the document auto-index cascade — it never populates - // `vector_doc_map` (that reverse map is keyed by document id, - // which this path doesn't have). Empty `doc_id` tells + // `vector_doc_map` (that reverse map is keyed by storage key, + // which this path doesn't have). `None` `doc_id` tells // `apply_undo_vector` to skip the `vector_doc_map` mutation. undo_log.push(UndoEntry::InsertVector { index_key, vector_id, collection: collection.to_string(), field: field_name.to_string(), - doc_id: String::new(), + doc_id: None, }); Ok(self.response_ok(dummy_task)) } @@ -128,14 +128,14 @@ impl CoreLoop { && index.delete(vector_id) { // Same direct primary-vector path as `VectorOp::Insert` - // above — no `vector_doc_map` entry to restore, so an - // empty `doc_id` skips that mutation in `apply_undo_vector`. + // above — no `vector_doc_map` entry to restore, so a + // `None` `doc_id` skips that mutation in `apply_undo_vector`. undo_log.push(UndoEntry::DeleteVector { index_key, vector_id, collection: collection.to_string(), field: String::new(), - doc_id: String::new(), + doc_id: None, }); } self.response_ok(dummy_task) diff --git a/nodedb/src/data/executor/handlers/transaction/undo/apply.rs b/nodedb/src/data/executor/handlers/transaction/undo/apply.rs index 62609a762..6c756e434 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/apply.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/apply.rs @@ -35,11 +35,11 @@ impl CoreLoop { // Reverse the forward insert's `vector_doc_map` write — // without this a rolled-back insert leaves a stale // doc→vector_id mapping behind (unbounded leak), mirroring - // `apply_undo_spatial`'s `spatial_doc_map.remove`. Empty + // `apply_undo_spatial`'s `spatial_doc_map.remove`. `None` // `doc_id` marks the direct primary-vector write path // (`PhysicalPlan::Vector`), which never populates // `vector_doc_map` — skip the mutation for that path. - if !doc_id.is_empty() { + if let Some(doc_id) = doc_id { self.vector_doc_map.remove(&( index_key.0, index_key.1, @@ -78,10 +78,10 @@ impl CoreLoop { // doc→vector reverse lookup missing, so a later delete of // the same document can never find (and soft-delete) its // vector: a permanent orphan. Mirrors - // `apply_undo_spatial`'s `spatial_doc_map.insert`. Empty + // `apply_undo_spatial`'s `spatial_doc_map.insert`. `None` // `doc_id` marks the direct primary-vector write path, // which never populates `vector_doc_map` — skip it there. - if !doc_id.is_empty() { + if let Some(doc_id) = doc_id { self.vector_doc_map.insert( (index_key.0, index_key.1, collection, field, doc_id), vector_id, @@ -910,14 +910,20 @@ mod tests { crate::data::executor::core_loop::CoreLoop::vector_index_key(DB, TID, "c", "emb") } - fn vector_doc_key() -> (nodedb_types::DatabaseId, TenantId, String, String, String) { + fn vector_doc_key() -> ( + nodedb_types::DatabaseId, + TenantId, + String, + String, + nodedb_types::StorageKey, + ) { let key = vector_index_key(); ( key.0, key.1, "c".to_string(), "emb".to_string(), - "d1".to_string(), + nodedb_types::StorageKey::for_surrogate(nodedb_types::Surrogate::new(1)), ) } @@ -949,7 +955,7 @@ mod tests { vector_id, collection: "c".to_string(), field: "emb".to_string(), - doc_id: "d1".to_string(), + doc_id: Some(vector_doc_key().4), }; core.apply_undo_vector(TID, 0, undo).unwrap(); @@ -988,7 +994,7 @@ mod tests { vector_id, collection: "c".to_string(), field: "emb".to_string(), - doc_id: "d1".to_string(), + doc_id: Some(vector_doc_key().4), }; core.apply_undo_vector(TID, 0, undo).unwrap(); diff --git a/nodedb/src/data/executor/handlers/transaction/undo/entry.rs b/nodedb/src/data/executor/handlers/transaction/undo/entry.rs index 64d30ae38..ab205f59c 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/entry.rs @@ -93,11 +93,13 @@ pub(in crate::data::executor) enum UndoEntry { InsertVector { index_key: (nodedb_types::DatabaseId, TenantId, String), vector_id: u32, - /// Collection, field, and doc id — the `vector_doc_map` key + /// Collection, field, and storage key — the `vector_doc_map` key /// components the forward insert wrote, needed to remove them. + /// `None` marks the direct primary-vector write path + /// (`PhysicalPlan::Vector`), which never populates `vector_doc_map`. collection: String, field: String, - doc_id: String, + doc_id: Option, }, /// Undo a VectorDelete by un-deleting (clearing tombstone) and restoring /// the `vector_doc_map` entry the forward delete removed — mirroring @@ -108,11 +110,13 @@ pub(in crate::data::executor) enum UndoEntry { DeleteVector { index_key: (nodedb_types::DatabaseId, TenantId, String), vector_id: u32, - /// Collection, field, and doc id — the `vector_doc_map` key + /// Collection, field, and storage key — the `vector_doc_map` key /// components the forward delete removed, needed to restore them. + /// `None` marks the direct primary-vector write path + /// (`PhysicalPlan::Vector`), which never populates `vector_doc_map`. collection: String, field: String, - doc_id: String, + doc_id: Option, }, /// Undo a spatial R-tree insert by removing the entry from the per-field /// R-tree and deleting its reverse `spatial_doc_map` record. diff --git a/nodedb/src/data/executor/handlers/update_from_join.rs b/nodedb/src/data/executor/handlers/update_from_join.rs index 7e090fa10..1f98db2b8 100644 --- a/nodedb/src/data/executor/handlers/update_from_join.rs +++ b/nodedb/src/data/executor/handlers/update_from_join.rs @@ -297,8 +297,8 @@ impl CoreLoop { Err(e) => return self.response_error(task, e), }; wire.push(( - r.doc_id, - r.surrogate.map(|s| s.as_u32()), + r.key.to_string(), + r.key.surrogate().as_u32(), doc_format::encode_resolved_wire_body(&r.doc), doc_format::encode_resolved_wire_body(&old_doc), )); diff --git a/nodedb/src/data/executor/handlers/update_from_join_collect.rs b/nodedb/src/data/executor/handlers/update_from_join_collect.rs index 6e6382879..d3dfa79d0 100644 --- a/nodedb/src/data/executor/handlers/update_from_join_collect.rs +++ b/nodedb/src/data/executor/handlers/update_from_join_collect.rs @@ -218,8 +218,7 @@ impl CoreLoop { // reindex and the expanded `PointPut`'s identity (RESOLVE path) // both need it, so it's carried through rather than re-parsed. rows.push(ResolvedUpdateRow { - doc_id: key.to_string(), - surrogate: Some(key.surrogate()), + key, body: updated_bytes, old_body: current_bytes, doc: target_doc, diff --git a/nodedb/src/data/executor/handlers/update_from_join_types.rs b/nodedb/src/data/executor/handlers/update_from_join_types.rs index f788fc92c..ff2198cb3 100644 --- a/nodedb/src/data/executor/handlers/update_from_join_types.rs +++ b/nodedb/src/data/executor/handlers/update_from_join_types.rs @@ -4,7 +4,7 @@ //! the write pass and the RESOLVE pass consume, and the operation's //! parameters. -use nodedb_types::Surrogate; +use nodedb_types::StorageKey; use nodedb_physical::physical_plan::{ResolvedSumTarget, ReturningSpec, UpdateValue}; @@ -14,11 +14,9 @@ use nodedb_physical::physical_plan::{ResolvedSumTarget, ReturningSpec, UpdateVal /// classifier so the two cannot diverge on which rows match or what post-image /// each carries. pub(in crate::data::executor) struct ResolvedUpdateRow { - /// Target storage key (hex-encoded surrogate on a surrogate-keyed row). - pub doc_id: String, - /// The row's registered surrogate, parsed from `doc_id`. `None` for a - /// legacy non-surrogate-keyed row. - pub surrogate: Option, + /// Target storage key. `.surrogate()` recovers the numeric surrogate; + /// `.to_identity()` / `Display` recover the client-visible or rendered forms. + pub key: StorageKey, /// Post-image body: strict Binary Tuple for a strict target, MessagePack /// for a schemaless target. pub body: Vec, @@ -47,8 +45,8 @@ pub(in crate::data::executor) struct UpdateFromJoinParams<'a> { /// handler runs the identical scan/join/assignment/encode pipeline as the /// write path but writes NOTHING — no `sparse.put`, no vector re-index, no /// write-set, no events — and returns the matched rows as msgpack - /// `Vec<(doc_id, Option, post_image_body)>` for the expander - /// to rewrite into concrete `PointPut` ops. `false` = the normal write path. + /// `Vec` for the expander to rewrite into concrete + /// `PointPut` ops. `false` = the normal write path. pub resolve_only: bool, /// Control-Plane-shipped source rows for cross-core `UPDATE ... FROM`. When /// `Some`, the source join-map is built from these pre-scanned diff --git a/nodedb/src/data/executor/handlers/update_from_join_write.rs b/nodedb/src/data/executor/handlers/update_from_join_write.rs index f5ee441a4..a1ed6cdae 100644 --- a/nodedb/src/data/executor/handlers/update_from_join_write.rs +++ b/nodedb/src/data/executor/handlers/update_from_join_write.rs @@ -74,34 +74,12 @@ impl CoreLoop { for row in rows { let ResolvedUpdateRow { - doc_id, - surrogate: row_surrogate, + key: storage_key, body: updated_bytes, old_body, mut doc, } = row; - // `put_in_txn` addresses DOCUMENTS rows by `StorageKey` only, and - // the workspace carries no on-disk-format compatibility burden - // for a row shape that predates surrogate keying, so a row whose - // `doc_id` does not parse as one is refused rather than written - // through a raw string key. - let storage_key = match crate::engine::document::store::StorageKey::parse(&doc_id) { - Some(key) => key, - None => { - return Err(self.response_error( - task, - crate::Error::Storage { - engine: "document".into(), - detail: format!( - "UPDATE ... FROM target row '{doc_id}' in \ - '{target_collection}' has no surrogate storage key" - ), - }, - )); - } - }; - // Period lock, both images — matching `execute_point_update`: a // closed period must reject an edit to a row it already holds, // and must reject an edit that assigns the period column into it. @@ -202,7 +180,7 @@ impl CoreLoop { // `collect_update_from_join_rows`; `emit_put_event` derives // `WriteOp::Update` from the Some prior + Some new pair and // handles strict->msgpack conversion on both sides. - let row_identity = crate::engine::document::store::identity_of(&doc_id); + let row_identity = storage_key.to_identity(); // `row_identity` is read again below for `RETURNING`'s `id` // field, so the event-emit boundary gets a clone rather than // the move. @@ -220,7 +198,7 @@ impl CoreLoop { // post-apply `Put` redo (`updated_bytes` is moved as its last // use). Both are no-ops unless the collection has a vector // field, so a non-vector collection pays nothing. - if has_vectors && let Some(surrogate) = row_surrogate { + if has_vectors { if let Err(e) = self.update_reindex_vector_indexes(UpdateVectorReindex { database_id, tid, @@ -233,7 +211,7 @@ impl CoreLoop { return Err(self.response_error(task, e)); } write_set.push(WriteSetEntry { - surrogate: surrogate.as_u32(), + surrogate: storage_key.surrogate().as_u32(), is_delete: false, value: updated_bytes, collection: None, diff --git a/nodedb/src/data/executor/handlers/vector_multi.rs b/nodedb/src/data/executor/handlers/vector_multi.rs index 6fd9fff6d..135e8daac 100644 --- a/nodedb/src/data/executor/handlers/vector_multi.rs +++ b/nodedb/src/data/executor/handlers/vector_multi.rs @@ -298,7 +298,7 @@ impl CoreLoop { let hits: Vec = scored_docs .iter() .map(|(s, score)| super::super::response_codec::VectorSearchHit { - id: s.as_u32(), + id: super::hybrid_key::HybridFusionKey::for_surrogate(*s), distance: *score, doc_id: None, body: None, diff --git a/nodedb/src/data/executor/handlers/vector_search.rs b/nodedb/src/data/executor/handlers/vector_search.rs index 60a401303..01c6e753f 100644 --- a/nodedb/src/data/executor/handlers/vector_search.rs +++ b/nodedb/src/data/executor/handlers/vector_search.rs @@ -2,8 +2,8 @@ //! Vector search parameter types and shared helper functions. //! -//! DP emits each hit's `id` as the bound `Surrogate.as_u32()` (or the -//! local node id if the row is headless / pre-surrogate). `doc_id` is +//! DP emits each hit's `id` as its `HybridFusionKey`: `Bound` when the local +//! HNSW node resolves to a surrogate, `Headless` otherwise. `doc_id` is //! always `None` from DP; the Control Plane fills it via the catalog //! at the response boundary. @@ -16,20 +16,17 @@ use crate::data::executor::task::ExecutionTask; use crate::engine::vector::collection::VectorCollection; use crate::engine::vector::distance::DistanceMetric; -/// Build a search hit from raw search result data. `id` is the bound -/// surrogate when present, else the local node id (so headless rows +/// Build a search hit from raw search result data. `id` is the +/// `HybridFusionKey` the local HNSW node resolves to: `Bound` to its +/// surrogate's storage key when present, else `Headless` (so headless rows /// still round-trip). pub(super) fn build_search_hit( collection: Option<&VectorCollection>, local_id: u32, distance: f32, ) -> super::super::response_codec::VectorSearchHit { - let id = collection - .and_then(|c| c.get_surrogate(local_id)) - .map(|s| s.as_u32()) - .unwrap_or(local_id); super::super::response_codec::VectorSearchHit { - id, + id: vector_leg_key(collection, local_id), distance, doc_id: None, body: None, diff --git a/nodedb/src/data/executor/handlers/vector_search_exec.rs b/nodedb/src/data/executor/handlers/vector_search_exec.rs index 5efb3192f..3ce059ff8 100644 --- a/nodedb/src/data/executor/handlers/vector_search_exec.rs +++ b/nodedb/src/data/executor/handlers/vector_search_exec.rs @@ -13,7 +13,6 @@ use super::vector_search_ann::{ResolvedAnnOptions, apply_ann_options, quantizati use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; -use nodedb_types::Surrogate; /// Parameters for [`CoreLoop::search_ivf`]. struct SearchIvfParams<'a> { @@ -34,7 +33,8 @@ impl CoreLoop { /// (the Control Plane evaluates the predicate against `body`) and by /// the slow-path SELECT (the Control Plane response translator flattens /// the body's fields into the hit JSON so payload columns surface to - /// the client). When `attach == false` the hit is returned unchanged. + /// the client). When `attach == false`, or the hit carries no surrogate + /// binding, the hit is returned unchanged. /// /// The bytes are normalized to a standard msgpack map through the shared /// sparse-body normalizer, resolved from the collection's registered kind. @@ -54,7 +54,9 @@ impl CoreLoop { if !attach { return hit; } - let key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(hit.id)); + let Some(key) = hit.id.storage_key() else { + return hit; + }; if let Ok(Some(bytes)) = self.sparse.get(database_id, tid, collection, &key) { let format = self.sparse_body_format( crate::types::DatabaseId::new(database_id), @@ -408,8 +410,13 @@ impl CoreLoop { .collect(); if let Some(surrogate_bm) = filter_bitmap { - // Bitmap is a set of surrogates; hit.id is now the surrogate. - hits.retain(|h| surrogate_bm.0.contains(h.id)); + // Bitmap is a set of surrogates: keep only bound hits whose + // surrogate is in the bitmap. A headless hit has none, so it + // never survives a surrogate-bitmap filter. + hits.retain(|h| { + h.id.storage_key() + .is_some_and(|key| surrogate_bm.contains(key.surrogate())) + }); } if !rls_filters.is_empty() { // CP-side translator runs the predicate; DP only attaches body. diff --git a/nodedb/src/data/executor/handlers/vector_sparse.rs b/nodedb/src/data/executor/handlers/vector_sparse.rs index 818249c78..d85d7a1ff 100644 --- a/nodedb/src/data/executor/handlers/vector_sparse.rs +++ b/nodedb/src/data/executor/handlers/vector_sparse.rs @@ -144,21 +144,23 @@ impl CoreLoop { let results = crate::engine::vector::sparse::search::dot_product_topk(index, &query, top_k); // Convert to VectorSearchHit. The sparse index keys documents by their - // hex surrogate `doc_id`; emit that surrogate as the hit `id` and leave - // `doc_id` unset, exactly like the dense vector search — the Control-Plane - // response translator (`translate_vector_search_payload`) then resolves - // the surrogate back to the user PK for the projection. Fall back to the - // index's internal id only if a `doc_id` is somehow not a hex surrogate. + // hex surrogate `doc_id`; emit that surrogate's storage key as the hit + // `id` and leave `doc_id` unset, exactly like the dense vector search — + // the Control-Plane response translator (`translate_vector_search_payload`) + // then resolves the surrogate back to the user PK for the projection. + // Fall back to `Headless(internal_id)` only if `doc_id` is somehow not + // a hex surrogate. let hits: Vec = results .iter() .map(|r| { - let surrogate = r + let id = r .doc_id .as_deref() - .and_then(|d| u32::from_str_radix(d, 16).ok()) - .unwrap_or(r.internal_id); + .and_then(nodedb_types::StorageKey::parse) + .map(super::hybrid_key::HybridFusionKey::Bound) + .unwrap_or(super::hybrid_key::HybridFusionKey::Headless(r.internal_id)); super::super::response_codec::VectorSearchHit { - id: surrogate, + id, distance: r.score, doc_id: None, body: None, diff --git a/nodedb/src/data/executor/response_codec/hits.rs b/nodedb/src/data/executor/response_codec/hits.rs index ff982bb66..a93312f83 100644 --- a/nodedb/src/data/executor/response_codec/hits.rs +++ b/nodedb/src/data/executor/response_codec/hits.rs @@ -11,10 +11,12 @@ use serde::Serialize; +use crate::data::executor::handlers::hybrid_key::HybridFusionKey; + #[derive(Serialize, zerompk::ToMessagePack, zerompk::FromMessagePack)] #[msgpack(map)] pub(in crate::data::executor) struct VectorSearchHit { - pub id: u32, + pub id: HybridFusionKey, pub distance: f32, #[serde(skip_serializing_if = "Option::is_none")] pub doc_id: Option, @@ -194,13 +196,13 @@ mod tests { fn encode_vector_hits() { let hits = vec![ VectorSearchHit { - id: 1, + id: HybridFusionKey::for_surrogate(nodedb_types::Surrogate::new(1)), distance: 0.5, doc_id: None, body: None, }, VectorSearchHit { - id: 2, + id: HybridFusionKey::for_surrogate(nodedb_types::Surrogate::new(2)), distance: 0.8, doc_id: None, body: None, diff --git a/nodedb/src/data/executor/response_codec/raw.rs b/nodedb/src/data/executor/response_codec/raw.rs index cf5727ea7..253c6f526 100644 --- a/nodedb/src/data/executor/response_codec/raw.rs +++ b/nodedb/src/data/executor/response_codec/raw.rs @@ -139,14 +139,16 @@ pub fn flatten_to_relational_rows(bytes: &[u8]) -> Vec { /// Flatten a gathered array of vector-family search hits (dense / sparse / /// multi-vector) into bare relational rows for the post-processing tail. /// -/// A hit is `{id: , distance, doc_id?, body?: }`. -/// The document `body`'s columns become top-level (so ORDER BY / DISTINCT / -/// projection can reference any document column), `distance` is surfaced, and -/// `_surrogate` carries the internal id. The output `id` is, in order: the -/// document body's own `id`; else the user PK from `resolve_pk(surrogate)`; -/// else the `doc_id`; else the raw surrogate. `resolve_pk` lets a body-less hit -/// (sparse / multi-vector, or a `skip_payload_fetch` dense hit) still surface -/// the user PK — mirroring the Control-Plane response translator. +/// A hit is `{id: , distance, doc_id?, +/// body?: }`. The document `body`'s columns become top-level (so +/// ORDER BY / DISTINCT / projection can reference any document column), +/// `distance` is surfaced, and `_surrogate` carries the internal id (or, for +/// a headless hit with no surrogate binding, the `__local_` sentinel +/// text). The output `id` is, in order: the document body's own `id`; else +/// the user PK from `resolve_pk(surrogate)`; else the `doc_id`; else the raw +/// surrogate (or sentinel). `resolve_pk` lets a body-less hit (sparse / +/// multi-vector, or a `skip_payload_fetch` dense hit) still surface the user +/// PK — mirroring the Control-Plane response translator. /// /// The body is *bare* msgpack (the storage wire shape), so it is decoded and /// re-encoded with the native `Value` codec (`value_from_msgpack` / @@ -157,12 +159,13 @@ pub fn flatten_vector_hits_to_relational_rows( bytes: &[u8], resolve_pk: impl Fn(u32) -> Option, ) -> Vec { + use crate::data::executor::handlers::hybrid_key::HybridFusionKey; use nodedb_types::Value; #[derive(zerompk::FromMessagePack)] #[msgpack(map)] struct Hit { - id: u32, + id: HybridFusionKey, distance: f32, doc_id: Option, body: Option>, @@ -188,21 +191,28 @@ pub fn flatten_vector_hits_to_relational_rows( fields .entry("distance".to_string()) .or_insert(Value::Float(h.distance as f64)); + // `None` for a headless hit (no surrogate binding) — never + // resolvable to a user PK. + let surrogate = h.id.storage_key().map(|k| k.surrogate().as_u32()); + let raw_identity = match surrogate { + Some(s) => Value::Integer(s as i64), + None => Value::String(h.id.to_string()), + }; if !fields.contains_key("id") { // Prefer a catalog-resolved PK, then the DP-set doc_id, then the - // raw surrogate as a last resort. - match resolve_pk(h.id).or(h.doc_id) { + // raw surrogate (or the headless sentinel) as a last resort. + match surrogate.and_then(&resolve_pk).or(h.doc_id) { Some(pk) => { fields.insert("id".to_string(), Value::String(pk)); } None => { - fields.insert("id".to_string(), Value::Integer(h.id as i64)); + fields.insert("id".to_string(), raw_identity.clone()); } } } fields .entry("_surrogate".to_string()) - .or_insert(Value::Integer(h.id as i64)); + .or_insert(raw_identity); nodedb_types::value_to_msgpack(&Value::Object(fields)).ok() }) .collect(); @@ -374,15 +384,23 @@ mod tests { } /// A vector-family hit in the Data-Plane wire shape (bare `#[msgpack(map)]`). + /// `id` is the `HybridFusionKey` wire string: an 8-hex-char storage key + /// for a bound hit, matching `test_storage_key`. #[derive(zerompk::ToMessagePack)] #[msgpack(map)] struct TestHit { - id: u32, + id: String, distance: f32, doc_id: Option, body: Option>, } + /// Render surrogate `n` as the 8-hex-char storage key text a bound + /// `HybridFusionKey` carries on the wire. + fn test_storage_key(n: u32) -> String { + nodedb_types::StorageKey::for_surrogate(nodedb_types::Surrogate::new(n)).to_string() + } + #[test] fn vector_hits_merge_body_and_surface_metadata() { let body = bare_doc(vec![ @@ -390,7 +408,7 @@ mod tests { ("tag", Value::String("keep".into())), ]); let input = zerompk::to_msgpack_vec(&vec![TestHit { - id: 7, + id: test_storage_key(7), distance: 0.5, doc_id: None, body: Some(body), @@ -412,7 +430,7 @@ mod tests { // Sparse / multivec hit: no body → `id` comes from the PK resolver, not // the raw surrogate. let input = zerompk::to_msgpack_vec(&vec![TestHit { - id: 42, + id: test_storage_key(42), distance: 1.25, doc_id: None, body: None, @@ -431,7 +449,7 @@ mod tests { #[test] fn body_less_vector_hit_falls_back_to_surrogate_when_unresolved() { let input = zerompk::to_msgpack_vec(&vec![TestHit { - id: 9, + id: test_storage_key(9), distance: 0.0, doc_id: None, body: None, diff --git a/nodedb/src/data/executor/wal_replay_document_vector.rs b/nodedb/src/data/executor/wal_replay_document_vector.rs index bb39a2235..c6166e809 100644 --- a/nodedb/src/data/executor/wal_replay_document_vector.rs +++ b/nodedb/src/data/executor/wal_replay_document_vector.rs @@ -167,7 +167,6 @@ impl CoreLoop { tid: tenant_id, collection: &collection, storage_key, - surrogate, value: &value, wal_lsn: record_lsn, }, diff --git a/nodedb/src/query/resolved_update_row.rs b/nodedb/src/query/resolved_update_row.rs index c59d6c4d1..c3a90cfb9 100644 --- a/nodedb/src/query/resolved_update_row.rs +++ b/nodedb/src/query/resolved_update_row.rs @@ -9,5 +9,6 @@ /// Both images travel — a materialized sum folds a delta from the pair, and /// a rewritten join key moves value between two targets, neither derivable /// from the post-image alone. Bodies are schemaless wire form, never a -/// stored Binary Tuple, or the write path would double-encode them. -pub type ResolvedUpdateRowWire = (String, Option, Vec, Vec); +/// stored Binary Tuple, or the write path would double-encode them. The +/// surrogate is never absent: every matched row is a storage-keyed row. +pub type ResolvedUpdateRowWire = (String, u32, Vec, Vec); diff --git a/nodedb/tests/inproc/cases/cross_engine_bitmap_currency.rs b/nodedb/tests/inproc/cases/cross_engine_bitmap_currency.rs index 7be2d6d08..725b4ba88 100644 --- a/nodedb/tests/inproc/cases/cross_engine_bitmap_currency.rs +++ b/nodedb/tests/inproc/cases/cross_engine_bitmap_currency.rs @@ -109,13 +109,14 @@ fn parse_json(payload: &[u8]) -> serde_json::Value { } /// Extract surrogate u32 values from a vector search response payload. -/// Vector results encode `id` as the raw surrogate u32. +/// A bound vector hit encodes `id` as its storage key. fn extract_vector_surrogates(payload: &[u8]) -> Vec { let v = parse_json(payload); v.as_array() .unwrap_or(&vec![]) .iter() - .filter_map(|hit| hit.get("id").and_then(|id| id.as_u64()).map(|n| n as u32)) + .filter_map(|hit| hit.get("id").and_then(|id| id.as_str())) + .filter_map(|key| nodedb_types::StorageKey::parse(key).map(|k| k.surrogate().as_u32())) .collect() } diff --git a/nodedb/tests/inproc/cases/cross_engine_three_way_fts_vector_doc.rs b/nodedb/tests/inproc/cases/cross_engine_three_way_fts_vector_doc.rs index fddc1d875..e8c9d7c87 100644 --- a/nodedb/tests/inproc/cases/cross_engine_three_way_fts_vector_doc.rs +++ b/nodedb/tests/inproc/cases/cross_engine_three_way_fts_vector_doc.rs @@ -112,13 +112,14 @@ fn parse_json(payload: &[u8]) -> serde_json::Value { } /// Extract surrogate u32 values from a vector search response. -/// Vector hits encode the surrogate as `id: u32`. +/// A bound vector hit encodes `id` as its storage key. fn extract_vector_surrogates(payload: &[u8]) -> Vec { parse_json(payload) .as_array() .unwrap_or(&vec![]) .iter() - .filter_map(|h| h.get("id").and_then(|v| v.as_u64()).map(|n| n as u32)) + .filter_map(|h| h.get("id").and_then(|v| v.as_str())) + .filter_map(|key| nodedb_types::StorageKey::parse(key).map(|k| k.surrogate().as_u32())) .collect() } diff --git a/nodedb/tests/inproc/cases/surrogate_round_trip.rs b/nodedb/tests/inproc/cases/surrogate_round_trip.rs index 9527e3919..5733ad9a3 100644 --- a/nodedb/tests/inproc/cases/surrogate_round_trip.rs +++ b/nodedb/tests/inproc/cases/surrogate_round_trip.rs @@ -154,7 +154,8 @@ fn extract_vector_surrogates(payload: &[u8]) -> Vec { .as_array() .unwrap_or(&vec![]) .iter() - .filter_map(|h| h.get("id").and_then(|v| v.as_u64()).map(|n| n as u32)) + .filter_map(|h| h.get("id").and_then(|v| v.as_str())) + .filter_map(|key| nodedb_types::StorageKey::parse(key).map(|k| k.surrogate().as_u32())) .collect() } From c0e41e1fa85e924cf9cf62ca1971e66efdca6c9b Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 12 Sep 2026 19:30:46 +0800 Subject: [PATCH 16/17] refactor(events): build RowId from declared identity, not raw doc id MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Row is now an enum (Row/Batch/Edge/Heartbeat) carrying the same RowIdentity a row's INSERT minted, instead of an ad hoc Arc. Edge rows compose their src/label/dst text once at construction so the event ring pays for one identity, not four strings. Threads a declared_primary_key column through the write paths that build events and redo entries — TRUNCATE, bulk DELETE/UPDATE, UPDATE FROM, and the materialized-sum resolve pass — so a row emitted from these paths reports its declared PRIMARY KEY value, or the decimal surrogate when the collection has none, matching what live writes and WAL replay already emit. Updates the executor, WAL replication encode/decode, event plane consumers, CDC router, and CRDT sync packager to construct and read the new RowId shape, and updates cluster and in-process tests to match. --- .../misc_suite/cases/cluster_triggers.rs | 2 +- .../src/physical_plan/document/op.rs | 8 ++ .../src/physical_plan/document/types.rs | 5 + nodedb/src/bridge/envelope/response.rs | 6 + .../calvin/scheduler/driver/core/routing.rs | 2 + nodedb/src/control/event_trigger.rs | 4 +- .../control/planner/calvin/tx_class/shared.rs | 1 + .../src/control/planner/calvin/write_class.rs | 2 + .../control/planner/materialized_sum/index.rs | 3 + .../planner/materialized_sum/settle.rs | 2 + .../control/planner/rls_injection/document.rs | 1 + .../src/control/planner/rls_injection/plan.rs | 1 + .../planner/sql_plan_convert/set_ops.rs | 2 + .../dispatch_utils/durability_barrier.rs | 1 + .../native/dispatch/plan_builder/document.rs | 2 + .../ddl/neutral/collection/enforcement.rs | 3 + .../predicate/txn_buffering/classify.rs | 1 + .../control/server/wal_dispatch/document.rs | 16 ++- .../server/wal_dispatch/write_set_redo.rs | 68 ++++++++-- .../wal_replication/decode/document.rs | 38 +++--- .../wal_replication/decode/entry_document.rs | 12 +- .../wal_replication/encode/document.rs | 29 ++--- .../wal_replication/encode/entry_document.rs | 20 ++- .../src/control/wal_replication/types/mod.rs | 4 +- .../wal_replication/types/replicated_write.rs | 7 + .../wal_replication/types/wire_shapes.rs | 22 ++++ .../src/data/executor/core_loop/deferred.rs | 2 +- .../src/data/executor/core_loop/event_emit.rs | 41 ++++-- nodedb/src/data/executor/dispatch/document.rs | 20 ++- .../enforcement/materialized_sum/apply.rs | 22 +++- .../enforcement/materialized_sum/delta.rs | 1 + .../enforcement/materialized_sum/rmw.rs | 11 +- .../data/executor/enforcement/write_hook.rs | 1 + .../data/executor/handlers/bulk_dml/delete.rs | 6 + .../handlers/bulk_dml/delete_cascade.rs | 23 +++- .../data/executor/handlers/bulk_dml/update.rs | 13 +- .../handlers/control/crdt_apply/gated.rs | 1 + .../handlers/control/crdt_apply/local.rs | 1 + .../executor/handlers/control/crdt_doc.rs | 15 ++- .../handlers/control/crdt_materialize.rs | 51 ++++++-- .../handlers/document/apply_balance_delta.rs | 5 + .../handlers/document/resolve/apply.rs | 3 +- .../handlers/document/resolve/apply_row.rs | 13 +- .../handlers/document/write/batch_insert.rs | 28 ++-- .../merge_orchestrated/apply/insert_rows.rs | 16 ++- .../merge_orchestrated/apply/orchestrate.rs | 6 +- .../merge_orchestrated/apply/update_rows.rs | 14 +- .../merge_orchestrated/apply_support.rs | 10 +- .../merge_orchestrated/delete_arms.rs | 12 +- .../data/executor/handlers/point/delete.rs | 1 + .../data/executor/handlers/point/insert.rs | 5 +- .../src/data/executor/handlers/point/put.rs | 5 +- .../executor/handlers/point/update/exec.rs | 4 +- .../executor/handlers/transaction/batch.rs | 14 +- .../transaction/index_write_values.rs | 2 + .../transaction/sub_plan_doc/delete.rs | 3 + .../handlers/transaction/sub_plan_doc/put.rs | 3 + .../handlers/transaction/undo/document.rs | 13 ++ .../handlers/transaction/undo/entry.rs | 4 + .../handlers/transaction/undo/rollback.rs | 2 + nodedb/src/data/executor/handlers/truncate.rs | 50 +++++++- .../executor/handlers/update_from_join.rs | 3 +- .../handlers/update_from_join_write.rs | 23 +++- .../executor/handlers/upsert/exec/dispatch.rs | 1 + .../executor/handlers/upsert/exec/insert.rs | 8 +- .../handlers/upsert/exec/overwrite.rs | 8 +- .../src/data/executor/handlers/write_batch.rs | 11 +- nodedb/src/data/executor/wal_replay/crdt.rs | 3 +- nodedb/src/event/audit_dml/consumer.rs | 2 +- nodedb/src/event/bus.rs | 2 +- nodedb/src/event/cdc/router.rs | 4 +- nodedb/src/event/consumer.rs | 2 +- nodedb/src/event/consumer_helpers.rs | 2 +- nodedb/src/event/crdt_sync/packager.rs | 2 +- nodedb/src/event/plane.rs | 2 +- nodedb/src/event/types.rs | 120 ++++++++++++++++-- nodedb/src/event/wal_replay.rs | 24 ++++ nodedb/src/event/wal_replay_parse.rs | 34 +++-- nodedb/src/query/materialized_sum_delta.rs | 1 + nodedb/src/query/materialized_sum_images.rs | 1 + nodedb/src/query/materialized_sum_keys.rs | 1 + nodedb/tests/inproc/cases/bitemporal_cdc.rs | 2 +- nodedb/tests/inproc/cases/cdc_arc_fanout.rs | 2 +- nodedb/tests/inproc/cases/event_trigger.rs | 2 +- .../inproc/cases/shutdown_event_plane.rs | 2 +- 85 files changed, 737 insertions(+), 208 deletions(-) diff --git a/nodedb-cluster-tests/tests/misc_suite/cases/cluster_triggers.rs b/nodedb-cluster-tests/tests/misc_suite/cases/cluster_triggers.rs index c09b8cc24..799b46536 100644 --- a/nodedb-cluster-tests/tests/misc_suite/cases/cluster_triggers.rs +++ b/nodedb-cluster-tests/tests/misc_suite/cases/cluster_triggers.rs @@ -181,7 +181,7 @@ fn event_source_preserved_through_write_event() { database_id: nodedb::types::DatabaseId::DEFAULT, collection: Arc::from("orders"), op: WriteOp::Insert, - row_id: RowId::new("doc-1"), + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("doc-1")), lsn: Lsn::new(100), tenant_id: TenantId::new(1), vshard_id: VShardId::new(0), diff --git a/nodedb-physical/src/physical_plan/document/op.rs b/nodedb-physical/src/physical_plan/document/op.rs index 3cccfa423..cf978f848 100644 --- a/nodedb-physical/src/physical_plan/document/op.rs +++ b/nodedb-physical/src/physical_plan/document/op.rs @@ -328,6 +328,10 @@ pub enum DocumentOp { /// divergence, before writing. #[serde(default)] resolved_sum_targets: Vec, + /// See `PointUpdate::declared_primary_key`. Names the column each + /// removed row's identity is read from for its event and redo entry. + #[serde(default)] + declared_primary_key: Option, }, /// Estimate count via HLL cardinality stats. @@ -561,6 +565,10 @@ pub enum DocumentOp { join_column: String, /// Join value that resolved to `surrogate`. join_value: String, + /// The TARGET collection's declared `PRIMARY KEY` column, when it + /// has one. Names the target row in its event and redo entry. + #[serde(default)] + declared_primary_key: Option, }, /// Read-only resolve pass over the wrapped write op: runs its full diff --git a/nodedb-physical/src/physical_plan/document/types.rs b/nodedb-physical/src/physical_plan/document/types.rs index 8c615b0f7..9b8f7df7a 100644 --- a/nodedb-physical/src/physical_plan/document/types.rs +++ b/nodedb-physical/src/physical_plan/document/types.rs @@ -150,6 +150,11 @@ pub struct MaterializedSumBinding { pub join_column: String, /// Expression evaluated against the source INSERT row to compute the delta. pub value_expr: nodedb_query::expr::SqlExpr, + /// The TARGET collection's declared `PRIMARY KEY` column, when it has + /// one. Resolved from the catalog at plan time. Names the target row in + /// the event and redo entry its balance write produces. + #[serde(default)] + pub declared_primary_key: Option, } /// Period lock configuration propagated to Data Plane. diff --git a/nodedb/src/bridge/envelope/response.rs b/nodedb/src/bridge/envelope/response.rs index 2f09d9bed..cafd7fbce 100644 --- a/nodedb/src/bridge/envelope/response.rs +++ b/nodedb/src/bridge/envelope/response.rs @@ -6,6 +6,7 @@ use super::error_code::ErrorCode; use super::payload::Payload; use super::status::Status; use crate::types::{Lsn, RequestId}; +use nodedb_types::RowIdentity; /// One row-level effect of an applied write, carried back from the Data Plane /// so the Control Plane can mint a durable redo record *after* apply. @@ -19,6 +20,11 @@ use crate::types::{Lsn, RequestId}; pub struct WriteSetEntry { /// The row's stable global surrogate. pub surrogate: u32, + /// The row's client identity, by the rule INSERT mints it with. + /// + /// The redo record journals this text as its `document_id`, so a WAL + /// replay names the same row a live event names. + pub identity: RowIdentity, /// `true` for a delete effect (no body), `false` for a put (post-image in /// `value`). pub is_delete: bool, diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs index 10a2e2522..5ad829365 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs @@ -547,6 +547,7 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "docs"), restart_identity: false, resolved_sum_targets: Vec::new(), + declared_primary_key: None, }); let want = collection_vshard("docs").as_u32(); assert_eq!(vshards_of(&plan), vec![want]); @@ -565,6 +566,7 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, &collection), restart_identity: false, resolved_sum_targets: Vec::new(), + declared_primary_key: None, }); let expected = collection_vshard_in_database(DatabaseId::new(7), &collection); match plan_vshard_in_database(&plan, DatabaseId::new(7)) { diff --git a/nodedb/src/control/event_trigger.rs b/nodedb/src/control/event_trigger.rs index 217e58a09..b9347ce83 100644 --- a/nodedb/src/control/event_trigger.rs +++ b/nodedb/src/control/event_trigger.rs @@ -409,7 +409,9 @@ mod tests { sequence: 1, collection: Arc::from("odd\"; DROP TABLE audit; --"), op: WriteOp::Insert, - row_id: RowId::new("doc'; DELETE FROM audit; --"), + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key( + "doc'; DELETE FROM audit; --", + )), lsn: Lsn::new(1), database_id: DatabaseId::DEFAULT, tenant_id: TenantId::new(1), diff --git a/nodedb/src/control/planner/calvin/tx_class/shared.rs b/nodedb/src/control/planner/calvin/tx_class/shared.rs index f7b67133c..16c222ad3 100644 --- a/nodedb/src/control/planner/calvin/tx_class/shared.rs +++ b/nodedb/src/control/planner/calvin/tx_class/shared.rs @@ -494,6 +494,7 @@ mod routing_agreement_tests { delta: "25".to_owned(), join_column: "account_id".to_owned(), join_value: "acc-1".to_owned(), + declared_primary_key: None, }) } diff --git a/nodedb/src/control/planner/calvin/write_class.rs b/nodedb/src/control/planner/calvin/write_class.rs index 948e44896..b26888129 100644 --- a/nodedb/src/control/planner/calvin/write_class.rs +++ b/nodedb/src/control/planner/calvin/write_class.rs @@ -390,6 +390,7 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "docs"), restart_identity: false, resolved_sum_targets: Vec::new(), + declared_primary_key: None, }); assert!(is_write_plan(&plan), "DocumentOp::Truncate must be a write"); } @@ -609,6 +610,7 @@ mod tests { delta: "25".to_owned(), join_column: "account_id".to_owned(), join_value: "acc-1".to_owned(), + declared_primary_key: None, }) } diff --git a/nodedb/src/control/planner/materialized_sum/index.rs b/nodedb/src/control/planner/materialized_sum/index.rs index 16635e8d3..dcd25ded8 100644 --- a/nodedb/src/control/planner/materialized_sum/index.rs +++ b/nodedb/src/control/planner/materialized_sum/index.rs @@ -115,6 +115,9 @@ impl MaterializedSumIndex { target_column: def.target_column.clone(), join_column: def.join_column.clone(), value_expr: def.value_expr.clone(), + // `target` is the collection the binding writes into, + // so its declared key names the target row. + declared_primary_key: target.declared_primary_key.clone(), }); } } diff --git a/nodedb/src/control/planner/materialized_sum/settle.rs b/nodedb/src/control/planner/materialized_sum/settle.rs index 2fbf2521a..28d74dcf5 100644 --- a/nodedb/src/control/planner/materialized_sum/settle.rs +++ b/nodedb/src/control/planner/materialized_sum/settle.rs @@ -240,6 +240,7 @@ pub(super) fn balance_task(spec: BalanceTaskSpec<'_>) -> PhysicalTask { delta: spec.delta.to_string(), join_column: spec.binding.join_column.clone(), join_value: spec.join_value, + declared_primary_key: spec.binding.declared_primary_key.clone(), }), post_set_op: PostSetOp::None, txn_id: spec.txn_id, @@ -360,6 +361,7 @@ mod tests { target_column: "balance".to_string(), join_column: "account_id".to_string(), value_expr: nodedb_query::expr::SqlExpr::Column("amount".to_string()), + declared_primary_key: None, } } diff --git a/nodedb/src/control/planner/rls_injection/document.rs b/nodedb/src/control/planner/rls_injection/document.rs index 7593de0a4..dde01115f 100644 --- a/nodedb/src/control/planner/rls_injection/document.rs +++ b/nodedb/src/control/planner/rls_injection/document.rs @@ -469,6 +469,7 @@ mod tests { ), restart_identity: false, resolved_sum_targets: Vec::new(), + declared_primary_key: None, }); assert_write_refused(inject(&mut plan, &store), "orders"); } diff --git a/nodedb/src/control/planner/rls_injection/plan.rs b/nodedb/src/control/planner/rls_injection/plan.rs index 0df53ece6..57cf47739 100644 --- a/nodedb/src/control/planner/rls_injection/plan.rs +++ b/nodedb/src/control/planner/rls_injection/plan.rs @@ -430,6 +430,7 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "docs"), restart_identity: false, resolved_sum_targets: Vec::new(), + declared_primary_key: None, }); let before = plan.clone(); assert!(inject(&mut plan, &store).is_ok()); diff --git a/nodedb/src/control/planner/sql_plan_convert/set_ops.rs b/nodedb/src/control/planner/sql_plan_convert/set_ops.rs index 4493c7f02..a9d435fe6 100644 --- a/nodedb/src/control/planner/sql_plan_convert/set_ops.rs +++ b/nodedb/src/control/planner/sql_plan_convert/set_ops.rs @@ -71,6 +71,8 @@ pub(super) fn convert_truncate( // Filled in by the materialized-sum resolution pass, which recon- // scans the rows this TRUNCATE will remove. resolved_sum_targets: Vec::new(), + // Names the column each removed row's identity is read from. + declared_primary_key: super::dml::declared_primary_key_name(ctx, collection)?, }), post_set_op: PostSetOp::None, txn_id: None, diff --git a/nodedb/src/control/server/dispatch_utils/durability_barrier.rs b/nodedb/src/control/server/dispatch_utils/durability_barrier.rs index 2d78b51a3..8a8de71b1 100644 --- a/nodedb/src/control/server/dispatch_utils/durability_barrier.rs +++ b/nodedb/src/control/server/dispatch_utils/durability_barrier.rs @@ -157,6 +157,7 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), restart_identity: false, resolved_sum_targets: Vec::new(), + declared_primary_key: None, }); assert!(plan_is_write(&plan)); assert_eq!(funnel_minted_redo_engine(&plan), None); diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/document.rs b/nodedb/src/control/server/native/dispatch/plan_builder/document.rs index d70e66853..1a6760a78 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/document.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/document.rs @@ -429,6 +429,8 @@ pub(crate) fn build_truncate( restart_identity: false, // Filled in by the materialized-sum resolution pass. resolved_sum_targets: Vec::new(), + // See `build_update`: reads the declared PRIMARY KEY from the catalog. + declared_primary_key: declared_primary_key(ctx, collection)?, })) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/enforcement.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/enforcement.rs index ab88d6d67..05273014c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/enforcement.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/enforcement.rs @@ -353,6 +353,9 @@ pub fn find_materialized_sum_bindings( target_column: def.target_column.clone(), join_column: def.join_column.clone(), value_expr: def.value_expr.clone(), + // `target_coll` is the collection the binding writes into, + // so its declared key names the target row. + declared_primary_key: target_coll.declared_primary_key.clone(), }); } } diff --git a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs index 57f421195..29c9350d4 100644 --- a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs +++ b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs @@ -2037,6 +2037,7 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), restart_identity: false, resolved_sum_targets: Vec::new(), + declared_primary_key: None, }), PhysicalPlan::Kv(KvOp::Truncate { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), diff --git a/nodedb/src/control/server/wal_dispatch/document.rs b/nodedb/src/control/server/wal_dispatch/document.rs index 088b993df..b5d070cea 100644 --- a/nodedb/src/control/server/wal_dispatch/document.rs +++ b/nodedb/src/control/server/wal_dispatch/document.rs @@ -11,6 +11,9 @@ use crate::wal::manager::WalManager; /// Encode a document PUT redo record: `(collection, document_id, value, /// Option, surrogate)`. Must match `wal_replay_redo_document`'s decode. +/// +/// `document_id` is the row's client identity. The Event Plane replay reads +/// it back verbatim as the event's row id. pub(crate) fn encode_document_put_record( collection: &str, document_id: &str, @@ -28,6 +31,8 @@ pub(crate) fn encode_document_put_record( /// Encode a document DELETE redo record: `(collection, document_id, /// Option, surrogate)` — surrogate keys the redb storage row. +/// +/// `document_id` is the row's client identity, as for the PUT record. pub(crate) fn encode_document_delete_record( collection: &str, document_id: &str, @@ -66,7 +71,7 @@ pub(super) fn wal_append_document_op( } => { let entry = encode_document_put_record( collection.as_str(), - document_id, + document_id.as_str(), value, surrogate.as_u32(), )?; @@ -86,7 +91,7 @@ pub(super) fn wal_append_document_op( } => { let entry = encode_document_put_record( collection.as_str(), - document_id, + document_id.as_str(), value, surrogate.as_u32(), )?; @@ -100,8 +105,11 @@ pub(super) fn wal_append_document_op( } => { // 4-tuple keys secondary vector-index removal by surrogate on restart — // a 3-tuple would leave the deleted embedding to resurrect. - let entry = - encode_document_delete_record(collection.as_str(), document_id, surrogate.as_u32())?; + let entry = encode_document_delete_record( + collection.as_str(), + document_id.as_str(), + surrogate.as_u32(), + )?; Some(wal.append_delete(tenant_id, vshard_id, database_id, &entry)?) } // NotAWrite — reads / query ops / DDL that produces no engine mutation here diff --git a/nodedb/src/control/server/wal_dispatch/write_set_redo.rs b/nodedb/src/control/server/wal_dispatch/write_set_redo.rs index 298eff040..7381f11be 100644 --- a/nodedb/src/control/server/wal_dispatch/write_set_redo.rs +++ b/nodedb/src/control/server/wal_dispatch/write_set_redo.rs @@ -4,15 +4,13 @@ //! //! Some write handlers mint no autocommit WAL redo of their own, so a //! vector-index rebuild at startup would resurrect a stale embedding. The Data -//! Plane carries surrogate + post-image in [`Response::write_set`]; the -//! Control Plane mints the durable redo here. +//! Plane carries surrogate, client identity, and post-image in +//! [`Response::write_set`]; the Control Plane mints the durable redo here. use crate::bridge::envelope::{PhysicalPlan, Response, Status, WriteSetEntry}; -use crate::engine::document::store::StorageKey; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; use crate::wal::manager::WalManager; use nodedb_physical::physical_plan::DocumentOp; -use nodedb_types::Surrogate; use super::document::{encode_document_delete_record, encode_document_put_record}; @@ -54,8 +52,10 @@ pub fn plan_post_apply_redo(plan: &PhysicalPlan) -> Option { } /// Append a document redo record for each write-set entry, returning the last -/// allocated LSN. Each entry is keyed by `StorageKey::for_surrogate(surrogate)` -/// so replay keys on the same identity. Called under the write-admission guard. +/// allocated LSN. Each record journals `entry.identity` as its `document_id` +/// and `entry.surrogate` in the `u32` slot, so the Event Plane replay names +/// the row the way a live event does and the Data Plane replay keys on the +/// surrogate. Called under the write-admission guard. pub fn append_write_set_redo( wal: &WalManager, tenant_id: TenantId, @@ -67,7 +67,6 @@ pub fn append_write_set_redo( let mut last: Option = None; for entry in write_set { let entry_collection = entry.collection.as_deref().unwrap_or(collection); - let doc_id = StorageKey::for_surrogate(Surrogate::new(entry.surrogate)).to_string(); // A cross-collection entry homes to a different vShard, so it's re-derived // per entry rather than reusing the caller-hoisted `vshard_id`. let entry_vshard_id = match &entry.collection { @@ -75,12 +74,16 @@ pub fn append_write_set_redo( None => vshard_id, }; let lsn = if entry.is_delete { - let record = encode_document_delete_record(entry_collection, &doc_id, entry.surrogate)?; + let record = encode_document_delete_record( + entry_collection, + entry.identity.as_str(), + entry.surrogate, + )?; wal.append_delete(tenant_id, entry_vshard_id, database_id, &record)? } else { let record = encode_document_put_record( entry_collection, - &doc_id, + entry.identity.as_str(), &entry.value, entry.surrogate, )?; @@ -119,8 +122,8 @@ pub fn mint_dispatch_local_redo( mod tests { use super::*; use nodedb_physical::physical_plan::ReturningSpec; - use nodedb_types::QualifiedCollection; use nodedb_types::sync::wire::SyncProvenance; + use nodedb_types::{QualifiedCollection, RowIdentity, Surrogate}; fn open_wal(dir: &std::path::Path) -> WalManager { WalManager::open_for_testing(&dir.join("test.wal")).expect("open wal") @@ -245,6 +248,7 @@ mod tests { let wal = open_wal(dir.path()); let entries = vec![WriteSetEntry { surrogate: 9, + identity: RowIdentity::for_surrogate(Surrogate::new(9)), is_delete: false, value: vec![1, 2, 3], collection: None, @@ -273,8 +277,8 @@ mod tests { .expect("decode put payload"); assert_eq!(collection, "docs"); assert_eq!( - document_id, - StorageKey::for_surrogate(Surrogate::new(9)).to_string() + document_id, "9", + "a minted row journals its decimal surrogate" ); assert_eq!(value, vec![1, 2, 3]); assert_eq!(surrogate, 9); @@ -286,6 +290,7 @@ mod tests { let wal = open_wal(dir.path()); let entries = vec![WriteSetEntry { surrogate: 9, + identity: RowIdentity::for_surrogate(Surrogate::new(9)), is_delete: true, value: Vec::new(), collection: None, @@ -307,12 +312,47 @@ mod tests { .expect("decode delete payload"); assert_eq!(collection, "docs"); assert_eq!( - document_id, - StorageKey::for_surrogate(Surrogate::new(9)).to_string() + document_id, "9", + "a minted row journals its decimal surrogate" ); assert_eq!(surrogate, 9); } + #[test] + fn write_set_journals_declared_pk_identity_verbatim() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(dir.path()); + let entries = vec![WriteSetEntry { + surrogate: 9, + identity: RowIdentity::from_user_key("order-1"), + is_delete: false, + value: vec![1, 2, 3], + collection: None, + }]; + + append_write_set_redo( + &wal, + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + "docs", + &entries, + ) + .expect("append"); + + let record = last_record_of_type(&wal, nodedb_wal::record::RecordType::Put); + let (_collection, document_id, _value, _prov, surrogate) = + zerompk::from_msgpack::<(String, String, Vec, Option, u32)>( + &record.payload, + ) + .expect("decode put payload"); + assert_eq!( + document_id, "order-1", + "declared PK is journaled as written" + ); + assert_eq!(surrogate, 9, "the surrogate slot still keys storage replay"); + } + #[test] fn empty_write_set_appends_nothing() { let dir = tempfile::tempdir().expect("tempdir"); diff --git a/nodedb/src/control/wal_replication/decode/document.rs b/nodedb/src/control/wal_replication/decode/document.rs index 9a64ce420..77e8ea0ee 100644 --- a/nodedb/src/control/wal_replication/decode/document.rs +++ b/nodedb/src/control/wal_replication/decode/document.rs @@ -8,7 +8,7 @@ use super::ctx::{DecodeCtx, bind_or_lookup}; use crate::bridge::envelope::PhysicalPlan; -use crate::control::wal_replication::types::ReplicatedSumTarget; +use crate::control::wal_replication::types::{BalanceDeltaFields, ReplicatedSumTarget}; use nodedb_physical::physical_plan::{DocumentOp, ResolvedSumTarget, ReturningSpec, UpdateValue}; /// A decoded RETURNING projection spec plus the read filters gating it. @@ -368,12 +368,14 @@ pub(super) fn truncate( collection: &str, restart_identity: bool, resolved_sum_targets: &WireSumResolution<'_>, + declared_primary_key: Option, ) -> PhysicalPlan { PhysicalPlan::Document(DocumentOp::Truncate { collection: nodedb_types::QualifiedCollection::from_stored(collection.to_owned()), restart_identity, // Read off the record — see this module's doc. resolved_sum_targets: plan_targets(resolved_sum_targets), + declared_primary_key, }) } @@ -399,23 +401,16 @@ pub(super) fn insert_select( /// Reconstruct an `ApplyBalanceDelta` plan. No surrogate binding: the document /// id here IS the hex surrogate, not a primary key. Idempotent like `KvIncr`. -pub(super) fn apply_balance_delta( - collection: &str, - document_id: &str, - surrogate: u32, - column: &str, - delta: &str, - join_column: &str, - join_value: &str, -) -> PhysicalPlan { +pub(super) fn apply_balance_delta(fields: BalanceDeltaFields<'_>) -> PhysicalPlan { PhysicalPlan::Document(DocumentOp::ApplyBalanceDelta { - collection: nodedb_types::QualifiedCollection::from_stored(collection.to_owned()), - document_id: document_id.to_owned(), - surrogate: nodedb_types::Surrogate::new(surrogate), - column: column.to_owned(), - delta: delta.to_owned(), - join_column: join_column.to_owned(), - join_value: join_value.to_owned(), + collection: nodedb_types::QualifiedCollection::from_stored(fields.collection.to_owned()), + document_id: fields.document_id.to_owned(), + surrogate: nodedb_types::Surrogate::new(fields.surrogate), + column: fields.column.to_owned(), + delta: fields.delta.to_owned(), + join_column: fields.join_column.to_owned(), + join_value: fields.join_value.to_owned(), + declared_primary_key: fields.declared_primary_key.map(str::to_owned), }) } @@ -788,6 +783,7 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "docs"), restart_identity: true, resolved_sum_targets: Vec::new(), + declared_primary_key: Some("sku".to_string()), }); let entry = to_replicated_entry(tenant, DatabaseId::DEFAULT, vshard, &plan) .expect("encode must not error") @@ -800,10 +796,16 @@ mod tests { PhysicalPlan::Document(DocumentOp::Truncate { collection, restart_identity, - .. + resolved_sum_targets: _, + declared_primary_key, }) => { assert_eq!(collection.as_str(), "docs"); assert!(restart_identity, "restart_identity must round-trip"); + assert_eq!( + declared_primary_key.as_deref(), + Some("sku"), + "declared_primary_key must round-trip" + ); } other => panic!("expected Document(Truncate), got {other:?}"), } diff --git a/nodedb/src/control/wal_replication/decode/entry_document.rs b/nodedb/src/control/wal_replication/decode/entry_document.rs index 2bc739e99..91bfc1adc 100644 --- a/nodedb/src/control/wal_replication/decode/entry_document.rs +++ b/nodedb/src/control/wal_replication/decode/entry_document.rs @@ -6,7 +6,7 @@ //! these variants. use super::super::decode_sync_engines::decode_returning; -use super::super::types::{ReplicatedSumTarget, ReplicatedWrite}; +use super::super::types::{BalanceDeltaFields, ReplicatedSumTarget, ReplicatedWrite}; use super::ctx::DecodeCtx; use super::document; use super::document::{PointInsertOptions, ReturningFields, UpsertExtras, WireSumResolution}; @@ -168,10 +168,12 @@ pub(super) fn decode_arm(ctx: &DecodeCtx, write: &ReplicatedWrite) -> crate::Res restart_identity, resolved_sum_targets, resolved_sum_target_bindings, + declared_primary_key, } => Ok(document::truncate( collection, *restart_identity, &sums(resolved_sum_target_bindings, resolved_sum_targets), + declared_primary_key.clone(), )), ReplicatedWrite::BulkDml { collection, @@ -216,15 +218,17 @@ pub(super) fn decode_arm(ctx: &DecodeCtx, write: &ReplicatedWrite) -> crate::Res delta, join_column, join_value, - } => Ok(document::apply_balance_delta( + declared_primary_key, + } => Ok(document::apply_balance_delta(BalanceDeltaFields { collection, document_id, - *surrogate, + surrogate: *surrogate, column, delta, join_column, join_value, - )), + declared_primary_key: declared_primary_key.as_deref(), + })), ReplicatedWrite::DocumentResolvedWrite { mutations, response_payload, diff --git a/nodedb/src/control/wal_replication/encode/document.rs b/nodedb/src/control/wal_replication/encode/document.rs index 4391efbd0..900db2ede 100644 --- a/nodedb/src/control/wal_replication/encode/document.rs +++ b/nodedb/src/control/wal_replication/encode/document.rs @@ -5,7 +5,7 @@ //! A document write that maintains a derived total carries the join-key → //! target-surrogate resolution, copied onto the record so no applier re-derives it. -use super::super::types::{ReplicatedSumTarget, ReplicatedWrite}; +use super::super::types::{BalanceDeltaFields, ReplicatedSumTarget, ReplicatedWrite}; use nodedb_physical::physical_plan::{DocumentResolvedMutation, ResolvedSumTarget, UpdateValue}; use nodedb_types::Surrogate; @@ -193,12 +193,14 @@ pub(super) fn truncate( collection: &str, restart_identity: bool, resolved_sum_targets: &[ResolvedSumTarget], + declared_primary_key: Option<&str>, ) -> ReplicatedWrite { ReplicatedWrite::DocTruncate { collection: collection.to_owned(), restart_identity, resolved_sum_targets: wire_targets(resolved_sum_targets), resolved_sum_target_bindings: wire_target_bindings(resolved_sum_targets), + declared_primary_key: declared_primary_key.map(str::to_owned), } } @@ -320,22 +322,15 @@ pub(super) fn resolved_write( /// Replicates as the delta it is, modelled on `KvIncr`: each replica applies it /// once in log order onto its own prior balance. The decimal travels as a /// string because `f64` is lossy past 15 significant digits. -pub(super) fn apply_balance_delta( - collection: &str, - document_id: &str, - surrogate: u32, - column: &str, - delta: &str, - join_column: &str, - join_value: &str, -) -> ReplicatedWrite { +pub(super) fn apply_balance_delta(fields: BalanceDeltaFields<'_>) -> ReplicatedWrite { ReplicatedWrite::ApplyBalanceDelta { - collection: collection.to_owned(), - document_id: document_id.to_owned(), - surrogate, - column: column.to_owned(), - delta: delta.to_owned(), - join_column: join_column.to_owned(), - join_value: join_value.to_owned(), + collection: fields.collection.to_owned(), + document_id: fields.document_id.to_owned(), + surrogate: fields.surrogate, + column: fields.column.to_owned(), + delta: fields.delta.to_owned(), + join_column: fields.join_column.to_owned(), + join_value: fields.join_value.to_owned(), + declared_primary_key: fields.declared_primary_key.map(str::to_owned), } } diff --git a/nodedb/src/control/wal_replication/encode/entry_document.rs b/nodedb/src/control/wal_replication/encode/entry_document.rs index 9d9f5d682..ef62fee7a 100644 --- a/nodedb/src/control/wal_replication/encode/entry_document.rs +++ b/nodedb/src/control/wal_replication/encode/entry_document.rs @@ -7,7 +7,7 @@ #![deny(clippy::wildcard_enum_match_arm)] -use super::super::types::ReplicatedWrite; +use super::super::types::{BalanceDeltaFields, ReplicatedWrite}; use super::document; use super::document::{SumFields, WireReturning}; use super::entry::encode_returning; @@ -223,7 +223,13 @@ pub(super) fn document_write(op: &DocumentOp) -> Option { restart_identity, // See `PointPut`. resolved_sum_targets, - } => document::truncate(collection.as_str(), *restart_identity, resolved_sum_targets), + declared_primary_key, + } => document::truncate( + collection.as_str(), + *restart_identity, + resolved_sum_targets, + declared_primary_key.as_deref(), + ), // OLLP-prepared bulk plans route via the cross-shard Calvin path, not // single-shard Raft proposal. DocumentOp::BulkDelete { .. } | DocumentOp::BulkUpdate { .. } => return None, @@ -245,15 +251,17 @@ pub(super) fn document_write(op: &DocumentOp) -> Option { delta, join_column, join_value, - } => document::apply_balance_delta( - collection.as_str(), + declared_primary_key, + } => document::apply_balance_delta(BalanceDeltaFields { + collection: collection.as_str(), document_id, - surrogate.as_u32(), + surrogate: surrogate.as_u32(), column, delta, join_column, join_value, - ), + declared_primary_key: declared_primary_key.as_deref(), + }), // Not a write — reads / scans / index DDL-metadata / system ops. DocumentOp::ResolveWrite(_) diff --git a/nodedb/src/control/wal_replication/types/mod.rs b/nodedb/src/control/wal_replication/types/mod.rs index 2ad067db9..7af59d455 100644 --- a/nodedb/src/control/wal_replication/types/mod.rs +++ b/nodedb/src/control/wal_replication/types/mod.rs @@ -26,6 +26,6 @@ pub use aliases::{AsyncRaftProposer, RaftAppliedIndexSink, RaftCompactor, RaftPr pub use replicated_entry::ReplicatedEntry; pub use replicated_write::ReplicatedWrite; pub use wire_shapes::{ - ColumnarResolvedRow, ConstraintChangeOp, DocumentResolvedMutationWire, KvResolvedMutationWire, - ReplicatedBatchEdge, ReplicatedSumTarget, + BalanceDeltaFields, ColumnarResolvedRow, ConstraintChangeOp, DocumentResolvedMutationWire, + KvResolvedMutationWire, ReplicatedBatchEdge, ReplicatedSumTarget, }; diff --git a/nodedb/src/control/wal_replication/types/replicated_write.rs b/nodedb/src/control/wal_replication/types/replicated_write.rs index 154c756a8..11d74c603 100644 --- a/nodedb/src/control/wal_replication/types/replicated_write.rs +++ b/nodedb/src/control/wal_replication/types/replicated_write.rs @@ -666,6 +666,10 @@ pub enum ReplicatedWrite { /// See `PointPut::resolved_sum_target_bindings`. #[serde(default)] resolved_sum_target_bindings: Vec, + /// See `PointUpdate::declared_primary_key`. Names the column each + /// removed row's identity is read from on every applier. + #[serde(default)] + declared_primary_key: Option, }, KvTruncate { collection: String, @@ -751,6 +755,9 @@ pub enum ReplicatedWrite { join_column: String, /// Join value that resolved to `surrogate`. join_value: String, + /// See `PointUpdate::declared_primary_key`, for the TARGET collection. + #[serde(default)] + declared_primary_key: Option, }, /// Resolved-row-set form of a columnar predicate `UPDATE` / `DELETE` on a diff --git a/nodedb/src/control/wal_replication/types/wire_shapes.rs b/nodedb/src/control/wal_replication/types/wire_shapes.rs index aeeb26524..5e4a6e473 100644 --- a/nodedb/src/control/wal_replication/types/wire_shapes.rs +++ b/nodedb/src/control/wal_replication/types/wire_shapes.rs @@ -28,6 +28,28 @@ pub struct ReplicatedBatchEdge { pub dst_surrogate: u32, } +/// The fields of one `ApplyBalanceDelta`, borrowed from the plan or the wire +/// entry. Shared by the encode and decode helpers so both name every field. +#[derive(Debug, Clone, Copy)] +pub struct BalanceDeltaFields<'a> { + /// TARGET collection, db-qualified. + pub collection: &'a str, + /// Target row's storage key — hex-encoded surrogate. + pub document_id: &'a str, + /// Target row's global identity. + pub surrogate: u32, + /// The balance column this delta moves. + pub column: &'a str, + /// Signed amount as an exact decimal string. + pub delta: &'a str, + /// Binding's join column, for the typed not-found error on apply. + pub join_column: &'a str, + /// Join value that resolved to `surrogate`. + pub join_value: &'a str, + /// The TARGET collection's declared `PRIMARY KEY` column, when it has one. + pub declared_primary_key: Option<&'a str>, +} + /// One entry of a write's materialized-sum resolution: which target row a /// binding's `(target collection, join value)` pair names. Supersedes the /// `(join_value, surrogate)` pairs in `*_sum_targets`, which can't tell apart diff --git a/nodedb/src/data/executor/core_loop/deferred.rs b/nodedb/src/data/executor/core_loop/deferred.rs index 1a2303afa..e1bf27b22 100644 --- a/nodedb/src/data/executor/core_loop/deferred.rs +++ b/nodedb/src/data/executor/core_loop/deferred.rs @@ -50,7 +50,7 @@ impl CoreLoop { sequence: self.event_sequence, collection: Arc::from(write.collection.as_str()), op: write.op, - row_id: RowId::new(write.identity.as_str()), + row_id: RowId::row(write.identity), lsn: self.watermark, database_id, tenant_id, diff --git a/nodedb/src/data/executor/core_loop/event_emit.rs b/nodedb/src/data/executor/core_loop/event_emit.rs index 572a20549..608163887 100644 --- a/nodedb/src/data/executor/core_loop/event_emit.rs +++ b/nodedb/src/data/executor/core_loop/event_emit.rs @@ -212,22 +212,14 @@ impl CoreLoop { task: &super::super::task::ExecutionTask, edge: GraphEdgeEvent<'_>, ) { - let row_id = crate::event::graph_cdc::edge_row_id(edge.src_id, edge.label, edge.dst_id); - let identity = crate::engine::document::store::RowIdentity::from_user_key(row_id); + let row_id = crate::event::types::RowId::edge(edge.src_id, edge.label, edge.dst_id); let (new_value, old_value): (Option<&[u8]>, Option<&[u8]>) = if matches!(edge.op, crate::event::WriteOp::Delete) { (None, None) } else { (edge.properties, None) }; - self.emit_write_event( - task, - edge.collection, - edge.op, - identity, - new_value, - old_value, - ); + self.emit_event_with_row_id(task, edge.collection, edge.op, row_id, new_value, old_value); } /// Set the Event Plane producer (called after open, before event loop). @@ -235,7 +227,7 @@ impl CoreLoop { self.event_producer = Some(producer); } - /// Emit a write event to the Event Plane. + /// Emit a write event for one row to the Event Plane. /// /// Called after a successful write (PointPut, PointDelete, PointUpdate, /// BatchInsert, BulkDelete, atomic KV ops, etc.). The Data Plane NEVER @@ -258,6 +250,29 @@ impl CoreLoop { identity: crate::engine::document::store::RowIdentity, new_value: Option<&[u8]>, old_value: Option<&[u8]>, + ) { + self.emit_event_with_row_id( + task, + collection, + op, + crate::event::types::RowId::row(identity), + new_value, + old_value, + ); + } + + /// Emit a write event carrying any [`crate::event::types::RowId`]. + /// + /// [`Self::emit_write_event`] is the entry point for single rows. Edge + /// events name an `(src, label, dst)` triple and call this directly. + pub(in crate::data::executor) fn emit_event_with_row_id( + &mut self, + task: &super::super::task::ExecutionTask, + collection: &str, + op: crate::event::WriteOp, + row_id: crate::event::types::RowId, + new_value: Option<&[u8]>, + old_value: Option<&[u8]>, ) { let producer = match self.event_producer.as_mut() { Some(p) => p, @@ -273,7 +288,7 @@ impl CoreLoop { sequence: self.event_sequence, collection: Arc::from(collection), op, - row_id: crate::event::types::RowId::new(identity.into_string()), + row_id, lsn: self.watermark, database_id: task.request.database_id, tenant_id: task.request.tenant_id, @@ -307,7 +322,7 @@ impl CoreLoop { sequence: self.event_sequence, collection: Arc::from("_heartbeat"), op: crate::event::WriteOp::Heartbeat, - row_id: crate::event::types::RowId::new(""), + row_id: crate::event::types::RowId::Heartbeat, // watermark = last committed LSN. Correct for heartbeats: uncommitted // writes should NOT advance the Event Plane's watermark. lsn: self.watermark, diff --git a/nodedb/src/data/executor/dispatch/document.rs b/nodedb/src/data/executor/dispatch/document.rs index 7dbbefa65..d4da40e52 100644 --- a/nodedb/src/data/executor/dispatch/document.rs +++ b/nodedb/src/data/executor/dispatch/document.rs @@ -281,9 +281,7 @@ impl CoreLoop { rls_filters, rls_write_check, resolved_sum_targets, - // Read only by overlay staging; a durable delete removes rows - // by storage key. - declared_primary_key: _, + declared_primary_key, } => self.execute_bulk_delete( task, tid, @@ -298,6 +296,7 @@ impl CoreLoop { surrogates: ollp_predicted_surrogates.as_deref(), edges: ollp_predicted_edges.as_deref(), }, + declared_primary_key: declared_primary_key.as_deref(), }, ), @@ -329,9 +328,18 @@ impl CoreLoop { DocumentOp::Truncate { collection, + restart_identity: _, resolved_sum_targets, - .. - } => self.execute_truncate(task, tid, collection.as_str(), resolved_sum_targets), + declared_primary_key, + } => self.execute_truncate( + task, + tid, + super::super::handlers::truncate::TruncateParams { + collection: collection.as_str(), + resolved_sum_targets, + declared_primary_key: declared_primary_key.as_deref(), + }, + ), DocumentOp::EstimateCount { collection, field } => { self.execute_estimate_count(task, tid, collection.as_str(), field) @@ -470,6 +478,7 @@ impl CoreLoop { delta, join_column, join_value, + declared_primary_key, } => self.execute_apply_balance_delta( task, super::super::handlers::document::apply_balance_delta::ApplyBalanceDeltaParams { @@ -481,6 +490,7 @@ impl CoreLoop { delta, join_column, join_value, + declared_primary_key: declared_primary_key.as_deref(), }, ), } diff --git a/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs b/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs index 39f53e7f4..8210ca72a 100644 --- a/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs +++ b/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs @@ -70,7 +70,7 @@ use redb::WriteTransaction; use rust_decimal::Decimal; use nodedb_physical::physical_plan::MaterializedSumBinding; -use nodedb_types::Surrogate; +use nodedb_types::{RowIdentity, Surrogate}; use super::delta::fold_sum_deltas; use super::rmw::BalanceRmw; @@ -87,6 +87,10 @@ pub(in crate::data::executor) struct TargetWrite { /// The target row's surrogate, so an undo entry addresses the same identity /// the forward write used. pub surrogate: Surrogate, + /// The target row's client identity: its declared primary key when the + /// target declares one, else its decimal surrogate. The redo entry and the + /// target row's event both name the row by it. + pub identity: RowIdentity, /// The MessagePack body this write handed to `apply_point_put` — NOT the /// bytes that reached storage. /// @@ -248,6 +252,7 @@ impl CoreLoop { join_column: &binding.join_column, join_value, wal_lsn: ctx.wal_lsn, + target_declared_primary_key: binding.declared_primary_key.as_deref(), }, ) } @@ -347,6 +352,7 @@ mod tests { target_column: "balance".to_string(), join_column: "account_id".to_string(), value_expr: nodedb_query::expr::SqlExpr::Column("amount".to_string()), + declared_primary_key: None, } } @@ -915,6 +921,7 @@ mod tests { surrogates: None, edges: None, }, + declared_primary_key: None, }, ); @@ -937,7 +944,15 @@ mod tests { let resolved = resolved_onto_target(&[(ACCOUNT_A, SURROGATE_A), (ACCOUNT_B, SURROGATE_B)]); let task = make_default_task(); - let response = core.execute_truncate(&task, TID, SOURCE, &resolved); + let response = core.execute_truncate( + &task, + TID, + crate::data::executor::handlers::truncate::TruncateParams { + collection: SOURCE, + resolved_sum_targets: &resolved, + declared_primary_key: None, + }, + ); assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); assert_eq!(balance_of(&core, SURROGATE_A), "0"); @@ -1092,6 +1107,7 @@ mod tests { surrogates: None, edges: None, }, + declared_primary_key: None, }, ); @@ -1170,6 +1186,7 @@ mod tests { surrogates: None, edges: None, }, + declared_primary_key: None, }, ); @@ -1236,6 +1253,7 @@ mod tests { surrogates: None, edges: None, }, + declared_primary_key: None, }, ); diff --git a/nodedb/src/data/executor/enforcement/materialized_sum/delta.rs b/nodedb/src/data/executor/enforcement/materialized_sum/delta.rs index d917c7984..8b451f212 100644 --- a/nodedb/src/data/executor/enforcement/materialized_sum/delta.rs +++ b/nodedb/src/data/executor/enforcement/materialized_sum/delta.rs @@ -97,6 +97,7 @@ mod tests { target_column: "balance".to_string(), join_column: "account_id".to_string(), value_expr: nodedb_query::expr::SqlExpr::Column("amount".to_string()), + declared_primary_key: None, } } diff --git a/nodedb/src/data/executor/enforcement/materialized_sum/rmw.rs b/nodedb/src/data/executor/enforcement/materialized_sum/rmw.rs index 858696bd7..e9d67ba78 100644 --- a/nodedb/src/data/executor/enforcement/materialized_sum/rmw.rs +++ b/nodedb/src/data/executor/enforcement/materialized_sum/rmw.rs @@ -26,7 +26,7 @@ use redb::WriteTransaction; use rust_decimal::Decimal; -use nodedb_types::Surrogate; +use nodedb_types::{RowIdentity, Surrogate}; use super::apply::TargetWrite; use super::delta::json_to_decimal; @@ -56,6 +56,9 @@ pub(in crate::data::executor) struct BalanceRmw<'a> { /// Join value that resolved to `surrogate`, for the typed not-found error. pub join_value: &'a str, pub wal_lsn: Option, + /// The TARGET collection's declared `PRIMARY KEY` column, when it has + /// one. Names the target row in its event and redo entry. + pub target_declared_primary_key: Option<&'a str>, } impl BalanceRmw<'_> { @@ -189,9 +192,15 @@ impl CoreLoop { } }; + // The identity INSERT minted for the target row, read from the + // MessagePack body just written: the declared primary key when the + // target declares one, else the decimal surrogate. + let identity = + RowIdentity::of_stored_row(&body, params.target_declared_primary_key, storage_key); Ok(TargetWrite { collection: params.target_collection.to_string(), surrogate: params.surrogate, + identity, body, outcome, }) diff --git a/nodedb/src/data/executor/enforcement/write_hook.rs b/nodedb/src/data/executor/enforcement/write_hook.rs index 0e29d5035..ba7948708 100644 --- a/nodedb/src/data/executor/enforcement/write_hook.rs +++ b/nodedb/src/data/executor/enforcement/write_hook.rs @@ -194,6 +194,7 @@ pub(in crate::data::executor) fn target_write_set( .iter() .map(|target| crate::bridge::envelope::WriteSetEntry { surrogate: target.surrogate.as_u32(), + identity: target.identity.clone(), is_delete: false, value: target.body.clone(), collection: Some(target.collection.clone()), diff --git a/nodedb/src/data/executor/handlers/bulk_dml/delete.rs b/nodedb/src/data/executor/handlers/bulk_dml/delete.rs index 890fff20a..1c5d0e7b3 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/delete.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/delete.rs @@ -45,6 +45,9 @@ pub(in crate::data::executor) struct BulkDeleteParams<'a> { /// Plane from its recon scan of the same predicate. pub resolved_sum_targets: &'a [ResolvedSumTarget], pub ollp: OllpPrediction<'a>, + /// The collection's declared `PRIMARY KEY` column, when it has one. Names + /// each removed row in its redo entry and delete event. + pub declared_primary_key: Option<&'a str>, } impl CoreLoop { @@ -67,6 +70,7 @@ impl CoreLoop { rls_write_check, resolved_sum_targets, ollp, + declared_primary_key, } = params; let ollp_predicted_surrogates = ollp.surrogates; let ollp_predicted_edges = ollp.edges; @@ -353,6 +357,8 @@ impl CoreLoop { doc_id: doc_id.as_str(), storage_key: *storage_key, deleted_bytes: bytes, + strict_schema: strict_schema.as_ref(), + declared_primary_key, has_vectors, index_paths: &index_paths, pre_delete_doc, diff --git a/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs b/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs index 0ebb807e6..7e5c3949a 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs @@ -8,10 +8,12 @@ //! on failure — each step logs and continues rather than aborting a //! statement it cannot undo. +use nodedb_types::columnar::StrictSchema; use tracing::warn; use crate::bridge::envelope::WriteSetEntry; use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::transaction::stage_write::stored_row_identity; use crate::data::executor::task::ExecutionTask; use crate::engine::document::store::{IndexPath, StorageKey}; @@ -27,6 +29,11 @@ pub(in crate::data::executor) struct BulkDeleteRowCascade<'a> { /// The row's pre-deletion bytes, as `sparse.delete` returned them — /// never re-read. pub deleted_bytes: &'a [u8], + /// The collection's strict schema, when it stores Binary Tuples. Decodes + /// `deleted_bytes` so the row's identity column is readable. + pub strict_schema: Option<&'a StrictSchema>, + /// The collection's declared `PRIMARY KEY` column, when it has one. + pub declared_primary_key: Option<&'a str>, pub has_vectors: bool, pub index_paths: &'a [IndexPath], /// The pre-deletion document, when the caller captured one (`RETURNING` @@ -54,12 +61,24 @@ impl CoreLoop { doc_id, storage_key, deleted_bytes, + strict_schema, + declared_primary_key, has_vectors, index_paths, pre_delete_doc, returning, } = cascade; + // The identity INSERT minted for this row: its declared primary key + // when the collection declares one, else its decimal surrogate. The + // redo entry and the delete event both name the row by it. + let row_identity = stored_row_identity( + deleted_bytes, + strict_schema, + declared_primary_key, + storage_key, + ); + // Cascade: inverted index. let row_surrogate = storage_key.surrogate(); if let Err(e) = self.inverted.remove_document( @@ -137,6 +156,7 @@ impl CoreLoop { if has_vectors { write_set.push(WriteSetEntry { surrogate: row_surrogate.as_u32(), + identity: row_identity.clone(), is_delete: true, value: Vec::new(), collection: None, @@ -158,11 +178,10 @@ impl CoreLoop { collection, deleted_bytes, ); - let event_identity = storage_key.to_identity(); self.emit_document_delete_event( task, collection, - event_identity, + row_identity, Some(old_converted.as_deref().unwrap_or(deleted_bytes)), ); if returning && let Some(doc) = pre_delete_doc { diff --git a/nodedb/src/data/executor/handlers/bulk_dml/update.rs b/nodedb/src/data/executor/handlers/bulk_dml/update.rs index 3c4c3c8e7..12a0b029d 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/update.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/update.rs @@ -11,6 +11,7 @@ use crate::data::executor::handlers::point::update_reindex_vector::UpdateVectorR use crate::data::executor::handlers::returning_doc; use crate::data::executor::handlers::returning_rows; use crate::data::executor::handlers::rls_write_gate; +use crate::data::executor::handlers::transaction::stage_write::stored_row_identity; use crate::data::executor::response_codec; use crate::data::executor::task::ExecutionTask; use nodedb_physical::physical_plan::{OllpPredictedEdge, ResolvedSumTarget, ReturningSpec}; @@ -365,7 +366,16 @@ impl CoreLoop { // Event Plane's WAL-replay bulk variants are aggregate // metadata reconstructed only when the live per-row events // were lost — the live path always emits per row. - let row_identity = storage_key.to_identity(); + // + // The identity is the one INSERT minted: the declared primary + // key when the collection declares one, else the decimal + // surrogate. The redo entry below journals the same identity. + let row_identity = stored_row_identity( + &updated_bytes, + strict_schema.as_ref(), + declared_primary_key, + storage_key, + ); // `row_identity` is read again below for `RETURNING`'s `id` field, // so the event-emit boundary gets a clone rather than the move. self.emit_put_event( @@ -391,6 +401,7 @@ impl CoreLoop { if has_vectors { write_set.push(WriteSetEntry { surrogate: surrogate.as_u32(), + identity: row_identity, is_delete: false, value: updated_bytes, collection: None, diff --git a/nodedb/src/data/executor/handlers/control/crdt_apply/gated.rs b/nodedb/src/data/executor/handlers/control/crdt_apply/gated.rs index 240ce53a5..0ea52f780 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_apply/gated.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_apply/gated.rs @@ -331,6 +331,7 @@ impl CoreLoop { task, tenant_id.as_u64(), collection, + document_id, surrogate, &bytes, ); diff --git a/nodedb/src/data/executor/handlers/control/crdt_apply/local.rs b/nodedb/src/data/executor/handlers/control/crdt_apply/local.rs index ad454feb3..49bbef079 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_apply/local.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_apply/local.rs @@ -140,6 +140,7 @@ impl CoreLoop { task, tenant_id.as_u64(), collection, + document_id, surrogate, &bytes, ); diff --git a/nodedb/src/data/executor/handlers/control/crdt_doc.rs b/nodedb/src/data/executor/handlers/control/crdt_doc.rs index 582f29c23..16445425d 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_doc.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_doc.rs @@ -113,11 +113,14 @@ impl CoreLoop { let response = if let Some(bytes) = materialized { self.materialize_document_write( task, - tenant_id.as_u64(), - collection, - surrogate, - &bytes, - true, + super::crdt_materialize::CrdtMaterializeWrite { + tid: tenant_id.as_u64(), + collection, + document_id, + surrogate, + value: &bytes, + index_text: true, + }, ); if let Some(spec) = returning { // No strict schema: a CRDT row's stored body is whatever @@ -267,7 +270,7 @@ impl CoreLoop { self.emit_document_delete_event( task, collection, - storage_key.to_identity(), + RowIdentity::from_user_key(document_id), Some(old_converted.as_deref().unwrap_or(prior_bytes)), ); } diff --git a/nodedb/src/data/executor/handlers/control/crdt_materialize.rs b/nodedb/src/data/executor/handlers/control/crdt_materialize.rs index c56af9c9f..cf110acdb 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_materialize.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_materialize.rs @@ -29,7 +29,7 @@ use tracing::warn; -use nodedb_types::Surrogate; +use nodedb_types::{RowIdentity, Surrogate}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::point::apply_put::PointPutParams; @@ -38,6 +38,20 @@ use crate::engine::crdt::tenant_state::TenantCrdtEngine; use crate::engine::document::crdt_store::loro_value_to_json; use crate::engine::document::store::StorageKey; +/// One CRDT row write to materialize into the sparse store. +pub(super) struct CrdtMaterializeWrite<'a> { + pub tid: u64, + pub collection: &'a str, + /// The Loro row id the client addresses the row by: its client identity. + pub document_id: &'a str, + pub surrogate: Surrogate, + pub value: &'a [u8], + /// Gates inverted BM25 text indexing: `false` on the CRDT sync path (a + /// separate `FtsIndex` frame delivers text), `true` for user SQL DML on a + /// `crdt='true'` collection. + pub index_text: bool, +} + impl CoreLoop { /// Read the merged Loro row back and encode it into the schemaless /// MessagePack bytes the native put path accepts. @@ -72,7 +86,8 @@ impl CoreLoop { /// Routes through `apply_point_put` inside a single write transaction, then /// commits and emits the `WriteEvent`, mirroring `execute_point_put`. The /// storage key is the hex-encoded surrogate (identical to the native path), - /// NOT the CRDT `document_id` (the user-facing Loro row id); bitemporal + /// NOT the CRDT `document_id` (the user-facing Loro row id, which the + /// `WriteEvent` names the row by); bitemporal /// collections append a version per applied delta (handled inside /// `apply_point_put`), non-bitemporal collections overwrite by key /// (idempotent under replay). Inverted BM25 text indexing is skipped @@ -85,26 +100,42 @@ impl CoreLoop { task: &ExecutionTask, tid: u64, collection: &str, + document_id: &str, surrogate: Surrogate, value: &[u8], ) { - self.materialize_document_write(task, tid, collection, surrogate, value, false); + self.materialize_document_write( + task, + CrdtMaterializeWrite { + tid, + collection, + document_id, + surrogate, + value, + index_text: false, + }, + ); } /// Shared body of the sparse-store materialization. `index_text` gates /// inverted BM25 text indexing: `false` on the CRDT sync path (a separate /// `FtsIndex` frame delivers text), `true` for user SQL DML on a /// `crdt='true'` collection (no separate frame — the merged row is the - /// only source). + /// only source). `document_id` is the Loro row id the client addresses + /// the row by: the row's client identity. pub(super) fn materialize_document_write( &mut self, task: &ExecutionTask, - tid: u64, - collection: &str, - surrogate: Surrogate, - value: &[u8], - index_text: bool, + write: CrdtMaterializeWrite<'_>, ) { + let CrdtMaterializeWrite { + tid, + collection, + document_id, + surrogate, + value, + index_text, + } = write; let database_id = task.request.database_id.as_u64(); let storage_key = StorageKey::for_surrogate(surrogate); @@ -159,7 +190,7 @@ impl CoreLoop { task, tid, collection, - storage_key.to_identity(), + RowIdentity::from_user_key(document_id), value, prior.prior_value.as_deref(), ); diff --git a/nodedb/src/data/executor/handlers/document/apply_balance_delta.rs b/nodedb/src/data/executor/handlers/document/apply_balance_delta.rs index ab6845469..3e6202049 100644 --- a/nodedb/src/data/executor/handlers/document/apply_balance_delta.rs +++ b/nodedb/src/data/executor/handlers/document/apply_balance_delta.rs @@ -52,6 +52,8 @@ pub(in crate::data::executor) struct ApplyBalanceDeltaParams<'a> { pub delta: &'a str, pub join_column: &'a str, pub join_value: &'a str, + /// The TARGET collection's declared `PRIMARY KEY` column, when it has one. + pub declared_primary_key: Option<&'a str>, } impl CoreLoop { @@ -69,6 +71,7 @@ impl CoreLoop { delta, join_column, join_value, + declared_primary_key, } = params; debug!( core = self.core_id, @@ -115,6 +118,7 @@ impl CoreLoop { join_column, join_value, wal_lsn: task.wal_lsn(), + target_declared_primary_key: declared_primary_key, }, ) { Ok(write) => write, @@ -145,6 +149,7 @@ impl CoreLoop { let mut response = self.response_affected(task, 1); response.write_set = vec![crate::bridge::envelope::WriteSetEntry { surrogate: write.surrogate.as_u32(), + identity: write.identity, is_delete: false, value: write.body, // Always `Some`: the row lives in the TARGET collection, and the diff --git a/nodedb/src/data/executor/handlers/document/resolve/apply.rs b/nodedb/src/data/executor/handlers/document/resolve/apply.rs index 87344aa40..1ca8ffe3e 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/apply.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/apply.rs @@ -71,13 +71,14 @@ impl CoreLoop { value, precondition, resolved_sum_targets, - document_id: _, + document_id, pk_bytes: _, } => self.apply_resolved_document_put( task, ApplyResolvedPut { tid, collection: collection.as_str(), + document_id, surrogate: *surrogate, value, precondition: precondition.as_deref(), diff --git a/nodedb/src/data/executor/handlers/document/resolve/apply_row.rs b/nodedb/src/data/executor/handlers/document/resolve/apply_row.rs index 46710cf76..4219941ee 100644 --- a/nodedb/src/data/executor/handlers/document/resolve/apply_row.rs +++ b/nodedb/src/data/executor/handlers/document/resolve/apply_row.rs @@ -15,12 +15,14 @@ use crate::data::executor::enforcement::write_hook::{self, HookCtx, ImageBody, W use crate::data::executor::handlers::point::apply_delete::PointDeleteParams; use crate::data::executor::handlers::point::apply_put::PointPutParams; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::StorageKey; +use crate::engine::document::store::{RowIdentity, StorageKey}; /// One already-decided row write, as the apply loop hands it over. pub(super) struct ApplyResolvedPut<'a> { pub tid: u64, pub collection: &'a str, + /// The row's client identity, as `RETURNING` and CDC name it. + pub document_id: &'a str, pub surrogate: Surrogate, /// Pre-encode MessagePack body — the write path encodes the strict Binary /// Tuple from it. @@ -50,6 +52,7 @@ impl CoreLoop { let ApplyResolvedPut { tid, collection, + document_id, surrogate, value, precondition, @@ -57,6 +60,9 @@ impl CoreLoop { } = put; let database_id = task.request.database_id.as_u64(); let storage_key = StorageKey::for_surrogate(surrogate); + // The resolve pass carried the identity INSERT minted for this row; + // the event and the redo entry both name the row by it. + let row_identity = RowIdentity::from_user_key(document_id); let has_vectors = self.collection_has_vectors(database_id, tid, collection); // HNSW insert appends rather than replaces, so the prior embedding @@ -150,7 +156,7 @@ impl CoreLoop { task, tid, collection, - storage_key.to_identity(), + row_identity.clone(), &stored_bytes, precondition, ); @@ -162,6 +168,7 @@ impl CoreLoop { if has_vectors { write_set.push(WriteSetEntry { surrogate: surrogate.as_u32(), + identity: row_identity, is_delete: false, value: value.to_vec(), collection: None, @@ -255,7 +262,7 @@ impl CoreLoop { self.emit_document_delete_event( task, collection, - StorageKey::for_surrogate(surrogate).to_identity(), + RowIdentity::from_user_key(document_id), Some(old_converted.as_deref().unwrap_or(prior_bytes)), ); } diff --git a/nodedb/src/data/executor/handlers/document/write/batch_insert.rs b/nodedb/src/data/executor/handlers/document/write/batch_insert.rs index cf6af8102..8ed3d7d42 100644 --- a/nodedb/src/data/executor/handlers/document/write/batch_insert.rs +++ b/nodedb/src/data/executor/handlers/document/write/batch_insert.rs @@ -145,13 +145,15 @@ impl CoreLoop { let has_vectors = self.collection_has_vectors(database_id, tid, collection) || self.collection_has_sparse(database_id, tid, collection); - // Row key + storage key for post-commit event emission and cache + // Row identity + storage key for post-commit event emission and cache // invalidation, captured as each row applies successfully; the value // bytes are re-borrowed from `documents` after commit rather than // cloned here. On any error we return early (dropping `txn`, which // rolls back every row applied so far). - let mut applied: Vec<(String, nodedb_types::StorageKey)> = - Vec::with_capacity(documents.len()); + let mut applied: Vec<( + crate::engine::document::store::RowIdentity, + nodedb_types::StorageKey, + )> = Vec::with_capacity(documents.len()); let mut write_set: Vec = Vec::new(); // Per-row secondary-index tuples (added ∪ removed ∪ bitemporal), // parallel to `applied`. Recorded into the per-index write-value @@ -178,7 +180,10 @@ impl CoreLoop { for (i, (document_id, value)) in documents.iter().enumerate() { let surrogate = surrogates[i]; let key = nodedb_types::StorageKey::for_surrogate(surrogate); - let row_key = key.to_string(); + // The plan's `document_id` is the identity INSERT minted for this + // row: the declared primary key value, else the decimal surrogate. + let row_identity = + crate::engine::document::store::RowIdentity::from_user_key(document_id.as_str()); // Every row of a batch insert is INSERT-shaped, so every row is a // chain link. The chain rewrites the BODY, so it runs before the // body is encoded and stored. The link covers the user-visible @@ -187,9 +192,7 @@ impl CoreLoop { let chained = match chain.chain_insert(self, database_id, tid, document_id, value) { Ok(chained) => chained, Err(e) => { - // `key` is `Copy`, so this costs nothing and `row_key` - // stays borrowed by the parameters of the call this arm - // is handling. + // `key` is `Copy`, so this costs nothing. failure = Some((key, e)); break; } @@ -234,9 +237,7 @@ impl CoreLoop { ) { Ok(enforcement) => enforcement, Err(e) => { - // `key` is `Copy`, so this costs nothing and `row_key` - // stays borrowed by the parameters of the call this arm - // is handling. + // `key` is `Copy`, so this costs nothing. failure = Some((key, e)); break; } @@ -249,6 +250,7 @@ impl CoreLoop { if has_vectors { write_set.push(WriteSetEntry { surrogate: surrogate.as_u32(), + identity: row_identity.clone(), is_delete: false, value: value.clone(), collection: None, @@ -260,7 +262,7 @@ impl CoreLoop { tuples.extend(outcome.bitemporal_index_tuples); row_index_tuples.push(tuples); } - applied.push((row_key, key)); + applied.push((row_identity, key)); } if let Some((failed_key, error)) = failure { @@ -331,8 +333,7 @@ impl CoreLoop { m.record_document_insert(); } - for (i, (_, key)) in applied.iter().enumerate() { - let identity = key.to_identity(); + for (i, (identity, _)) in applied.into_iter().enumerate() { self.emit_put_event(task, tid, collection, identity, &documents[i].1, None); } @@ -599,6 +600,7 @@ mod tests { target_column: "balance".to_string(), join_column: "account_id".to_string(), value_expr: nodedb_query::expr::SqlExpr::Column("amount".to_string()), + declared_primary_key: None, } } diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/insert_rows.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/insert_rows.rs index ede0e5691..e79998529 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/insert_rows.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/insert_rows.rs @@ -14,7 +14,7 @@ use crate::data::executor::enforcement::write_hook; use crate::data::executor::handlers::point::apply_put::PointPutParams; use crate::data::executor::handlers::transaction::undo::UndoEntry; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::StorageKey; +use crate::engine::document::store::{RowIdentity, StorageKey}; use nodedb_types::Surrogate; use super::super::abort::MergeAbort; @@ -35,6 +35,9 @@ pub(super) struct InsertRowsCtx<'a> { /// Source join value → Control-Plane-pre-assigned surrogate, verified /// against `inserts` by the caller before this arm runs. pub(super) surrogate_for: &'a HashMap<&'a str, u32>, + /// The target's declared `PRIMARY KEY` column, when it has one. Names + /// each row in its event and redo entry. + pub(super) declared_primary_key: Option<&'a str>, } /// Mutable accumulators the INSERT arm folds into. Owned by the caller for @@ -70,6 +73,7 @@ impl CoreLoop { returning, resolved_sum_targets, surrogate_for, + declared_primary_key, } = ctx; let InsertRowsTally { affected, @@ -100,8 +104,11 @@ impl CoreLoop { } }; let storage_key = StorageKey::for_surrogate(surrogate); - let row_key = storage_key.to_string(); - applied_keys.push(row_key.clone()); + applied_keys.push(storage_key.to_string()); + // The identity INSERT minted for this row, from its MessagePack + // body: the declared primary key, else the decimal surrogate. + let row_identity = + RowIdentity::of_stored_row(&ins.body, declared_primary_key, storage_key); match self.apply_point_put( txn, PointPutParams { @@ -158,6 +165,7 @@ impl CoreLoop { if has_vectors { write_set.push(WriteSetEntry { surrogate: surrogate.as_u32(), + identity: row_identity.clone(), is_delete: false, value: ins.body.clone(), collection: None, @@ -179,7 +187,7 @@ impl CoreLoop { } } } - put_events.push((row_key, ins.body.as_slice(), None)); + put_events.push((row_identity, ins.body.as_slice(), None)); *affected += 1; } Err(e) => { diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/orchestrate.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/orchestrate.rs index 0ba3a6027..9129c39b8 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/orchestrate.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/orchestrate.rs @@ -133,6 +133,7 @@ impl CoreLoop { has_vectors, returning: params.returning.is_some(), resolved_sum_targets: params.resolved_sum_targets, + declared_primary_key: params.declared_primary_key, }, UpdateRowsTally { affected: &mut affected, @@ -159,6 +160,7 @@ impl CoreLoop { returning: params.returning.is_some(), resolved_sum_targets: params.resolved_sum_targets, surrogate_for: &surrogate_for, + declared_primary_key: params.declared_primary_key, }, InsertRowsTally { affected: &mut affected, @@ -210,8 +212,7 @@ impl CoreLoop { self.checkpoint_coordinator .mark_dirty("sparse", put_events.len()); - for (row_key, body, prior) in &put_events { - let identity = crate::engine::document::store::identity_of(row_key); + for (identity, body, prior) in put_events { self.emit_put_event( task, tid, @@ -234,6 +235,7 @@ impl CoreLoop { has_vectors, returning: params.returning.is_some(), resolved_targets: params.resolved_sum_targets, + declared_primary_key: params.declared_primary_key, }, MergeDeleteTally { affected: &mut affected, diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs index a7c2b0e0e..54024e2fe 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/apply/update_rows.rs @@ -12,6 +12,7 @@ use crate::data::executor::enforcement::write_hook; use crate::data::executor::handlers::point::apply_put::PointPutParams; use crate::data::executor::handlers::transaction::undo::UndoEntry; use crate::data::executor::task::ExecutionTask; +use crate::engine::document::store::RowIdentity; use super::super::abort::MergeAbort; use super::super::apply_support::{MergePutEvent, record_put_index_undo, returning_doc}; @@ -29,6 +30,9 @@ pub(super) struct UpdateRowsCtx<'a> { /// Whether the statement carries a `RETURNING` projection. pub(super) returning: bool, pub(super) resolved_sum_targets: &'a [nodedb_physical::physical_plan::ResolvedSumTarget], + /// The target's declared `PRIMARY KEY` column, when it has one. Names + /// each row in its event and redo entry. + pub(super) declared_primary_key: Option<&'a str>, } /// Mutable accumulators the UPDATE arm folds into. Owned by the caller for @@ -63,6 +67,7 @@ impl CoreLoop { has_vectors, returning, resolved_sum_targets, + declared_primary_key, } = ctx; let UpdateRowsTally { affected, @@ -76,8 +81,10 @@ impl CoreLoop { for upd in updates { let surrogate = upd.key.surrogate(); - let row_key = upd.key.to_string(); - applied_keys.push(row_key.clone()); + applied_keys.push(upd.key.to_string()); + // The identity INSERT minted for this row, from its MessagePack + // body: the declared primary key, else the decimal surrogate. + let row_identity = RowIdentity::of_stored_row(&upd.body, declared_primary_key, upd.key); // `apply_point_put`'s vector step APPENDS (it never replaces), // so an in-place UPDATE must first soft-delete the surrogate's // prior embedding or the stale vector keeps scoring in KNN @@ -155,6 +162,7 @@ impl CoreLoop { if has_vectors { write_set.push(WriteSetEntry { surrogate: surrogate.as_u32(), + identity: row_identity.clone(), is_delete: false, value: upd.body.clone(), collection: None, @@ -176,7 +184,7 @@ impl CoreLoop { } } } - put_events.push((row_key, upd.body.as_slice(), outcome.prior_value)); + put_events.push((row_identity, upd.body.as_slice(), outcome.prior_value)); *affected += 1; } Err(e) => { diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/apply_support.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/apply_support.rs index 3845f4109..1eb0bb74a 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/apply_support.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/apply_support.rs @@ -7,15 +7,15 @@ use crate::data::executor::handlers::point::apply_put::PointPutOutcome; use crate::data::executor::handlers::rls_write_gate; use crate::data::executor::handlers::transaction::undo::UndoEntry; -use crate::engine::document::store::StorageKey; +use crate::engine::document::store::{RowIdentity, StorageKey}; use super::plan::MergePlanActions; /// One committed Phase-A put captured for post-commit event emission: -/// `(row_key, new stored body borrowed from the plan, prior stored value)`. -/// The body borrows from the merge plan (owned for the whole apply) rather than -/// being cloned. -pub(super) type MergePutEvent<'a> = (String, &'a [u8], Option>); +/// `(row identity, new stored body borrowed from the plan, prior stored value)`. +/// The identity is the one INSERT minted for the row. The body borrows from +/// the merge plan (owned for the whole apply) rather than being cloned. +pub(super) type MergePutEvent<'a> = (RowIdentity, &'a [u8], Option>); /// Record the in-memory index mutations a successful /// [`crate::data::executor::core_loop::CoreLoop::apply_point_put`] performed as diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs index f8d7526ce..81ce590c8 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs @@ -17,6 +17,7 @@ use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::enforcement::write_hook; use crate::data::executor::handlers::point::apply_delete::PointDeleteParams; use crate::data::executor::task::ExecutionTask; +use crate::engine::document::store::RowIdentity; use super::apply_support::returning_doc; use super::plan::MergeDelete; @@ -35,6 +36,9 @@ pub(super) struct MergeDeleteArms<'a> { pub(super) returning: bool, /// Join-key VALUE → target row surrogate, resolved on the Control Plane. pub(super) resolved_targets: &'a [nodedb_physical::physical_plan::ResolvedSumTarget], + /// The target's declared `PRIMARY KEY` column, when it has one. Names + /// each removed row in its event and redo entry. + pub(super) declared_primary_key: Option<&'a str>, } /// The statement-wide accumulators these arms contribute to, shared with the @@ -63,6 +67,7 @@ impl CoreLoop { has_vectors, returning, resolved_targets, + declared_primary_key, } = arms; let MergeDeleteTally { affected, @@ -73,6 +78,10 @@ impl CoreLoop { for del in deletes { let surrogate = del.key.surrogate(); let row_key = del.key.to_string(); + // The identity INSERT minted for this row, from the plan's + // captured MessagePack body: the declared primary key, else the + // decimal surrogate. + let row_identity = RowIdentity::of_stored_row(&del.body, declared_primary_key, del.key); // One write txn per arm: the removal and its index cascades // commit together, and a failing arm drops the txn // un-committed so it leaves nothing behind. @@ -153,6 +162,7 @@ impl CoreLoop { if has_vectors { write_set.push(WriteSetEntry { surrogate: surrogate.as_u32(), + identity: row_identity.clone(), is_delete: true, value: Vec::new(), collection: None, @@ -162,7 +172,7 @@ impl CoreLoop { self.emit_document_delete_event( task, collection, - del.key.to_identity(), + row_identity, outcome.prior_value.as_deref(), ); } diff --git a/nodedb/src/data/executor/handlers/point/delete.rs b/nodedb/src/data/executor/handlers/point/delete.rs index 66458181d..e24d63d59 100644 --- a/nodedb/src/data/executor/handlers/point/delete.rs +++ b/nodedb/src/data/executor/handlers/point/delete.rs @@ -344,6 +344,7 @@ mod tests { target_column: "balance".to_string(), join_column: "account_id".to_string(), value_expr: nodedb_query::expr::SqlExpr::Column("amount".to_string()), + declared_primary_key: None, } } diff --git a/nodedb/src/data/executor/handlers/point/insert.rs b/nodedb/src/data/executor/handlers/point/insert.rs index b62745f77..8ef89caf4 100644 --- a/nodedb/src/data/executor/handlers/point/insert.rs +++ b/nodedb/src/data/executor/handlers/point/insert.rs @@ -301,11 +301,13 @@ impl CoreLoop { // only writes the document; it no longer derives edges (which mis-homed // cross-shard edges by the document's vShard). + // The plan's `document_id` is the identity INSERT minted for this + // row, and the identity the WAL journals for it. self.emit_put_event( task, tid, collection, - storage_key.to_identity(), + document_identity.clone(), value, None, ); @@ -371,6 +373,7 @@ mod tests { target_column: "balance".to_string(), join_column: "account_id".to_string(), value_expr: nodedb_query::expr::SqlExpr::Column("amount".to_string()), + declared_primary_key: None, } } diff --git a/nodedb/src/data/executor/handlers/point/put.rs b/nodedb/src/data/executor/handlers/point/put.rs index 5eb6859a8..caa666913 100644 --- a/nodedb/src/data/executor/handlers/point/put.rs +++ b/nodedb/src/data/executor/handlers/point/put.rs @@ -214,11 +214,13 @@ impl CoreLoop { // Emit write event to Event Plane. Insert vs Update is derived // from whether `prior` was present — a PointPut onto an existing // row is an Update from every downstream consumer's perspective. + // The plan's `document_id` is the row's client identity, and the + // identity the WAL journals for it. self.emit_put_event( task, tid, collection, - storage_key.to_identity(), + document_identity.clone(), value, prior.prior_value.as_deref(), ); @@ -279,6 +281,7 @@ mod tests { target_column: "balance".to_string(), join_column: "account_id".to_string(), value_expr: nodedb_query::expr::SqlExpr::Column("amount".to_string()), + declared_primary_key: None, } } diff --git a/nodedb/src/data/executor/handlers/point/update/exec.rs b/nodedb/src/data/executor/handlers/point/update/exec.rs index 89668c2d6..e061079ab 100644 --- a/nodedb/src/data/executor/handlers/point/update/exec.rs +++ b/nodedb/src/data/executor/handlers/point/update/exec.rs @@ -270,7 +270,7 @@ impl CoreLoop { task, tid, collection, - storage_key.to_identity(), + document_identity.clone(), &updated_bytes, Some(¤t_bytes), ); @@ -316,6 +316,7 @@ impl CoreLoop { if has_vectors { response.write_set = vec![WriteSetEntry { surrogate: surrogate.as_u32(), + identity: document_identity, is_delete: false, value: updated_bytes, collection: None, @@ -378,6 +379,7 @@ mod tests { target_column: "balance".to_string(), join_column: "account_id".to_string(), value_expr: nodedb_query::expr::SqlExpr::Column("amount".to_string()), + declared_primary_key: None, } } diff --git a/nodedb/src/data/executor/handlers/transaction/batch.rs b/nodedb/src/data/executor/handlers/transaction/batch.rs index 8990d3a87..a284f8a7d 100644 --- a/nodedb/src/data/executor/handlers/transaction/batch.rs +++ b/nodedb/src/data/executor/handlers/transaction/batch.rs @@ -431,9 +431,8 @@ impl CoreLoop { /// Emit deferred trigger events for every write recorded in the /// committed transaction's undo log. `UndoEntry::{PutDocument, - /// DeleteDocument}.document_id` is the row's storage key, so a deferred - /// trigger converts it to the client-visible identity here — the same - /// conversion an immediate trigger sees via `emit_put_event` / + /// DeleteDocument}.identity` is the row's client identity, the same one + /// an immediate trigger sees via `emit_put_event` / /// `emit_document_delete_event`. fn emit_deferred_writes(&mut self, task: &ExecutionTask, undo_log: Vec) { use crate::data::executor::core_loop::deferred::DeferredWrite; @@ -442,7 +441,7 @@ impl CoreLoop { .filter_map(|entry| match entry { UndoEntry::PutDocument { collection, - document_id, + identity, old_value, .. } => Some(DeferredWrite { @@ -452,19 +451,19 @@ impl CoreLoop { } else { crate::event::WriteOp::Insert }, - identity: document_id.to_identity(), + identity, new_value: None, old_value, }), UndoEntry::DeleteDocument { collection, - document_id, + identity, old_value, .. } => Some(DeferredWrite { collection, op: crate::event::WriteOp::Delete, - identity: document_id.to_identity(), + identity, new_value: None, old_value: Some(old_value), }), @@ -531,6 +530,7 @@ mod tests { target_column: "balance".to_string(), join_column: "account_id".to_string(), value_expr: nodedb_query::expr::SqlExpr::Column("amount".to_string()), + declared_primary_key: None, } } diff --git a/nodedb/src/data/executor/handlers/transaction/index_write_values.rs b/nodedb/src/data/executor/handlers/transaction/index_write_values.rs index 4f0ccdf66..9e972c223 100644 --- a/nodedb/src/data/executor/handlers/transaction/index_write_values.rs +++ b/nodedb/src/data/executor/handlers/transaction/index_write_values.rs @@ -208,6 +208,7 @@ mod tests { UndoEntry::PutDocument { collection: collection.to_string(), document_id: nodedb_types::StorageKey::for_surrogate(Surrogate::new(1)), + identity: nodedb_types::StorageKey::for_surrogate(Surrogate::new(1)).to_identity(), old_value: None, bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), @@ -221,6 +222,7 @@ mod tests { UndoEntry::DeleteDocument { collection: collection.to_string(), document_id: nodedb_types::StorageKey::for_surrogate(Surrogate::new(2)), + identity: nodedb_types::StorageKey::for_surrogate(Surrogate::new(2)).to_identity(), old_value: Vec::new(), bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs index c4d8984d4..d4987ef01 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs @@ -131,6 +131,7 @@ impl CoreLoop { undo_log.push(UndoEntry::PutDocument { collection: target.collection, document_id: nodedb_types::StorageKey::for_surrogate(target.surrogate), + identity: target.identity, old_value: target.outcome.prior_value, bitemporal_sys_from_ms: target.outcome.bitemporal_sys_from_ms, bitemporal_index_tuples: target.outcome.bitemporal_index_tuples, @@ -161,6 +162,8 @@ impl CoreLoop { undo_log.push(UndoEntry::DeleteDocument { collection: collection.to_string(), document_id: nodedb_types::StorageKey::for_surrogate(surrogate), + // The plan's `document_id` is the row's client identity. + identity: nodedb_types::RowIdentity::from_user_key(document_id), old_value: old, bitemporal_sys_from_ms: outcome.bitemporal_sys_from_ms, bitemporal_index_tuples: outcome.bitemporal_index_tuples, diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs index e6c1d3d94..bb2ddc9e6 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs @@ -293,6 +293,7 @@ impl CoreLoop { undo_log.push(UndoEntry::PutDocument { collection: target.collection, document_id: nodedb_types::StorageKey::for_surrogate(target.surrogate), + identity: target.identity, old_value: target.outcome.prior_value, bitemporal_sys_from_ms: target.outcome.bitemporal_sys_from_ms, bitemporal_index_tuples: target.outcome.bitemporal_index_tuples, @@ -320,6 +321,8 @@ impl CoreLoop { undo_log.push(UndoEntry::PutDocument { collection: collection.to_string(), document_id: storage_key, + // The plan's `document_id` is the row's client identity. + identity: nodedb_types::RowIdentity::from_user_key(document_id), old_value: outcome.prior_value, bitemporal_sys_from_ms: outcome.bitemporal_sys_from_ms, bitemporal_index_tuples: outcome.bitemporal_index_tuples, diff --git a/nodedb/src/data/executor/handlers/transaction/undo/document.rs b/nodedb/src/data/executor/handlers/transaction/undo/document.rs index bcc16db7e..6e7eddf13 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/document.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/document.rs @@ -37,6 +37,9 @@ impl CoreLoop { UndoEntry::PutDocument { collection, document_id, + // Rollback restores prior storage state; it emits no event, so + // the row's client identity has no reader here. + identity: _, old_value, bitemporal_sys_from_ms, bitemporal_index_tuples, @@ -128,6 +131,9 @@ impl CoreLoop { UndoEntry::DeleteDocument { collection, document_id, + // Rollback restores prior storage state; it emits no event, so + // the row's client identity has no reader here. + identity: _, old_value, bitemporal_sys_from_ms, bitemporal_index_tuples, @@ -422,6 +428,7 @@ mod tests { let entry = UndoEntry::PutDocument { collection: "c".into(), document_id: d1, + identity: d1.to_identity(), old_value: None, bitemporal_sys_from_ms: Some(t), bitemporal_index_tuples: vec![("status".into(), "active".into())], @@ -476,6 +483,7 @@ mod tests { let entry = UndoEntry::DeleteDocument { collection: "c".into(), document_id: d1, + identity: d1.to_identity(), old_value: b"v1".to_vec(), bitemporal_sys_from_ms: Some(2_000), bitemporal_index_tuples: vec![("status".into(), "active".into())], @@ -511,6 +519,7 @@ mod tests { let restore = UndoEntry::PutDocument { collection: "c".into(), document_id: storage_key(0), + identity: storage_key(0).to_identity(), old_value: None, bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), @@ -528,6 +537,7 @@ mod tests { let genesis = UndoEntry::PutDocument { collection: "c".into(), document_id: storage_key(0), + identity: storage_key(0).to_identity(), old_value: None, bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), @@ -552,6 +562,7 @@ mod tests { let overwrite = UndoEntry::PutDocument { collection: "c".into(), document_id: key1, + identity: key1.to_identity(), old_value: Some(b"old".to_vec()), bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), @@ -571,6 +582,7 @@ mod tests { let insert = UndoEntry::PutDocument { collection: "c".into(), document_id: key2, + identity: key2.to_identity(), old_value: None, bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), @@ -592,6 +604,7 @@ mod tests { let entry = UndoEntry::DeleteDocument { collection: "c".into(), document_id: key1, + identity: key1.to_identity(), old_value: b"prior".to_vec(), bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), diff --git a/nodedb/src/data/executor/handlers/transaction/undo/entry.rs b/nodedb/src/data/executor/handlers/transaction/undo/entry.rs index ab205f59c..adc7957f3 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/entry.rs @@ -37,6 +37,8 @@ pub(in crate::data::executor) enum UndoEntry { /// The redb storage key. `.surrogate()` recovers the numeric surrogate /// FTS index rollback needs. document_id: nodedb_types::StorageKey, + /// The row's client identity, as the deferred event names it. + identity: nodedb_types::RowIdentity, /// `None` if the document didn't exist before (inserted); `Some(bytes)` /// if it was overwritten (updated). old_value: Option>, @@ -68,6 +70,8 @@ pub(in crate::data::executor) enum UndoEntry { /// delete cascade removed this document's postings, and a /// rolled-back delete recomputes and re-inserts them under it. document_id: nodedb_types::StorageKey, + /// The row's client identity, as the deferred event names it. + identity: nodedb_types::RowIdentity, old_value: Vec, /// System-time key of the versioned tombstone row this op appended on a /// bitemporal collection. `None` = plain op → re-insert via the diff --git a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs index c5866efbd..1aec73948 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs @@ -207,6 +207,7 @@ mod tests { UndoEntry::PutDocument { collection: "c".into(), document_id: d1, + identity: d1.to_identity(), old_value: None, bitemporal_sys_from_ms: Some(1_000), bitemporal_index_tuples: vec![("status".into(), "active".into())], @@ -217,6 +218,7 @@ mod tests { UndoEntry::DeleteDocument { collection: "c".into(), document_id: d1, + identity: d1.to_identity(), old_value: b"v1".to_vec(), bitemporal_sys_from_ms: Some(2_000), bitemporal_index_tuples: vec![("status".into(), "active".into())], diff --git a/nodedb/src/data/executor/handlers/truncate.rs b/nodedb/src/data/executor/handlers/truncate.rs index 5b90269d3..745be0df3 100644 --- a/nodedb/src/data/executor/handlers/truncate.rs +++ b/nodedb/src/data/executor/handlers/truncate.rs @@ -2,15 +2,28 @@ //! TRUNCATE and ESTIMATE_COUNT handlers. +use nodedb_physical::physical_plan::{ResolvedSumTarget, StorageMode}; use tracing::{debug, warn}; use crate::bridge::envelope::{ErrorCode, Response, WriteSetEntry}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::enforcement::materialized_sum::divergence::SumTargetCheck; use crate::data::executor::enforcement::write_hook; +use crate::data::executor::handlers::transaction::stage_write::stored_row_identity; use crate::data::executor::response_codec; use crate::data::executor::task::ExecutionTask; +/// Borrowed arguments for [`CoreLoop::execute_truncate`]. +pub(in crate::data::executor) struct TruncateParams<'a> { + pub collection: &'a str, + /// Join-key VALUE → target row surrogate for every materialized-sum + /// target the removed rows contribute to, resolved on the Control Plane. + pub resolved_sum_targets: &'a [ResolvedSumTarget], + /// The collection's declared `PRIMARY KEY` column, when it has one. Names + /// each removed row in its redo entry and delete event. + pub declared_primary_key: Option<&'a str>, +} + impl CoreLoop { /// TRUNCATE: delete all documents in a collection without filter scanning. /// @@ -28,9 +41,13 @@ impl CoreLoop { &mut self, task: &ExecutionTask, tid: u64, - collection: &str, - resolved_sum_targets: &[nodedb_physical::physical_plan::ResolvedSumTarget], + params: TruncateParams<'_>, ) -> Response { + let TruncateParams { + collection, + resolved_sum_targets, + declared_primary_key, + } = params; debug!(core = self.core_id, %collection, "truncate"); // Collect all document IDs in this collection. @@ -81,6 +98,21 @@ impl CoreLoop { let has_vectors = self.collection_has_vectors(database_id, tid, collection); + // The stored pre-image is a Binary Tuple on a strict collection and + // MessagePack otherwise. Hoisted once so each removed row's identity is + // read through the matching decoder. + let strict_schema = self + .doc_configs + .get(&( + crate::types::DatabaseId::new(database_id), + crate::types::TenantId::new(tid), + collection.to_string(), + )) + .and_then(|c| match &c.storage_mode { + StorageMode::Strict { schema } => Some(schema.clone()), + StorageMode::Schemaless => None, + }); + // BALANCED, decided over every row about to be removed and BEFORE the // first removal — each row below commits in its own transaction, so a // check after the loop could not undo what it found. Emptying a @@ -156,6 +188,16 @@ impl CoreLoop { write_set.extend(write_hook::target_write_set(&target_writes)); if let Some(deleted_bytes) = deleted_bytes.as_deref() { let surrogate = storage_key.surrogate(); + // The identity INSERT minted for this row, read from the + // pre-image the delete returned: the declared primary key + // when the collection declares one, else the decimal + // surrogate. The redo entry and the delete event share it. + let row_identity = stored_row_identity( + deleted_bytes, + strict_schema.as_ref(), + declared_primary_key, + *storage_key, + ); if let Err(e) = self.inverted.remove_document( database_id, crate::types::TenantId::new(tid), @@ -181,6 +223,7 @@ impl CoreLoop { self.remove_document_vector_indexes(database_id, tid, collection, *storage_key); write_set.push(WriteSetEntry { surrogate: surrogate.as_u32(), + identity: row_identity.clone(), is_delete: true, value: Vec::new(), collection: None, @@ -223,11 +266,10 @@ impl CoreLoop { collection, deleted_bytes, ); - let identity = storage_key.to_identity(); self.emit_document_delete_event( task, collection, - identity, + row_identity, Some(old_converted.as_deref().unwrap_or(deleted_bytes)), ); truncated += 1; diff --git a/nodedb/src/data/executor/handlers/update_from_join.rs b/nodedb/src/data/executor/handlers/update_from_join.rs index 1f98db2b8..3cbd5b833 100644 --- a/nodedb/src/data/executor/handlers/update_from_join.rs +++ b/nodedb/src/data/executor/handlers/update_from_join.rs @@ -234,7 +234,8 @@ impl CoreLoop { target_collection, resolved_sum_targets, has_vectors, - is_strict: strict_schema.is_some(), + strict_schema: strict_schema.as_ref(), + declared_primary_key, want_returning: returning.is_some(), }, rows, diff --git a/nodedb/src/data/executor/handlers/update_from_join_write.rs b/nodedb/src/data/executor/handlers/update_from_join_write.rs index a1ed6cdae..01fe76e60 100644 --- a/nodedb/src/data/executor/handlers/update_from_join_write.rs +++ b/nodedb/src/data/executor/handlers/update_from_join_write.rs @@ -5,12 +5,14 @@ //! into its materialized-sum target and re-indexing its vectors. use nodedb_physical::physical_plan::ResolvedSumTarget; +use nodedb_types::columnar::StrictSchema; use crate::bridge::envelope::{ErrorCode, Response, WriteSetEntry}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::enforcement::write_hook; use crate::data::executor::handlers::point::update_reindex_vector::UpdateVectorReindex; use crate::data::executor::handlers::returning_doc; +use crate::data::executor::handlers::transaction::stage_write::stored_row_identity; use crate::data::executor::task::ExecutionTask; use super::update_from_join_types::ResolvedUpdateRow; @@ -35,7 +37,10 @@ pub(in crate::data::executor) struct WriteResolvedRowsCtx<'a> { pub target_collection: &'a str, pub resolved_sum_targets: &'a [ResolvedSumTarget], pub has_vectors: bool, - pub is_strict: bool, + /// The target collection's strict schema, when it stores Binary Tuples. + pub strict_schema: Option<&'a StrictSchema>, + /// The target collection's declared `PRIMARY KEY` column, when it has one. + pub declared_primary_key: Option<&'a str>, pub want_returning: bool, } @@ -55,9 +60,11 @@ impl CoreLoop { target_collection, resolved_sum_targets, has_vectors, - is_strict, + strict_schema, + declared_primary_key, want_returning, } = ctx; + let is_strict = strict_schema.is_some(); let database_id = task.request.database_id.as_u64(); let config_key = ( crate::types::DatabaseId::new(database_id), @@ -180,7 +187,16 @@ impl CoreLoop { // `collect_update_from_join_rows`; `emit_put_event` derives // `WriteOp::Update` from the Some prior + Some new pair and // handles strict->msgpack conversion on both sides. - let row_identity = storage_key.to_identity(); + // + // The identity is the one INSERT minted: the declared primary + // key when the collection declares one, else the decimal + // surrogate. The redo entry below journals the same identity. + let row_identity = stored_row_identity( + &updated_bytes, + strict_schema, + declared_primary_key, + storage_key, + ); // `row_identity` is read again below for `RETURNING`'s `id` // field, so the event-emit boundary gets a clone rather than // the move. @@ -212,6 +228,7 @@ impl CoreLoop { } write_set.push(WriteSetEntry { surrogate: storage_key.surrogate().as_u32(), + identity: row_identity.clone(), is_delete: false, value: updated_bytes, collection: None, diff --git a/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs b/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs index 3728b029a..651f32639 100644 --- a/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs +++ b/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs @@ -199,6 +199,7 @@ mod tests { target_column: "balance".to_string(), join_column: "account_id".to_string(), value_expr: nodedb_query::expr::SqlExpr::Column("amount".to_string()), + declared_primary_key: None, } } diff --git a/nodedb/src/data/executor/handlers/upsert/exec/insert.rs b/nodedb/src/data/executor/handlers/upsert/exec/insert.rs index fe24da059..02a7d30df 100644 --- a/nodedb/src/data/executor/handlers/upsert/exec/insert.rs +++ b/nodedb/src/data/executor/handlers/upsert/exec/insert.rs @@ -56,7 +56,8 @@ impl CoreLoop { } = ctx; let storage_key = StorageKey::for_surrogate(surrogate); - let row_identity = storage_key.to_identity(); + // The plan's `document_id` is the row's client identity: the write + // gate, the event, the redo entry, and `RETURNING` all name it. let document_identity = RowIdentity::from_user_key(document_id); // Insert: document doesn't exist, create new (same as PointPut). @@ -67,7 +68,7 @@ impl CoreLoop { if let Err(e) = rls_write_gate::admit_stored_row( rls_write_check, value, - &row_identity, + &document_identity, None, tid, collection, @@ -197,7 +198,7 @@ impl CoreLoop { task, tid, collection, - row_identity, + document_identity.clone(), value, prior.prior_value.as_deref(), ); @@ -221,6 +222,7 @@ impl CoreLoop { if has_vectors { response.write_set = vec![WriteSetEntry { surrogate: surrogate.as_u32(), + identity: document_identity, is_delete: false, value: value.to_vec(), collection: None, diff --git a/nodedb/src/data/executor/handlers/upsert/exec/overwrite.rs b/nodedb/src/data/executor/handlers/upsert/exec/overwrite.rs index 0fd220e71..f81bcd228 100644 --- a/nodedb/src/data/executor/handlers/upsert/exec/overwrite.rs +++ b/nodedb/src/data/executor/handlers/upsert/exec/overwrite.rs @@ -63,7 +63,8 @@ impl CoreLoop { } = ctx; let storage_key = StorageKey::for_surrogate(surrogate); - let row_identity = storage_key.to_identity(); + // The plan's `document_id` is the row's client identity: the write + // gate, the event, the redo entry, and `RETURNING` all name it. let document_identity = RowIdentity::from_user_key(document_id); // Decode existing document to nodedb_types::Value. @@ -156,7 +157,7 @@ impl CoreLoop { if let Err(e) = rls_write_gate::admit_stored_row( rls_write_check, &merged_body, - &row_identity, + &document_identity, None, tid, collection, @@ -269,7 +270,7 @@ impl CoreLoop { task, tid, collection, - row_identity, + document_identity.clone(), &stored_bytes, Some(¤t_bytes), ); @@ -296,6 +297,7 @@ impl CoreLoop { if has_vectors { response.write_set = vec![WriteSetEntry { surrogate: surrogate.as_u32(), + identity: document_identity, is_delete: false, value: merged_body, collection: None, diff --git a/nodedb/src/data/executor/handlers/write_batch.rs b/nodedb/src/data/executor/handlers/write_batch.rs index e38588be0..4be2e97df 100644 --- a/nodedb/src/data/executor/handlers/write_batch.rs +++ b/nodedb/src/data/executor/handlers/write_batch.rs @@ -167,7 +167,7 @@ impl CoreLoop { // bytes captured per row above. if let PhysicalPlan::Document(DocumentOp::PointPut { collection, - surrogate, + document_id, value, .. }) = task.plan() @@ -177,9 +177,12 @@ impl CoreLoop { Ok(p) => p.prior_value.as_deref(), Err(_) => None, }; - let identity = - crate::engine::document::store::StorageKey::for_surrogate(*surrogate) - .to_identity(); + // The plan's `document_id` is the row's client identity, + // the same one `execute_point_put` emits and the WAL + // journals. + let identity = crate::engine::document::store::RowIdentity::from_user_key( + document_id.as_str(), + ); self.emit_put_event(task, tid, collection.as_str(), identity, value, prior); } self.response_ok(task) diff --git a/nodedb/src/data/executor/wal_replay/crdt.rs b/nodedb/src/data/executor/wal_replay/crdt.rs index 58cc10014..bcf3463c7 100644 --- a/nodedb/src/data/executor/wal_replay/crdt.rs +++ b/nodedb/src/data/executor/wal_replay/crdt.rs @@ -305,7 +305,7 @@ impl CoreLoop { } }; - if let Some((_document_id, surrogate, Some(bytes))) = projection { + if let Some((document_id, surrogate, Some(bytes))) = projection { let task = Self::replay_task( tid, database_id, @@ -325,6 +325,7 @@ impl CoreLoop { &task, tid.as_u64(), collection, + document_id, surrogate, &bytes, ); diff --git a/nodedb/src/event/audit_dml/consumer.rs b/nodedb/src/event/audit_dml/consumer.rs index bec25b8c6..6024b95c6 100644 --- a/nodedb/src/event/audit_dml/consumer.rs +++ b/nodedb/src/event/audit_dml/consumer.rs @@ -99,7 +99,7 @@ mod tests { sequence: 1, collection: Arc::from("orders"), op, - row_id: RowId::new("o-1"), + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("o-1")), lsn: Lsn::new(100), database_id: DatabaseId::new(42), tenant_id: TenantId::new(1), diff --git a/nodedb/src/event/bus.rs b/nodedb/src/event/bus.rs index e3cdde9cc..cb69f0417 100644 --- a/nodedb/src/event/bus.rs +++ b/nodedb/src/event/bus.rs @@ -222,7 +222,7 @@ mod tests { sequence: seq, collection: Arc::from("test"), op: WriteOp::Insert, - row_id: RowId::new("row-1"), + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("row-1")), lsn: Lsn::new(seq), database_id: DatabaseId::new(7), tenant_id: TenantId::new(1), diff --git a/nodedb/src/event/cdc/router.rs b/nodedb/src/event/cdc/router.rs index 1f3cfeff8..edea07f20 100644 --- a/nodedb/src/event/cdc/router.rs +++ b/nodedb/src/event/cdc/router.rs @@ -336,7 +336,9 @@ mod tests { sequence: seq, collection: Arc::from(collection), op: WriteOp::Insert, - row_id: RowId::new(format!("row-{seq}")), + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key(format!( + "row-{seq}" + ))), lsn: Lsn::new(seq * 10), database_id: DatabaseId::new(7), tenant_id: TenantId::new(1), diff --git a/nodedb/src/event/consumer.rs b/nodedb/src/event/consumer.rs index 6ebf9f6e1..fbdba1d84 100644 --- a/nodedb/src/event/consumer.rs +++ b/nodedb/src/event/consumer.rs @@ -599,7 +599,7 @@ mod tests { sequence: seq, collection: Arc::from("test"), op: WriteOp::Insert, - row_id: RowId::new("row-1"), + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("row-1")), lsn: Lsn::new(seq * 10), database_id: DatabaseId::new(7), tenant_id: TenantId::new(1), diff --git a/nodedb/src/event/consumer_helpers.rs b/nodedb/src/event/consumer_helpers.rs index 5d57a76c4..872e60d31 100644 --- a/nodedb/src/event/consumer_helpers.rs +++ b/nodedb/src/event/consumer_helpers.rs @@ -310,7 +310,7 @@ mod tests { sequence: 1, collection: Arc::from("events"), op, - row_id: RowId::new("row-1"), + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("row-1")), lsn: Lsn::new(1), database_id: DatabaseId::DEFAULT, tenant_id: TenantId::new(1), diff --git a/nodedb/src/event/crdt_sync/packager.rs b/nodedb/src/event/crdt_sync/packager.rs index 399ee4b2c..41bb39f97 100644 --- a/nodedb/src/event/crdt_sync/packager.rs +++ b/nodedb/src/event/crdt_sync/packager.rs @@ -169,7 +169,7 @@ mod tests { sequence: 1, collection: Arc::from("orders"), op, - row_id: RowId::new("o-1"), + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("o-1")), lsn: Lsn::new(100), database_id: DatabaseId::new(7), tenant_id: TenantId::new(1), diff --git a/nodedb/src/event/plane.rs b/nodedb/src/event/plane.rs index ff4fdedf2..ee0676fc2 100644 --- a/nodedb/src/event/plane.rs +++ b/nodedb/src/event/plane.rs @@ -421,7 +421,7 @@ mod tests { sequence: seq, collection: Arc::from("test"), op: WriteOp::Insert, - row_id: RowId::new("row-1"), + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("row-1")), lsn: Lsn::new(seq * 10), database_id: DatabaseId::new(7), tenant_id: TenantId::new(1), diff --git a/nodedb/src/event/types.rs b/nodedb/src/event/types.rs index b764bfd00..5c4172d8c 100644 --- a/nodedb/src/event/types.rs +++ b/nodedb/src/event/types.rs @@ -10,30 +10,101 @@ use std::sync::Arc; +use nodedb_types::RowIdentity; use sonic_rs; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; -/// Identifies a row within a collection. Wraps the document/row ID string. +/// The row an event names, as the client sees it. /// -/// Separate from `DocumentId` (nodedb-types) because event row IDs are -/// ephemeral references into the event payload, not owned document handles. +/// A `Row` carries the same [`RowIdentity`] INSERT minted for the row: the +/// declared `PRIMARY KEY` value when the collection declares one, else the +/// decimal surrogate. The live emit path and WAL replay both build it from +/// that identity, never from a storage key. #[derive(Debug, Clone, PartialEq, Eq, Hash)] -pub struct RowId(pub Arc); +pub enum RowId { + /// A document, KV, or graph-node row. + Row(RowIdentity), + /// A KV batch op that names no single row. + Batch, + /// A graph edge, named by its endpoints and label. Boxed so every + /// event on the ring pays for one identity, not four strings. + Edge(Box), + /// An idle heartbeat, which names no row. + Heartbeat, +} + +/// Text a [`RowId::Batch`] renders as. +const BATCH_ROW_ID: &str = "_batch"; + +/// A graph edge's `(src, label, dst)` identity with its composite text. +/// +/// The text is what [`crate::event::graph_cdc::edge_row_id`] builds, rendered +/// once at construction so `as_str` allocates nothing. +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +pub struct EdgeRowId { + src: String, + label: String, + dst: String, + rendered: String, +} + +impl EdgeRowId { + pub fn src(&self) -> &str { + &self.src + } + + pub fn label(&self) -> &str { + &self.label + } + + pub fn dst(&self) -> &str { + &self.dst + } + + /// The composite `src\u{1}label\u{1}dst` text. + pub fn as_str(&self) -> &str { + &self.rendered + } +} impl RowId { - pub fn new(id: impl Into>) -> Self { - Self(id.into()) + /// Name a single row by its client identity. + pub fn row(identity: RowIdentity) -> Self { + Self::Row(identity) } + /// Name a graph edge by its `(src, label, dst)` triple. + pub fn edge(src: impl Into, label: impl Into, dst: impl Into) -> Self { + let src = src.into(); + let label = label.into(); + let dst = dst.into(); + let rendered = crate::event::graph_cdc::edge_row_id(&src, &label, &dst); + Self::Edge(Box::new(EdgeRowId { + src, + label, + dst, + rendered, + })) + } + + /// The row id as text, without allocating. + /// + /// `Row` yields the identity text, `Batch` yields `"_batch"`, `Edge` + /// yields the rendered composite, and `Heartbeat` yields `""`. pub fn as_str(&self) -> &str { - &self.0 + match self { + Self::Row(identity) => identity.as_str(), + Self::Batch => BATCH_ROW_ID, + Self::Edge(edge) => edge.as_str(), + Self::Heartbeat => "", + } } } impl std::fmt::Display for RowId { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.write_str(&self.0) + f.write_str(self.as_str()) } } @@ -200,11 +271,40 @@ mod tests { #[test] fn row_id_display() { - let id = RowId::new("doc-123"); + let id = RowId::row(RowIdentity::from_user_key("doc-123")); assert_eq!(id.as_str(), "doc-123"); assert_eq!(id.to_string(), "doc-123"); } + #[test] + fn row_id_surrogate_identity_is_decimal() { + let id = RowId::row(RowIdentity::for_surrogate(nodedb_types::Surrogate::new(9))); + assert_eq!(id.as_str(), "9"); + } + + #[test] + fn row_id_batch_and_heartbeat_text() { + assert_eq!(RowId::Batch.as_str(), "_batch"); + assert_eq!(RowId::Heartbeat.as_str(), ""); + } + + #[test] + fn row_id_edge_matches_graph_cdc_composition() { + let id = RowId::edge("a", "KNOWS", "b"); + assert_eq!( + id.as_str(), + crate::event::graph_cdc::edge_row_id("a", "KNOWS", "b").as_str() + ); + match id { + RowId::Edge(edge) => { + assert_eq!(edge.src(), "a"); + assert_eq!(edge.label(), "KNOWS"); + assert_eq!(edge.dst(), "b"); + } + other => panic!("expected edge row id, got {other:?}"), + } + } + #[test] fn write_op_display() { assert_eq!(WriteOp::Insert.to_string(), "INSERT"); @@ -226,7 +326,7 @@ mod tests { sequence: 1, collection: Arc::from("orders"), op: WriteOp::Insert, - row_id: RowId::new("order-1"), + row_id: RowId::row(RowIdentity::from_user_key("order-1")), lsn: Lsn::new(100), database_id: DatabaseId::DEFAULT, tenant_id: TenantId::new(1), diff --git a/nodedb/src/event/wal_replay.rs b/nodedb/src/event/wal_replay.rs index 1300bf461..524a9855c 100644 --- a/nodedb/src/event/wal_replay.rs +++ b/nodedb/src/event/wal_replay.rs @@ -403,6 +403,30 @@ mod tests { assert!(record_to_events(&record, &mut seq).is_empty()); } + #[test] + fn document_put_replays_journaled_identity_verbatim() { + // The current 5-tuple arity carries the row's `RowIdentity` text as + // `document_id`. A declared-PK string and a decimal surrogate both + // replay as `RowId::row(from_user_key(..))`, never reinterpreted. + use crate::event::types::RowId; + use nodedb_types::RowIdentity; + for (journaled, lsn) in [("order-1", 210u64), ("9", 211u64)] { + let provenance: Option = None; + let payload = + zerompk::to_msgpack_vec(&("orders", journaled, b"value", provenance, 9u32)) + .unwrap(); + let record = make_record(RecordType::Put, &payload, 1, 0, lsn); + let mut seq = 0u64; + let event = one_event(&record, &mut seq); + assert_eq!( + event.row_id, + RowId::row(RowIdentity::from_user_key(journaled)), + "replayed row id is the journaled identity text" + ); + assert_eq!(event.row_id.as_str(), journaled); + } + } + #[test] fn parse_document_put_with_provenance() { // New 4-element arity: (collection, document_id, value, Option). diff --git a/nodedb/src/event/wal_replay_parse.rs b/nodedb/src/event/wal_replay_parse.rs index 3e698aa7c..aa5e6e4dd 100644 --- a/nodedb/src/event/wal_replay_parse.rs +++ b/nodedb/src/event/wal_replay_parse.rs @@ -15,6 +15,7 @@ use std::sync::Arc; +use nodedb_types::RowIdentity; use nodedb_types::sync::wire::SyncProvenance; use tracing::warn; @@ -106,7 +107,7 @@ pub(super) fn parse_put_record( sequence: *sequence, collection: Arc::from(collection.as_str()), op: WriteOp::Insert, - row_id: RowId::new(key_str.as_ref()), + row_id: RowId::row(RowIdentity::from_user_key(key_str.into_owned())), lsn, database_id, tenant_id, @@ -131,7 +132,7 @@ pub(super) fn parse_put_record( op: WriteOp::BulkInsert { count: entries.len() as u32, }, - row_id: RowId::new("_batch"), + row_id: RowId::Batch, lsn, database_id, tenant_id, @@ -148,8 +149,9 @@ pub(super) fn parse_put_record( // Try document put with surrogate (current arity): // (collection, document_id, value, provenance, surrogate_u32). The trailing - // surrogate is consumed by the Data Plane's vector-index replay; the event - // stream keys on `document_id`, so it is ignored here. + // surrogate is consumed by the Data Plane's vector-index replay. The event + // stream keys on `document_id`, which the writer journals as the row's + // `RowIdentity` text, so it is wrapped verbatim and never reinterpreted. if let Ok((collection, document_id, value, _prov, _surrogate)) = zerompk::from_msgpack::<(String, String, Vec, Option, u32)>(payload) { @@ -160,7 +162,7 @@ pub(super) fn parse_put_record( sequence: *sequence, collection: Arc::from(collection.as_str()), op: WriteOp::Insert, - row_id: RowId::new(document_id.as_str()), + row_id: RowId::row(RowIdentity::from_user_key(document_id)), lsn, database_id, tenant_id, @@ -186,7 +188,7 @@ pub(super) fn parse_put_record( sequence: *sequence, collection: Arc::from(collection.as_str()), op: WriteOp::Insert, - row_id: RowId::new(document_id.as_str()), + row_id: RowId::row(RowIdentity::from_user_key(document_id)), lsn, database_id, tenant_id, @@ -215,7 +217,7 @@ pub(super) fn parse_put_record( sequence: *sequence, collection: Arc::from(collection.as_str()), op: WriteOp::Insert, - row_id: RowId::new(document_id.as_str()), + row_id: RowId::row(RowIdentity::from_user_key(document_id)), lsn, database_id, tenant_id, @@ -247,9 +249,7 @@ pub(super) fn parse_put_record( sequence: *sequence, collection: Arc::from(collection.as_str()), op: WriteOp::Insert, - row_id: RowId::new( - crate::event::graph_cdc::edge_row_id(&src_id, &label, &dst_id).as_str(), - ), + row_id: RowId::edge(src_id, label, dst_id), lsn, database_id, tenant_id, @@ -314,7 +314,7 @@ pub(super) fn parse_graph_node_label_record( sequence: *sequence, collection: Arc::from(crate::event::graph_cdc::GRAPH_LABEL_STREAM), op, - row_id: RowId::new(node_id.as_str()), + row_id: RowId::row(RowIdentity::from_user_key(node_id)), lsn, database_id, tenant_id, @@ -350,7 +350,7 @@ pub(super) fn parse_delete_record( op: WriteOp::BulkDelete { count: keys.len() as u32, }, - row_id: RowId::new("_batch"), + row_id: RowId::Batch, lsn, database_id, tenant_id, @@ -376,7 +376,7 @@ pub(super) fn parse_delete_record( sequence: *sequence, collection: Arc::from(collection.as_str()), op: WriteOp::Delete, - row_id: RowId::new(document_id.as_str()), + row_id: RowId::row(RowIdentity::from_user_key(document_id)), lsn, database_id, tenant_id, @@ -400,7 +400,7 @@ pub(super) fn parse_delete_record( sequence: *sequence, collection: Arc::from(collection.as_str()), op: WriteOp::Delete, - row_id: RowId::new(document_id.as_str()), + row_id: RowId::row(RowIdentity::from_user_key(document_id)), lsn, database_id, tenant_id, @@ -422,7 +422,7 @@ pub(super) fn parse_delete_record( sequence: *sequence, collection: Arc::from(collection.as_str()), op: WriteOp::Delete, - row_id: RowId::new(document_id.as_str()), + row_id: RowId::row(RowIdentity::from_user_key(document_id)), lsn, database_id, tenant_id, @@ -450,9 +450,7 @@ pub(super) fn parse_delete_record( sequence: *sequence, collection: Arc::from(collection.as_str()), op: WriteOp::Delete, - row_id: RowId::new( - crate::event::graph_cdc::edge_row_id(&src_id, &label, &dst_id).as_str(), - ), + row_id: RowId::edge(src_id, label, dst_id), lsn, database_id, tenant_id, diff --git a/nodedb/src/query/materialized_sum_delta.rs b/nodedb/src/query/materialized_sum_delta.rs index 0a5885434..dd33fec31 100644 --- a/nodedb/src/query/materialized_sum_delta.rs +++ b/nodedb/src/query/materialized_sum_delta.rs @@ -107,6 +107,7 @@ mod tests { target_column: "balance".to_string(), join_column: "account_id".to_string(), value_expr: nodedb_query::expr::SqlExpr::Column("amount".to_string()), + declared_primary_key: None, } } diff --git a/nodedb/src/query/materialized_sum_images.rs b/nodedb/src/query/materialized_sum_images.rs index 5d09d68cc..404df2816 100644 --- a/nodedb/src/query/materialized_sum_images.rs +++ b/nodedb/src/query/materialized_sum_images.rs @@ -276,6 +276,7 @@ mod tests { target_column: "balance".to_string(), join_column: "account_id".to_string(), value_expr: nodedb_query::expr::SqlExpr::Column("amount".to_string()), + declared_primary_key: None, } } diff --git a/nodedb/src/query/materialized_sum_keys.rs b/nodedb/src/query/materialized_sum_keys.rs index d551b5e6f..ac2d3b4a4 100644 --- a/nodedb/src/query/materialized_sum_keys.rs +++ b/nodedb/src/query/materialized_sum_keys.rs @@ -122,6 +122,7 @@ mod tests { target_column: "balance".to_string(), join_column: "account_id".to_string(), value_expr: nodedb_query::expr::SqlExpr::Column("amount".to_string()), + declared_primary_key: None, } } diff --git a/nodedb/tests/inproc/cases/bitemporal_cdc.rs b/nodedb/tests/inproc/cases/bitemporal_cdc.rs index 23130f75a..1cd1ec7df 100644 --- a/nodedb/tests/inproc/cases/bitemporal_cdc.rs +++ b/nodedb/tests/inproc/cases/bitemporal_cdc.rs @@ -41,7 +41,7 @@ fn write_event(seq: u64, op: WriteOp, payload_bytes: Vec, is_delete: bool) - sequence: seq, collection: Arc::from("users"), op, - row_id: RowId::new("u-1"), + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("u-1")), lsn: Lsn::new(seq * 10), database_id: DatabaseId::new(7), tenant_id: TenantId::new(1), diff --git a/nodedb/tests/inproc/cases/cdc_arc_fanout.rs b/nodedb/tests/inproc/cases/cdc_arc_fanout.rs index b1cdfe375..59443a36b 100644 --- a/nodedb/tests/inproc/cases/cdc_arc_fanout.rs +++ b/nodedb/tests/inproc/cases/cdc_arc_fanout.rs @@ -56,7 +56,7 @@ fn write_event(seq: u64) -> WriteEvent { sequence: seq, collection: Arc::from("orders"), op: WriteOp::Insert, - row_id: RowId::new(format!("r-{seq}")), + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key(format!("r-{seq}"))), lsn: Lsn::new(seq * 10), database_id: DatabaseId::new(7), tenant_id: TenantId::new(1), diff --git a/nodedb/tests/inproc/cases/event_trigger.rs b/nodedb/tests/inproc/cases/event_trigger.rs index b5de3f6e4..a5878d016 100644 --- a/nodedb/tests/inproc/cases/event_trigger.rs +++ b/nodedb/tests/inproc/cases/event_trigger.rs @@ -28,7 +28,7 @@ fn make_event(source: EventSource, op: WriteOp, collection: &str) -> WriteEvent sequence: 1, collection: Arc::from(collection), op, - row_id: RowId::new("row-1"), + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("row-1")), lsn: Lsn::new(100), database_id: DatabaseId::new(7), tenant_id: TenantId::new(1), diff --git a/nodedb/tests/inproc/cases/shutdown_event_plane.rs b/nodedb/tests/inproc/cases/shutdown_event_plane.rs index 2026595d5..5a9d68e2a 100644 --- a/nodedb/tests/inproc/cases/shutdown_event_plane.rs +++ b/nodedb/tests/inproc/cases/shutdown_event_plane.rs @@ -34,7 +34,7 @@ fn make_write_event(seq: u64, lsn_val: u64) -> WriteEvent { sequence: seq, collection: Arc::from("test_collection"), op: WriteOp::Insert, - row_id: RowId::new("row-1"), + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("row-1")), lsn: Lsn::new(lsn_val), database_id: DatabaseId::DEFAULT, tenant_id: TenantId::new(1), From 0421065b63cd96782845a8469e18a82acf15b385 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 12 Sep 2026 22:18:41 +0800 Subject: [PATCH 17/17] refactor(identity): remove thin StorageKey wrapper functions Replace surrogate_to_doc_id, doc_id_to_surrogate, and identity_of with direct calls to StorageKey::for_surrogate, StorageKey::parse, and StorageKey::to_identity at every call site. The wrappers only saved a method call and no longer earn their keep now that most callers already hold a StorageKey. Moves the otel receiver off its hand-rolled hex shim onto the workspace hex crate, promoted from dev-dependencies to a regular dependency of nodedb. --- .../cases/cross_node_pk_lookup.rs | 4 +- .../cases/vector_index_dispatch_cross_node.rs | 5 +- nodedb-graph/src/csr/index/interning.rs | 2 +- nodedb-types/src/lib.rs | 4 +- nodedb-types/src/row_identity.rs | 49 +++---------- nodedb/Cargo.toml | 4 +- nodedb/src/control/otel/receiver.rs | 15 ---- .../server/http/routes/query_stream.rs | 13 +++- .../server/native/session/session_stream.rs | 12 ++-- .../server/pgwire/handler/stream_response.rs | 22 +++++- .../control/server/response_shape/compose.rs | 54 ++++++++------ .../control/server/response_shape/project.rs | 17 +++-- .../server/response_translate/text_hybrid.rs | 2 +- .../neutral/query_functions/balance_as_of.rs | 2 +- .../convert_currency_lookup.rs | 2 +- .../ddl/neutral/query_functions/helpers.rs | 6 +- .../query_functions/temporal_lookup.rs | 2 +- .../neutral/query_functions/verify_balance.rs | 4 +- .../src/control/server/wal_dispatch/core.rs | 6 +- .../executor/core_loop/doc_config_seed.rs | 6 +- .../executor/dispatch/bitmap/materialize.rs | 4 +- .../enforcement/materialized_sum/apply.rs | 15 ++-- .../handlers/document/write/batch_insert.rs | 14 ++-- .../executor/handlers/point/apply_put/core.rs | 14 ++-- nodedb/src/data/executor/handlers/spatial.rs | 70 +++++++++---------- .../transaction/overlay/columnar_merge.rs | 4 +- .../handlers/transaction/resolve/document.rs | 2 +- .../handlers/transaction/resolve/entry.rs | 36 +++++----- .../transaction/stage_write/stage_kv.rs | 18 +---- .../handlers/transaction/undo/rollback.rs | 2 +- nodedb/src/data/executor/wal_replay_fts.rs | 2 +- .../src/data/executor/wal_replay_spatial.rs | 17 +++-- nodedb/src/engine/document/store/key.rs | 8 +-- nodedb/src/engine/document/store/mod.rs | 2 +- .../graph/pattern/executor/core/triple.rs | 4 +- .../graph/pattern/executor/predicates.rs | 12 ++-- nodedb/src/engine/sparse/btree/chain_head.rs | 2 +- .../cross_engine_three_way_fts_vector_doc.rs | 2 +- .../executor_tests/test_conditional_update.rs | 2 +- .../test_range_scan_bitemporal.rs | 2 +- nodedb/tests/wire/cases/sql_hybrid_search.rs | 2 +- 41 files changed, 230 insertions(+), 235 deletions(-) diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cross_node_pk_lookup.rs b/nodedb-cluster-tests/tests/common_suite/cases/cross_node_pk_lookup.rs index 7ead95bf0..d6430d6b7 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cross_node_pk_lookup.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cross_node_pk_lookup.rs @@ -8,8 +8,8 @@ //! (`surrogate_pk{,_rev}_v3`) is SHARDED to the collection's data-group //! members. `document_strict` collections are single-vShard-homed, so when the //! coordinator is NOT a member of that group, resolution misses → the -//! coordinator ships `Surrogate::ZERO` to the owner → the owner does -//! `surrogate_to_doc_id(ZERO)` → the row is NOT FOUND. So cross-node PK reads +//! coordinator ships `Surrogate::ZERO` to the owner → the owner renders +//! `StorageKey::for_surrogate(ZERO)` → the row is NOT FOUND. So cross-node PK reads //! from a non-member coordinator silently returned EMPTY. //! //! Scans are unaffected: they route + scan on the owner with no surrogate diff --git a/nodedb-cluster-tests/tests/common_suite/cases/vector_index_dispatch_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/vector_index_dispatch_cross_node.rs index e12709380..db9ae3cc3 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/vector_index_dispatch_cross_node.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/vector_index_dispatch_cross_node.rs @@ -288,11 +288,14 @@ async fn search_probe(node: &TestClusterNode, query: Vec) -> Result match map.get("id") { - Some(Value::Integer(id)) => Some(*id as u32), + Some(Value::String(key)) => { + nodedb_types::StorageKey::parse(key).map(|k| k.surrogate().as_u32()) + } _ => None, }, _ => None, diff --git a/nodedb-graph/src/csr/index/interning.rs b/nodedb-graph/src/csr/index/interning.rs index 0ecfa1bbc..df9f702fb 100644 --- a/nodedb-graph/src/csr/index/interning.rs +++ b/nodedb-graph/src/csr/index/interning.rs @@ -130,7 +130,7 @@ impl CsrIndex { /// /// A graph node and its same-pk document share one global surrogate, so this /// is the bridge from a MATCH binding's node name to the document storage key - /// (`surrogate_to_doc_id`) used to fetch the node's properties. + /// (`StorageKey::for_surrogate`) used to fetch the node's properties. pub fn node_surrogate(&self, node: &str) -> Option { let &local_id = self.node_to_id.get(node)?; let raw = self.node_surrogate_raw(local_id); diff --git a/nodedb-types/src/lib.rs b/nodedb-types/src/lib.rs index a7b0b15b4..c7f9ec841 100644 --- a/nodedb-types/src/lib.rs +++ b/nodedb-types/src/lib.rs @@ -120,8 +120,8 @@ pub use quota::{ pub use result::{QueryResult, SearchResult, SubGraph}; pub use rls_write_check::{RlsWriteCheck, WriteGateDecision}; pub use row_identity::{ - DEFAULT_IDENTITY_COLUMN, HEADLESS_SENTINEL_PREFIX, RowIdentity, StorageKey, - doc_id_to_surrogate, extract_pk_value, identity_of, surrogate_to_doc_id, value_to_pk_string, + DEFAULT_IDENTITY_COLUMN, HEADLESS_SENTINEL_PREFIX, RowIdentity, StorageKey, extract_pk_value, + value_to_pk_string, }; pub use sparse_vector::{SparseVector, SparseVectorError}; pub use sql_quote::{quote_ident, quote_literal}; diff --git a/nodedb-types/src/row_identity.rs b/nodedb-types/src/row_identity.rs index 2ec4dbcc2..e7d1b1311 100644 --- a/nodedb-types/src/row_identity.rs +++ b/nodedb-types/src/row_identity.rs @@ -150,40 +150,6 @@ impl std::fmt::Display for RowIdentity { } } -/// Format a surrogate as the 8-character zero-padded lowercase hex string -/// used as the document's redb key. -/// -/// Thin wrapper over [`StorageKey::for_surrogate`], kept because 242 -/// call sites across the workspace hold the result as a plain `String` -/// (redb key params, msgpack field injection, WAL replay) rather than a -/// `StorageKey`. Converting all of them is a separate ripple from this one. -/// One allocation: the `Display` format. -pub fn surrogate_to_doc_id(surrogate: Surrogate) -> String { - StorageKey::for_surrogate(surrogate).to_string() -} - -/// Parse a hex-encoded document storage key back to a `Surrogate`. -/// -/// Returns `None` if the key is not exactly 8 lowercase hex characters — -/// this handles legacy non-surrogate document IDs gracefully. -/// -/// Thin wrapper over [`StorageKey::parse`], kept for the same reason as -/// [`surrogate_to_doc_id`]: 35 call sites hold a plain `&str` doc ID. -/// Allocation-free. -pub fn doc_id_to_surrogate(doc_id: &str) -> Option { - StorageKey::parse(doc_id).map(|key| key.surrogate()) -} - -/// The client-visible identity of a row stored under `doc_id`. -/// -/// A minted key renders its surrogate in decimal. Any other key is a user's -/// own value and passes through verbatim. -pub fn identity_of(doc_id: &str) -> RowIdentity { - StorageKey::parse(doc_id) - .map(|key| key.to_identity()) - .unwrap_or_else(|| RowIdentity::from_user_key(doc_id)) -} - /// Extract the stringified value of `field` from a MessagePack row body. /// /// Returns `None` when the body is not an object, lacks `field`, or the @@ -312,12 +278,13 @@ mod tests { } #[test] - fn identity_of_minted_key_is_decimal() { - assert_eq!(identity_of("0000002a").as_str(), "42"); - } - - #[test] - fn identity_of_user_key_passes_through() { - assert_eq!(identity_of("user-declared-id").as_str(), "user-declared-id"); + fn parse_none_yields_user_key_identity() { + assert_eq!( + StorageKey::parse("user-declared-id") + .map(|key| key.to_identity()) + .unwrap_or_else(|| RowIdentity::from_user_key("user-declared-id")) + .as_str(), + "user-declared-id" + ); } } diff --git a/nodedb/Cargo.toml b/nodedb/Cargo.toml index 287f706c8..97d185c73 100644 --- a/nodedb/Cargo.toml +++ b/nodedb/Cargo.toml @@ -160,6 +160,9 @@ tikv-jemalloc-ctl = { workspace = true } # Random number generation (weighted pick, provably fair gacha) rand = { workspace = true } +# Hex text for KV overlay keys and OTel trace/span ids +hex = { workspace = true } + [dev-dependencies] tokio = { workspace = true, features = ["test-util"] } # Named again here (same published version, unified by Cargo) so crash-harness @@ -176,7 +179,6 @@ nodedb-test-support = { workspace = true } rcgen = { workspace = true } zerompk = { workspace = true } loro = { workspace = true } -hex = { workspace = true } [target.'cfg(target_os = "linux")'.dependencies] # io_uring for Data Plane batched I/O (columnar reads, WAL) diff --git a/nodedb/src/control/otel/receiver.rs b/nodedb/src/control/otel/receiver.rs index 124f42a29..bf16480bc 100644 --- a/nodedb/src/control/otel/receiver.rs +++ b/nodedb/src/control/otel/receiver.rs @@ -490,18 +490,3 @@ fn decompress_body(headers: &HeaderMap, body: &Bytes) -> Vec { body.to_vec() } - -/// Minimal hex encoding for trace/span IDs (avoids adding `hex` crate). -mod hex { - pub fn encode(bytes: &[u8]) -> String { - let mut s = String::with_capacity(bytes.len() * 2); - for &b in bytes { - s.push(HEX_CHARS[(b >> 4) as usize]); - s.push(HEX_CHARS[(b & 0xf) as usize]); - } - s - } - const HEX_CHARS: [char; 16] = [ - '0', '1', '2', '3', '4', '5', '6', '7', '8', '9', 'a', 'b', 'c', 'd', 'e', 'f', - ]; -} diff --git a/nodedb/src/control/server/http/routes/query_stream.rs b/nodedb/src/control/server/http/routes/query_stream.rs index 0167ab9fb..df1e8b265 100644 --- a/nodedb/src/control/server/http/routes/query_stream.rs +++ b/nodedb/src/control/server/http/routes/query_stream.rs @@ -191,11 +191,20 @@ pub(super) fn ndjson_body_stream( // `{"id": …, "id_1": …}` rather than collapsing to one cell. // Only re-borrows the once-resolved inputs, so the very first // batch is redacted under the same policy as the last. - let shaped = shape_decoded_rows( + let shaped = match shape_decoded_rows( &value, projection.as_ref(), redaction.as_ref().map(|r| r.ctx(&state.redaction)), - ); + ) { + Ok(s) => s, + Err(e) => { + // In-band error line, matching the malformed-batch path + // above: the HTTP body itself never errors. + let line = format!("{}\n", serde_json::json!({ "error": format!("{e}") })); + yield Ok(Bytes::from(line)); + return; + } + }; for row in shaped.rows { if emitted >= limit { break; diff --git a/nodedb/src/control/server/native/session/session_stream.rs b/nodedb/src/control/server/native/session/session_stream.rs index 7ca30bfe2..e452226c9 100644 --- a/nodedb/src/control/server/native/session/session_stream.rs +++ b/nodedb/src/control/server/native/session/session_stream.rs @@ -40,16 +40,16 @@ fn decode_batch_to_columns_rows( json_text: &str, projection: Option<&OutputSchema>, redaction: Option>, -) -> (Vec, Vec>) { +) -> crate::Result<(Vec, Vec>)> { match sonic_rs::from_str::(json_text) { Ok(decoded) => { - let shaped = shape_decoded_rows(&decoded, projection, redaction); - to_native_columns_rows(&shaped) + let shaped = shape_decoded_rows(&decoded, projection, redaction)?; + Ok(to_native_columns_rows(&shaped)) } - Err(_) => ( + Err(_) => Ok(( vec!["result".into()], vec![vec![Value::String(json_text.to_string())]], - ), + )), } } @@ -106,7 +106,7 @@ pub(super) async fn emit_sql_stream( &json_text, projection.as_ref(), redaction.as_ref().map(|r| r.ctx(&state.redaction)), - ); + )?; if batch_rows.is_empty() { continue; } diff --git a/nodedb/src/control/server/pgwire/handler/stream_response.rs b/nodedb/src/control/server/pgwire/handler/stream_response.rs index eeb15a120..84103a59a 100644 --- a/nodedb/src/control/server/pgwire/handler/stream_response.rs +++ b/nodedb/src/control/server/pgwire/handler/stream_response.rs @@ -228,7 +228,14 @@ pub(crate) fn streaming_shaped_response( &value, Some(&schema_out), redaction.as_ref().map(|r| r.ctx(&state.redaction)), - ); + ) + .map_err(|e| { + PgWireError::UserError(Box::new(ErrorInfo::new( + "ERROR".to_owned(), + "XX000".to_owned(), + format!("failed to shape streamed batch: {e}"), + ))) + })?; for row in &shaped.rows { if emitted >= limit { break; @@ -332,11 +339,20 @@ pub(crate) async fn streaming_star_response( )); } - let shaped = shape_decoded_rows( + let shaped = match shape_decoded_rows( &serde_json::Value::Array(values), None, redaction.as_ref().map(|r| r.ctx(&state.redaction)), - ); + ) { + Ok(s) => s, + Err(e) => { + return single_pgwire_error(PgWireError::UserError(Box::new(ErrorInfo::new( + "ERROR".to_owned(), + "XX000".to_owned(), + format!("failed to shape streamed batch: {e}"), + )))); + } + }; // `SELECT *` derives its columns from the rows and has no client-requested // per-column formats, so it always renders text. let (response, _notice) = shaped_query_response(shaped, &[]); diff --git a/nodedb/src/control/server/response_shape/compose.rs b/nodedb/src/control/server/response_shape/compose.rs index e4270abd8..5e1286d97 100644 --- a/nodedb/src/control/server/response_shape/compose.rs +++ b/nodedb/src/control/server/response_shape/compose.rs @@ -87,12 +87,12 @@ pub fn shape_response_materialized( let translated = translate_search_response(&wrapped, plan, state, database_id, tenant_id); let shaped = match plan_kind { - PlanKind::ArraySlice => shape_array_slice(&translated, redaction), + PlanKind::ArraySlice => shape_array_slice(&translated, redaction)?, // `RETURNING` rows are held to the columns already announced to the // client, when any were — see `super::returning`. PlanKind::ReturningRows => shape_returning_rows(&translated, projection, redaction)?, PlanKind::SingleDocument | PlanKind::MultiRow => { - shape_generic_rows(&translated, projection, redaction) + shape_generic_rows(&translated, projection, redaction)? } // Handled by the early return above; kept exhaustive (no catch-all, // no panic) so a future PlanKind desync degrades to passthrough @@ -118,12 +118,12 @@ pub fn shape_payload_no_plan( ) -> Result { Ok(match plan_kind { PlanKind::Execution | PlanKind::DmlResult(_) => ShapeOutcome::Passthrough, - PlanKind::ArraySlice => ShapeOutcome::Rows(shape_array_slice(payload, redaction)), + PlanKind::ArraySlice => ShapeOutcome::Rows(shape_array_slice(payload, redaction)?), PlanKind::ReturningRows => { ShapeOutcome::Rows(shape_returning_rows(payload, projection, redaction)?) } PlanKind::SingleDocument | PlanKind::MultiRow => { - ShapeOutcome::Rows(shape_generic_rows(payload, projection, redaction)) + ShapeOutcome::Rows(shape_generic_rows(payload, projection, redaction)?) } }) } @@ -136,9 +136,12 @@ pub fn shape_payload_no_plan( /// Array slices never carry a SELECT-list projection today (matching the /// pre-extraction behavior), so `shape_decoded_rows` is always called with /// a `None` projection here — but redaction still applies to the cells. -fn shape_array_slice(payload: &[u8], redaction: Option>) -> ShapedRows { +fn shape_array_slice( + payload: &[u8], + redaction: Option>, +) -> crate::Result { if payload.is_empty() { - return empty_shaped(); + return Ok(empty_shaped()); } let (rows_json, truncated) = if let Ok(resp) = zerompk::from_msgpack::(payload) { @@ -152,11 +155,11 @@ fn shape_array_slice(payload: &[u8], redaction: Option>) -> Sha let notice = truncated.then(|| TRUNCATED_BEFORE_HORIZON_NOTICE.to_string()); let mut shaped = match sonic_rs::from_str::(&rows_json) { - Ok(value) => shape_decoded_rows(&value, None, redaction), + Ok(value) => shape_decoded_rows(&value, None, redaction)?, Err(_) => empty_shaped(), }; shaped.notice = notice; - shaped + Ok(shaped) } /// Shape a `SingleDocument` / `MultiRow` response: decode to JSON, then @@ -168,14 +171,14 @@ fn shape_generic_rows( payload: &[u8], projection: Option<&OutputSchema>, redaction: Option>, -) -> ShapedRows { +) -> crate::Result { if payload.is_empty() { - return empty_shaped(); + return Ok(empty_shaped()); } let text = decode_payload_to_json(payload); match sonic_rs::from_str::(&text) { Ok(value) => shape_decoded_rows(&value, projection, redaction), - Err(_) => single_result_row(text), + Err(_) => Ok(single_result_row(text)), } } @@ -196,9 +199,9 @@ pub fn shape_decoded_rows( decoded: &JsonValue, projection: Option<&OutputSchema>, redaction: Option>, -) -> ShapedRows { +) -> crate::Result { let mut rows = Vec::new(); - push_flat_rows(decoded.clone(), &mut rows); + push_flat_rows(decoded.clone(), &mut rows)?; // Column-level redaction runs on the flat row maps, AFTER the scan // envelope is unwrapped and BEFORE any projection or column derivation. @@ -231,12 +234,12 @@ pub fn shape_decoded_rows( // each cell in that type's PostgreSQL text form; native/http // ignore column types entirely. let column_types: Vec = s.columns.iter().map(|c| c.ty).collect(); - ShapedRows { + Ok(ShapedRows { columns: display_names, column_types, rows: projected_rows, notice: None, - } + }) } _ => { // Star / derived columns come from JSON rows with no catalog type, @@ -244,12 +247,12 @@ pub fn shape_decoded_rows( // schemaless collections. let columns = derive_columns(&rows); let column_types = ShapedRows::text_types(columns.len()); - ShapedRows { + Ok(ShapedRows { columns, column_types, rows, notice: None, - } + }) } } } @@ -553,7 +556,8 @@ mod tests { let sources = vec![(String::new(), "users".to_string())]; let decoded = one_row(serde_json::json!({"email": "a@b.c", "name": "Alice"})); - let shaped = shape_decoded_rows(&decoded, None, Some(ctx(&store, &roles, &sources))); + let shaped = shape_decoded_rows(&decoded, None, Some(ctx(&store, &roles, &sources))) + .expect("shape rows"); assert_eq!(shaped.rows[0]["email"], JsonValue::String("***".into())); assert_eq!(shaped.rows[0]["name"], JsonValue::String("Alice".into())); } @@ -572,8 +576,9 @@ mod tests { let sources = vec![(String::new(), "users".to_string())]; let decoded = one_row(serde_json::json!({"email": "a@b.c", "name": "Alice"})); - let baseline = shape_decoded_rows(&decoded, None, None); - let shaped = shape_decoded_rows(&decoded, None, Some(ctx(&store, &roles, &sources))); + let baseline = shape_decoded_rows(&decoded, None, None).expect("shape rows"); + let shaped = shape_decoded_rows(&decoded, None, Some(ctx(&store, &roles, &sources))) + .expect("shape rows"); assert_eq!(shaped.rows, baseline.rows); assert_eq!(shaped.columns, baseline.columns); } @@ -597,7 +602,8 @@ mod tests { &decoded, Some(&projection), Some(ctx(&store, &roles, &sources)), - ); + ) + .expect("shape rows"); assert_eq!(shaped.columns, vec!["contact".to_string()]); assert_eq!(shaped.rows[0]["contact"], JsonValue::String("***".into())); } @@ -624,7 +630,8 @@ mod tests { &decoded, Some(&projection), Some(ctx(&store, &roles, &sources)), - ); + ) + .expect("shape rows"); // `cell_keys` suffixes the duplicate display name. assert_eq!(shaped.rows[0]["id"], JsonValue::String("***".into())); assert_eq!(shaped.rows[0]["id_1"], JsonValue::String("b1".into())); @@ -644,7 +651,8 @@ mod tests { let sources = vec![(String::new(), "users".to_string())]; let decoded = one_row(serde_json::json!({"id": "u1", "email": "a@b.c"})); - let shaped = shape_decoded_rows(&decoded, None, Some(ctx(&store, &roles, &sources))); + let shaped = shape_decoded_rows(&decoded, None, Some(ctx(&store, &roles, &sources))) + .expect("shape rows"); assert!( shaped.columns.contains(&"email".to_string()), "redacted column must stay in the derived SELECT * schema: {:?}", diff --git a/nodedb/src/control/server/response_shape/project.rs b/nodedb/src/control/server/response_shape/project.rs index 5ba44ba4e..646868027 100644 --- a/nodedb/src/control/server/response_shape/project.rs +++ b/nodedb/src/control/server/response_shape/project.rs @@ -24,14 +24,18 @@ pub fn json_value_to_text(v: &serde_json::Value) -> String { } /// Flatten a parsed JSON value into row objects. +/// +/// The envelope `id` is a rendered [`StorageKey`](crate::engine::document::store::StorageKey) +/// by construction. +/// A value that fails `StorageKey::parse` is surfaced as `Err`, never accommodated. pub fn push_flat_rows( value: serde_json::Value, out: &mut Vec>, -) { +) -> crate::Result<()> { match value { serde_json::Value::Array(items) => { for item in items { - push_flat_rows(item, out); + push_flat_rows(item, out)?; } } serde_json::Value::Object(mut map) => { @@ -44,18 +48,23 @@ pub fn push_flat_rows( // boundary. `or_insert` leaves a declared primary key as the // authority. if let Some(serde_json::Value::String(key)) = map.remove("id") { - let identity = crate::engine::document::store::identity_of(&key); + let identity = crate::engine::document::store::StorageKey::parse(&key) + .ok_or_else(|| crate::Error::Internal { + detail: format!("scan envelope id is not a storage key: '{key}'"), + })? + .to_identity(); inner .entry("id") .or_insert(serde_json::Value::String(identity.into_string())); } out.push(inner); - return; + return Ok(()); } out.push(map); } _ => {} } + Ok(()) } /// The Data Plane's raw document-scan codec emits objects with exactly diff --git a/nodedb/src/control/server/response_translate/text_hybrid.rs b/nodedb/src/control/server/response_translate/text_hybrid.rs index 6812ed02b..c2c307386 100644 --- a/nodedb/src/control/server/response_translate/text_hybrid.rs +++ b/nodedb/src/control/server/response_translate/text_hybrid.rs @@ -5,7 +5,7 @@ //! search responses. //! //! `TextOp::Search` hits carry the standard `{id, data}` document-scan -//! envelope, keyed by `surrogate_to_doc_id(surrogate)` hex — the document +//! envelope, keyed by `StorageKey::for_surrogate(surrogate)` hex — the document //! body itself already carries the user's PK as an ordinary field (it was //! written verbatim from the user's INSERT), so the resolved value only //! needs injecting when the body has no `id` field of its own (a headless diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs index 221924de9..ca33bc44f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs @@ -153,7 +153,7 @@ pub async fn balance_as_of( .map_err(|e| err("22P02", &format!("invalid JSON in source scan: {e}")))?; // Unwrap the `{"id", "data"}` scan envelope so matching and `value_expr` // evaluation read the stored fields, not the wire wrapper. - let source_docs = unwrap_scan_docs(source_docs); + let source_docs = unwrap_scan_docs(source_docs)?; // Sum value_expr for source rows where join_column = key AND created_at > as_of. let mut recent_sum = rust_decimal::Decimal::ZERO; diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs index d96d7df50..81fc44d0b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs @@ -106,7 +106,7 @@ pub async fn convert_currency_lookup( .map_err(|e| err("22P02", &format!("invalid JSON in rate table scan: {e}")))?; // Unwrap the `{"id", "data"}` scan envelope so matching reads the stored // fields, not the wire wrapper. - let docs = unwrap_scan_docs(docs); + let docs = unwrap_scan_docs(docs)?; // Find latest row where key matches and time <= as_of. let mut best_rate: Option = None; diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/helpers.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/helpers.rs index b18f9618b..8cbfcb7e5 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/helpers.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/helpers.rs @@ -89,12 +89,12 @@ pub fn single_result(value: &str) -> Vec { /// pgwire/HTTP row shaper applies — so there is exactly one definition of /// "unwrap a scan envelope" in the tree. Rows that are not `{id, data}` /// wrapped (already-flat producers) pass through unchanged. -pub fn unwrap_scan_docs(docs: Vec) -> Vec> { +pub fn unwrap_scan_docs(docs: Vec) -> Result>, DdlError> { let mut out = Vec::with_capacity(docs.len()); for doc in docs { - push_flat_rows(doc, &mut out); + push_flat_rows(doc, &mut out).map_err(|e| err("XX000", &e.to_string()))?; } - out + Ok(out) } /// Unwrap a `DocumentOp::Scan` envelope while also returning the row's wire diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs index f5955747d..2093d0195 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs @@ -81,7 +81,7 @@ pub async fn temporal_lookup( // The raw document-scan codec wraps each row as `{"id": .., "data": {..}}`; // unwrap it so matching and redaction operate on the stored fields, not // the wire wrapper. - let docs = unwrap_scan_docs(docs); + let docs = unwrap_scan_docs(docs)?; // Find the row with latest time_column <= as_of for the given key. let mut best_doc: Option<&serde_json::Map> = None; diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs index e1b213dfd..b6e7a25fa 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs @@ -103,7 +103,7 @@ pub async fn verify_balance( .map_err(|e| err("22P02", &format!("invalid JSON in target scan: {e}")))?; // Unwrap the `{"id", "data"}` scan envelope so matching reads the stored // fields, not the wire wrapper. - let target_docs = unwrap_scan_docs(target_docs); + let target_docs = unwrap_scan_docs(target_docs)?; // Scan all source rows. let source_vshard = @@ -143,7 +143,7 @@ pub async fn verify_balance( .map_err(|e| err("22P02", &format!("invalid JSON in source scan: {e}")))?; // Unwrap the `{"id", "data"}` scan envelope so matching and `value_expr` // evaluation read the stored fields, not the wire wrapper. - let source_docs = unwrap_scan_docs(source_docs); + let source_docs = unwrap_scan_docs(source_docs)?; // For each target row, recompute balance from source rows. let mut discrepancies = 0u64; diff --git a/nodedb/src/control/server/wal_dispatch/core.rs b/nodedb/src/control/server/wal_dispatch/core.rs index 4f47e07b8..782074b07 100644 --- a/nodedb/src/control/server/wal_dispatch/core.rs +++ b/nodedb/src/control/server/wal_dispatch/core.rs @@ -241,7 +241,8 @@ mod tests { assert_eq!(decoded.text, "hello world"); assert_eq!( decoded.doc_id, - crate::engine::document::store::surrogate_to_doc_id(Surrogate::new(7)) + crate::engine::document::store::StorageKey::for_surrogate(Surrogate::new(7)) + .to_string() ); } @@ -274,7 +275,8 @@ mod tests { assert_eq!(decoded.collection, "docs"); assert_eq!( decoded.doc_id, - crate::engine::document::store::surrogate_to_doc_id(Surrogate::new(7)) + crate::engine::document::store::StorageKey::for_surrogate(Surrogate::new(7)) + .to_string() ); } diff --git a/nodedb/src/data/executor/core_loop/doc_config_seed.rs b/nodedb/src/data/executor/core_loop/doc_config_seed.rs index b35565b2d..1d0878903 100644 --- a/nodedb/src/data/executor/core_loop/doc_config_seed.rs +++ b/nodedb/src/data/executor/core_loop/doc_config_seed.rs @@ -42,13 +42,13 @@ impl CoreLoop { #[cfg(test)] mod tests { use nodedb_physical::physical_plan::StorageMode; - use nodedb_types::Surrogate; use nodedb_types::columnar::{ColumnDef, ColumnType, StrictSchema}; + use nodedb_types::{StorageKey, Surrogate}; use nodedb_wal::{RecordType, TombstoneSet, WalRecord, WalRecordArgs}; use crate::data::executor::core_loop::tests::make_core_with_dir; use crate::data::executor::strict_format; - use crate::engine::document::store::{CollectionConfig, surrogate_to_doc_id}; + use crate::engine::document::store::CollectionConfig; use crate::types::{DatabaseId, TenantId}; const DB: u64 = 0; @@ -75,7 +75,7 @@ mod tests { fn put_record(surrogate: u32) -> WalRecord { let payload = zerompk::to_msgpack_vec(&( COLL.to_string(), - surrogate_to_doc_id(Surrogate::new(surrogate)), + StorageKey::for_surrogate(Surrogate::new(surrogate)).to_string(), doc_bytes(), Option::::None, surrogate, diff --git a/nodedb/src/data/executor/dispatch/bitmap/materialize.rs b/nodedb/src/data/executor/dispatch/bitmap/materialize.rs index 7c111ba8f..11da0ad41 100644 --- a/nodedb/src/data/executor/dispatch/bitmap/materialize.rs +++ b/nodedb/src/data/executor/dispatch/bitmap/materialize.rs @@ -16,8 +16,8 @@ use nodedb_types::{Surrogate, SurrogateBitmap}; /// Parse a sequence of `(doc_id, _bytes)` pairs into a `SurrogateBitmap`. /// -/// Accepts 8-char lowercase-hex doc_ids produced by the document engine's -/// `surrogate_to_doc_id` encoding. Non-conforming ids are skipped without error. +/// Accepts 8-char lowercase-hex doc_ids produced by `StorageKey`'s `Display` +/// impl. Non-conforming ids are skipped without error. pub(crate) fn collect_surrogates(docs: &[(String, Vec)]) -> SurrogateBitmap { let mut bm = SurrogateBitmap::new(); for (doc_id, _) in docs { diff --git a/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs b/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs index 8210ca72a..77d8b3f83 100644 --- a/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs +++ b/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs @@ -52,8 +52,9 @@ //! # Identity comes from the plan, never from a store probe //! //! Rows are keyed by an 8-hex surrogate -//! ([`surrogate_to_doc_id`](crate::engine::document::store::surrogate_to_doc_id)), -//! so a join-key VALUE is not a storage key. The Control Plane resolves each +//! ([`StorageKey`](crate::engine::document::store::StorageKey), rendered via +//! its `Display` impl), so a join-key VALUE is not a storage key. The +//! Control Plane resolves each //! join value to its target row's surrogate at plan time and the resolution //! arrives on //! [`EnforcementCtx::resolved_targets`](crate::data::executor::enforcement::images::EnforcementCtx). @@ -313,7 +314,7 @@ mod tests { use crate::data::executor::handlers::document::write::DocumentBatchInsertParams; use crate::data::executor::handlers::update_from_join::UpdateFromJoinParams; use crate::data::executor::strict_format; - use crate::engine::document::store::{CollectionConfig, surrogate_to_doc_id}; + use crate::engine::document::store::CollectionConfig; use crate::types::TenantId; use nodedb_physical::physical_plan::{ResolvedSumTarget, StorageMode, UpdateValue}; use nodedb_types::columnar::{ColumnDef, ColumnType, StrictSchema}; @@ -428,7 +429,7 @@ mod tests { /// worse — the row survives the statement and is unreadable to every strict /// reader afterwards. /// - /// The target row is seeded under `surrogate_to_doc_id`, the key every + /// The target row is seeded under its `StorageKey`, the key every /// reader of that collection uses. Seeding under the raw join VALUE would /// only prove that a lookup keyed by the same wrong value finds it. #[test] @@ -984,7 +985,7 @@ mod tests { let rate = serde_json::json!({"rate_id": "r1", "amount": 80}); let source_rows = vec![( - surrogate_to_doc_id(Surrogate(9)), + nodedb_types::StorageKey::for_surrogate(Surrogate(9)).to_string(), doc_format::encode_to_msgpack(&rate), )]; @@ -1030,13 +1031,13 @@ mod tests { let documents = vec![ ( - surrogate_to_doc_id(Surrogate(1)), + nodedb_types::StorageKey::for_surrogate(Surrogate(1)).to_string(), doc_format::encode_to_msgpack( &serde_json::json!({"account_id": ACCOUNT_A, "amount": 25}), ), ), ( - surrogate_to_doc_id(Surrogate(2)), + nodedb_types::StorageKey::for_surrogate(Surrogate(2)).to_string(), doc_format::encode_to_msgpack( &serde_json::json!({"account_id": ACCOUNT_A, "amount": 75}), ), diff --git a/nodedb/src/data/executor/handlers/document/write/batch_insert.rs b/nodedb/src/data/executor/handlers/document/write/batch_insert.rs index 8ed3d7d42..2bc6ae6b6 100644 --- a/nodedb/src/data/executor/handlers/document/write/batch_insert.rs +++ b/nodedb/src/data/executor/handlers/document/write/batch_insert.rs @@ -392,11 +392,11 @@ mod tests { use crate::data::executor::core_loop::tests::make_core_with_dir; use crate::data::executor::doc_format; use crate::data::executor::task::ExecutionTask; - use crate::engine::document::store::{CollectionConfig, surrogate_to_doc_id}; + use crate::engine::document::store::CollectionConfig; use crate::engine::sparse::fts_redb::tables::DOC_LENGTHS; use crate::types::{DatabaseId, ReadConsistency, RequestId, TenantId, TraceId, VShardId}; use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan, ResolvedSumTarget}; - use nodedb_types::Surrogate; + use nodedb_types::{StorageKey, Surrogate}; use std::time::{Duration, Instant}; const TID: u64 = 1; @@ -675,8 +675,14 @@ mod tests { let mut core = sum_seeded_core(dir.path()); let documents = vec![ - (surrogate_to_doc_id(Surrogate(1)), sum_entry(SUM_A1, 25)), - (surrogate_to_doc_id(Surrogate(2)), sum_entry(SUM_A1, 75)), + ( + StorageKey::for_surrogate(Surrogate(1)).to_string(), + sum_entry(SUM_A1, 25), + ), + ( + StorageKey::for_surrogate(Surrogate(2)).to_string(), + sum_entry(SUM_A1, 75), + ), ]; let surrogates = vec![Surrogate(1), Surrogate(2)]; let resolved = vec![ResolvedSumTarget::new(SUM_TARGET, SUM_A1, SUM_T1)]; diff --git a/nodedb/src/data/executor/handlers/point/apply_put/core.rs b/nodedb/src/data/executor/handlers/point/apply_put/core.rs index cddfeb803..5e3f98b3f 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/core.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/core.rs @@ -334,7 +334,7 @@ mod tests { use crate::data::executor::handlers::point::apply_put::PointPutParams; use crate::data::executor::handlers::point::put::PointPutExec; use crate::data::executor::task::ExecutionTask; - use crate::engine::document::store::surrogate_to_doc_id; + use crate::engine::document::store::StorageKey; use crate::engine::sparse::fts_redb::tables::DOC_LENGTHS; use crate::types::{DatabaseId, ReadConsistency, RequestId, TenantId, TraceId, VShardId}; use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; @@ -364,6 +364,10 @@ mod tests { txn.commit().unwrap(); } + fn row_key() -> String { + StorageKey::for_surrogate(SURROGATE).to_string() + } + fn point_put_task(row_key: &str) -> ExecutionTask { ExecutionTask::new(Request { request_id: RequestId::new(1), @@ -410,7 +414,7 @@ mod tests { fn healthy_index_commits_the_row_and_indexes_it() { let dir = tempfile::tempdir().unwrap(); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let row_key = surrogate_to_doc_id(SURROGATE); + let row_key = row_key(); let task = point_put_task(&row_key); let resp = core.execute_point_put( @@ -442,7 +446,7 @@ mod tests { fn index_failure_rejects_the_write_and_leaves_no_row() { let dir = tempfile::tempdir().unwrap(); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let row_key = surrogate_to_doc_id(SURROGATE); + let row_key = row_key(); poison_inverted_index(&core); let task = point_put_task(&row_key); @@ -487,7 +491,7 @@ mod tests { fn apply_point_put_propagates_index_failure_instead_of_absorbing_it() { let dir = tempfile::tempdir().unwrap(); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let row_key = surrogate_to_doc_id(SURROGATE); + let row_key = row_key(); poison_inverted_index(&core); let txn = core.sparse.begin_write().unwrap(); @@ -528,7 +532,7 @@ mod tests { fn index_text_disabled_is_unaffected_by_a_broken_index() { let dir = tempfile::tempdir().unwrap(); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let row_key = surrogate_to_doc_id(SURROGATE); + let row_key = row_key(); poison_inverted_index(&core); let txn = core.sparse.begin_write().unwrap(); diff --git a/nodedb/src/data/executor/handlers/spatial.rs b/nodedb/src/data/executor/handlers/spatial.rs index 95a789252..cd90daece 100644 --- a/nodedb/src/data/executor/handlers/spatial.rs +++ b/nodedb/src/data/executor/handlers/spatial.rs @@ -16,10 +16,22 @@ use crate::bridge::scan_filter::ScanFilter; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; use nodedb_physical::physical_plan::SpatialPredicate; -use nodedb_types::{Surrogate, SurrogateBitmap}; +use nodedb_types::SurrogateBitmap; use super::spatial_refine::{apply_predicate, expand_bbox, extract_geometry, project_doc}; +/// Whether `doc_id`'s surrogate is a member of `prefilter`. +/// +/// `doc_id` is the hex-encoded surrogate for a document-collection row, or +/// a columnar-family user `id` that never parses as a storage key — the +/// latter is never admitted, matching a sparse miss on a parsed key. +fn prefilter_admits(prefilter: &SurrogateBitmap, doc_id: &str) -> bool { + match nodedb_types::StorageKey::parse(doc_id) { + Some(key) => prefilter.contains(key.surrogate()), + None => false, + } +} + /// Parameters for [`CoreLoop::execute_spatial_scan`]. pub(in crate::data::executor) struct SpatialScanParams<'a> { pub task: &'a ExecutionTask, @@ -211,15 +223,10 @@ impl CoreLoop { // Prefilter: skip candidates not in the surrogate bitmap before // any geometry evaluation. The doc_id is a hex-encoded surrogate. - if let Some(bitmap) = prefilter { - match storage_key { - Some(key) => { - if !bitmap.contains(key.surrogate()) { - continue; - } - } - None => continue, - } + if let Some(bitmap) = prefilter + && !prefilter_admits(bitmap, &doc_id) + { + continue; } // Fetch the candidate's document in an engine-aware way. A sparse @@ -397,15 +404,10 @@ impl CoreLoop { } // Prefilter: skip non-members before geometry evaluation. - if let Some(bitmap) = prefilter { - match u32::from_str_radix(doc_id, 16) { - Ok(raw) => { - if !bitmap.contains(Surrogate(raw)) { - continue; - } - } - Err(_) => continue, - } + if let Some(bitmap) = prefilter + && !prefilter_admits(bitmap, doc_id) + { + continue; } // A row skipped here silently drops out of the spatial result set, @@ -481,7 +483,7 @@ impl CoreLoop { #[cfg(test)] mod tests { - use super::SpatialScanParams; + use super::{SpatialScanParams, prefilter_admits}; use crate::bridge::envelope::{PhysicalPlan, Priority, Request, Status}; use crate::data::executor::task::ExecutionTask; use crate::engine::spatial::RTreeEntry; @@ -549,7 +551,8 @@ mod tests { lng: f64, lat: f64, ) -> String { - let doc_id = crate::engine::document::store::surrogate_to_doc_id(surrogate); + let storage_key = nodedb_types::StorageKey::for_surrogate(surrogate); + let doc_id = storage_key.to_string(); // Build a minimal msgpack document with a GeoJSON Point field. let geojson = serde_json::json!({ @@ -558,7 +561,6 @@ mod tests { }); let msgpack = nodedb_types::json_to_msgpack(&geojson).unwrap(); - let storage_key = nodedb_types::StorageKey::for_surrogate(surrogate); core.sparse .put(0, tid, collection, &storage_key, &msgpack) .unwrap(); @@ -620,33 +622,27 @@ mod tests { }) } + fn doc_id(surrogate: u32) -> String { + nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)).to_string() + } + #[test] fn prefilter_skips_non_member_doc_ids() { - // Direct unit on the prefilter check: the candidate-loop logic - // parses doc_id as hex Surrogate and skips non-members. + // Direct unit on the production prefilter check (`prefilter_admits`), + // not a re-implementation of it. let mut bitmap = SurrogateBitmap::new(); bitmap.insert(Surrogate(2)); - let candidate_doc_ids = [ - crate::engine::document::store::surrogate_to_doc_id(Surrogate(1)), - crate::engine::document::store::surrogate_to_doc_id(Surrogate(2)), - crate::engine::document::store::surrogate_to_doc_id(Surrogate(3)), - ]; + let candidate_doc_ids = [doc_id(1), doc_id(2), doc_id(3)]; let kept: Vec<_> = candidate_doc_ids .iter() - .filter(|doc_id| match u32::from_str_radix(doc_id, 16) { - Ok(raw) => bitmap.contains(Surrogate(raw)), - Err(_) => false, - }) + .filter(|doc_id| prefilter_admits(&bitmap, doc_id)) .cloned() .collect(); assert_eq!(kept.len(), 1); - assert_eq!( - kept[0], - crate::engine::document::store::surrogate_to_doc_id(Surrogate(2)) - ); + assert_eq!(kept[0], doc_id(2)); } // Note: the R-tree-branch by-surrogate candidate hydration diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/columnar_merge.rs b/nodedb/src/data/executor/handlers/transaction/overlay/columnar_merge.rs index 5e2b02431..365e91fb4 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/columnar_merge.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/columnar_merge.rs @@ -9,8 +9,8 @@ //! flushed-segment surrogate sidecar) rather than the hex-doc-id keying //! [`super::merge::merge_overlay_into_scan`] uses — a columnar row has no //! separate document id, so [`super::super::stage_write::stage_columnar`] -//! stages puts keyed by surrogate with `surrogate_to_doc_id` used only for -//! the overlay's doc-id side-map. A base row with no recorded surrogate +//! stages puts keyed by surrogate with `StorageKey::for_surrogate` used only +//! for the overlay's doc-id side-map. A base row with no recorded surrogate //! (legacy segments predating the surrogate sidecar) cannot be resolved //! against the overlay and is left untouched. //! diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/document.rs b/nodedb/src/data/executor/handlers/transaction/resolve/document.rs index 31607935b..19574e7f9 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/document.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/document.rs @@ -16,7 +16,7 @@ //! * A staged tombstone ([`Staged::Tombstone`]) → `RecordType::Delete`, //! `(collection, document_id, Option, surrogate)`. The redo //! delete shape carries the surrogate (unlike the autocommit delete shape) -//! because replay keys redb by `surrogate_to_doc_id(surrogate)`. +//! because replay keys redb by `StorageKey::for_surrogate(surrogate)`. //! //! ## Stored form vs replay input //! diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs b/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs index 84ae0204a..f231c16f6 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs @@ -384,11 +384,11 @@ mod tests { }; use nodedb_types::columnar::{ColumnDef, ColumnType, StrictSchema}; use nodedb_types::sync::wire::SyncProvenance; - use nodedb_types::{QualifiedCollection, RowIdentity, Surrogate}; + use nodedb_types::{QualifiedCollection, RowIdentity, StorageKey, Surrogate}; use crate::data::executor::handlers::graph::EdgePutParams; use crate::data::executor::strict_format; - use crate::engine::document::store::{CollectionConfig, surrogate_to_doc_id}; + use crate::engine::document::store::CollectionConfig; use crate::bridge::dispatch::{BridgeRequest, BridgeResponse}; use crate::bridge::envelope::{PhysicalPlan, Priority, Request, Status}; @@ -450,6 +450,10 @@ mod tests { (DatabaseId::DEFAULT, TenantId::new(TID), coll.to_string()) } + fn storage_key(surrogate: u32) -> StorageKey { + StorageKey::for_surrogate(Surrogate::new(surrogate)) + } + /// Decode the `RedoRecord` bytes carried in a resolve response payload. fn decode_redo(resp: &crate::bridge::envelope::Response) -> RedoRecord { assert_eq!(resp.status, Status::Ok, "resolve must succeed: {resp:?}"); @@ -754,7 +758,7 @@ mod tests { let txn = TxnId::new(41); let task = make_stage_task(txn); let surrogate = 5u32; - let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); + let row_key = storage_key(surrogate); // Seed a base row directly into the scan-visible sparse store. core.sparse @@ -810,7 +814,7 @@ mod tests { let task = make_stage_task(txn); for s in [1u32, 2u32] { - let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(s)); + let row_key = storage_key(s); core.sparse .put( DatabaseId::DEFAULT.as_u64(), @@ -869,7 +873,7 @@ mod tests { let task = make_stage_task(txn); for s in [1u32, 2u32] { - let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(s)); + let row_key = storage_key(s); core.sparse .put( DatabaseId::DEFAULT.as_u64(), @@ -974,7 +978,7 @@ mod tests { .expect("redo replay must succeed"); for s in [1u32, 2u32] { - let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(s)); + let row_key = storage_key(s); let stored = dst .sparse .get(DatabaseId::DEFAULT.as_u64(), TID, "notes", &row_key) @@ -1407,7 +1411,7 @@ mod tests { let task = make_task(); let txn = TxnId::new(20); let surrogate = 7u32; - let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); + let row_key = storage_key(surrogate); src.txn_overlay_mut(txn).insert_put( coll_key("sdocs"), @@ -1457,7 +1461,7 @@ mod tests { let task = make_task(); let txn = TxnId::new(21); let surrogate = 3u32; - let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); + let row_key = storage_key(surrogate); let body = schemaless_body("alice"); src.txn_overlay_mut(txn).insert_put( @@ -1494,7 +1498,7 @@ mod tests { let task = make_task(); let txn = TxnId::new(22); let surrogate = 11u32; - let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); + let row_key = storage_key(surrogate); src.txn_overlay_mut(txn).insert_tombstone( coll_key("notes"), @@ -1595,7 +1599,7 @@ mod tests { let task = make_task(); let txn = TxnId::new(23); let surrogate = 1u32; - let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); + let row_key = storage_key(surrogate); // Seed a base document row, then stage a DIFFERENT body for it. let seed = wrap_redo(&RedoRecord { @@ -1653,7 +1657,7 @@ mod tests { let task = make_task(); let txn = TxnId::new(24); let doc_surrogate = 5u32; - let doc_row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(doc_surrogate)); + let doc_row_key = storage_key(doc_surrogate); { let overlay = src.txn_overlay_mut(txn); @@ -2080,7 +2084,7 @@ mod tests { let task = make_task(); let txn = TxnId::new(35); let doc_surrogate = 6u32; - let doc_row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(doc_surrogate)); + let doc_row_key = storage_key(doc_surrogate); { let overlay = src.txn_overlay_mut(txn); @@ -2952,7 +2956,7 @@ mod tests { /// R-tree entry id for a surrogate, mirroring `execute_spatial_insert`'s /// `fnv1a_hash(doc_id.as_bytes())` keying. fn spatial_entry_id(surrogate: u32) -> u64 { - let doc_id = surrogate_to_doc_id(Surrogate::new(surrogate)); + let doc_id = storage_key(surrogate).to_string(); crate::util::fnv1a_hash(doc_id.as_bytes()) } @@ -3005,7 +3009,7 @@ mod tests { dst.spatial_doc_map.contains_key(&doc_map_key), "surrogate -> doc-id reverse map must be rebuilt" ); - let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); + let row_key = storage_key(surrogate); assert!( dst.sparse .get(DatabaseId::DEFAULT.as_u64(), TID, "places", &row_key) @@ -3072,7 +3076,7 @@ mod tests { 0, "redo delete must remove the R-tree entry" ); - let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); + let row_key = storage_key(surrogate); assert!( dst.sparse .get(DatabaseId::DEFAULT.as_u64(), TID, "places", &row_key) @@ -3136,7 +3140,7 @@ mod tests { !core.spatial_indexes.contains_key(&key), "resolve must not mutate the base spatial R-tree" ); - let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); + let row_key = storage_key(surrogate); assert!( core.sparse .get(DatabaseId::DEFAULT.as_u64(), TID, "places", &row_key) diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs index 7ba817446..74cf20569 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs @@ -19,11 +19,7 @@ use crate::types::TxnId; /// Lowercase-hex encode a raw KV key. [`unhex_key`] is the inverse. fn hex_key(key: &[u8]) -> String { - let mut s = String::with_capacity(key.len() * 2); - for b in key { - s.push_str(&format!("{b:02x}")); - } - s + hex::encode(key) } /// The overlay identity of a KV row: its raw key, hex encoded, taken @@ -36,17 +32,7 @@ pub(in crate::data::executor) fn kv_row_identity(raw_key: &[u8]) -> RowIdentity /// Decode a lowercase-hex KV overlay doc-id back to raw key bytes, the /// inverse of [`hex_key`]. Returns `None` for malformed hex. pub(in crate::data::executor) fn unhex_key(s: &str) -> Option> { - let bytes = s.as_bytes(); - if !bytes.len().is_multiple_of(2) { - return None; - } - let mut out = Vec::with_capacity(bytes.len() / 2); - for &[hi_byte, lo_byte] in bytes.as_chunks::<2>().0 { - let hi = (hi_byte as char).to_digit(16)?; - let lo = (lo_byte as char).to_digit(16)?; - out.push(((hi << 4) | lo) as u8); - } - Some(out) + hex::decode(s).ok() } impl CoreLoop { diff --git a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs index 1aec73948..283104e1d 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs @@ -310,7 +310,7 @@ mod tests { } fn row_key() -> String { - crate::engine::document::store::surrogate_to_doc_id(Surrogate::new(1)) + key(1).to_string() } fn parity_spatial_key() -> (nodedb_types::DatabaseId, TenantId, String, String) { diff --git a/nodedb/src/data/executor/wal_replay_fts.rs b/nodedb/src/data/executor/wal_replay_fts.rs index 5eff2402f..12c8a3185 100644 --- a/nodedb/src/data/executor/wal_replay_fts.rs +++ b/nodedb/src/data/executor/wal_replay_fts.rs @@ -10,7 +10,7 @@ //! ## Surrogate re-derivation on replay //! //! The WAL payload stores the document key as the hex-encoded surrogate -//! string produced by `surrogate_to_doc_id(surrogate)` (format `{:08x}`). +//! string produced by `StorageKey`'s `Display` impl (format `{:08x}`). //! On replay we parse it back via `u32::from_str_radix(&doc_id, 16)` — //! the same conversion used by the scan / prefilter paths. This does not //! require a catalog or surrogate-assigner round-trip: the 8-hex-char key diff --git a/nodedb/src/data/executor/wal_replay_spatial.rs b/nodedb/src/data/executor/wal_replay_spatial.rs index 6d750aa5a..85467a67f 100644 --- a/nodedb/src/data/executor/wal_replay_spatial.rs +++ b/nodedb/src/data/executor/wal_replay_spatial.rs @@ -10,7 +10,7 @@ //! ## Surrogate re-derivation on replay //! //! The WAL payload `doc_id` field holds the hex-encoded surrogate produced by -//! `surrogate_to_doc_id(surrogate)` (format `{:08x}`). On replay we parse it +//! `StorageKey`'s `Display` impl (format `{:08x}`). On replay we parse it //! back via `u32::from_str_radix(&doc_id, 16)` — no catalog round-trip needed. //! //! ## Geometry decode on replay @@ -370,10 +370,6 @@ mod tests { } } - fn doc_id() -> String { - format!("{SURROGATE:08x}") - } - fn storage_key() -> nodedb_types::StorageKey { nodedb_types::StorageKey::for_surrogate(nodedb_types::Surrogate::new(SURROGATE)) } @@ -408,7 +404,7 @@ mod tests { SyncProvenance::default(), COLLECTION, FIELD, - doc_id(), + storage_key().to_string(), geometry_bytes, ) .to_bytes() @@ -421,7 +417,7 @@ mod tests { SyncProvenance::default(), COLLECTION, FIELD, - doc_id(), + storage_key().to_string(), ) .to_bytes() .expect("encode SpatialDeletePayload"); @@ -444,7 +440,7 @@ mod tests { fn spatial_state(core: &CoreLoop) -> (usize, Option, Option) { let db = DatabaseId::new(DB); let tid = TenantId::new(TENANT); - let entry_id = fnv1a_hash(doc_id().as_bytes()); + let entry_id = fnv1a_hash(storage_key().to_string().as_bytes()); let entries = core .spatial_indexes .get(&(db, tid, COLLECTION.to_string(), FIELD.to_string())) @@ -475,7 +471,10 @@ mod tests { replay(&mut h.core, &record); let after_first = spatial_state(&h.core); assert_eq!(after_first.0, 1, "one geometry indexed"); - assert_eq!(after_first.1.as_deref(), Some(doc_id().as_str())); + assert_eq!( + after_first.1.as_deref(), + Some(storage_key().to_string().as_str()) + ); assert!(after_first.2.is_some(), "the document body must be written"); replay(&mut h.core, &record); diff --git a/nodedb/src/engine/document/store/key.rs b/nodedb/src/engine/document/store/key.rs index 6968222a3..71dfe5e0d 100644 --- a/nodedb/src/engine/document/store/key.rs +++ b/nodedb/src/engine/document/store/key.rs @@ -5,8 +5,6 @@ //! The definitions live in `nodedb_types::row_identity`: `nodedb-physical` //! and `nodedb-query` depend on `nodedb-types` but not on `nodedb`, and both //! need these types on their own fields. This module keeps every existing -//! `crate::engine::document::store::{StorageKey, RowIdentity, identity_of, -//! surrogate_to_doc_id, doc_id_to_surrogate}` path compiling unchanged. -pub use nodedb_types::{ - RowIdentity, StorageKey, doc_id_to_surrogate, identity_of, surrogate_to_doc_id, -}; +//! `crate::engine::document::store::{StorageKey, RowIdentity}` path +//! compiling unchanged. +pub use nodedb_types::{RowIdentity, StorageKey}; diff --git a/nodedb/src/engine/document/store/mod.rs b/nodedb/src/engine/document/store/mod.rs index 29dc00e98..74d58d5ac 100644 --- a/nodedb/src/engine/document/store/mod.rs +++ b/nodedb/src/engine/document/store/mod.rs @@ -10,4 +10,4 @@ pub use config::CollectionConfig; pub use engine::DocumentEngine; pub use extract::{extract_index_values, json_to_msgpack}; pub use index_path::IndexPath; -pub use key::{RowIdentity, StorageKey, doc_id_to_surrogate, identity_of, surrogate_to_doc_id}; +pub use key::{RowIdentity, StorageKey}; diff --git a/nodedb/src/engine/graph/pattern/executor/core/triple.rs b/nodedb/src/engine/graph/pattern/executor/core/triple.rs index 626b74d1a..95279e039 100644 --- a/nodedb/src/engine/graph/pattern/executor/core/triple.rs +++ b/nodedb/src/engine/graph/pattern/executor/core/triple.rs @@ -287,7 +287,7 @@ pub(in crate::engine::graph::pattern::executor) mod tests { /// graph's `(DatabaseId::DEFAULT, TenantId::new(1), "col")`. /// /// `csr` resolves a bound node name to its surrogate; the document is then - /// fetched at `surrogate_to_doc_id(surrogate)`, mirroring the real keying. + /// fetched at `StorageKey::for_surrogate(surrogate)`, mirroring the real keying. pub(crate) fn props_for<'a>(sparse: &'a SparseEngine, csr: &'a CsrIndex) -> PropertyLookup<'a> { PropertyLookup { sparse, @@ -885,7 +885,7 @@ pub(in crate::engine::graph::pattern::executor) mod tests { make_csr(&[("alice", "KNOWS", "carol"), ("bob", "KNOWS", "dave")]); let (sparse, _sdir) = make_sparse(); // alice/bob share their surrogate with their stored document (the real - // keying): node → surrogate → surrogate_to_doc_id → sparse. + // keying): node → surrogate → StorageKey → sparse. csr.set_node_surrogate("alice", nodedb_types::Surrogate::new(1)); csr.set_node_surrogate("bob", nodedb_types::Surrogate::new(2)); let props = props_for(&sparse, &csr); diff --git a/nodedb/src/engine/graph/pattern/executor/predicates.rs b/nodedb/src/engine/graph/pattern/executor/predicates.rs index e248fd25e..89a1c6b27 100644 --- a/nodedb/src/engine/graph/pattern/executor/predicates.rs +++ b/nodedb/src/engine/graph/pattern/executor/predicates.rs @@ -15,11 +15,11 @@ use crate::engine::sparse::btree::SparseEngine; /// /// A graph node's properties live as a document in the sparse engine. The /// document is NOT keyed by the user-visible node-id string — it is keyed by -/// `surrogate_to_doc_id(surrogate)`, the fixed-width hex form of the row's -/// global surrogate. A graph node and its same-pk document share one surrogate -/// (the CSR node surrogate is set from the edge surrogate allocated by the same -/// pk-keyed allocator), so the fetch chain is: -/// `node name → Surrogate (via `csr`) → surrogate_to_doc_id → sparse.get`. +/// `StorageKey::for_surrogate(surrogate)`, the fixed-width hex form of the +/// row's global surrogate. A graph node and its same-pk document share one +/// surrogate (the CSR node surrogate is set from the edge surrogate +/// allocated by the same pk-keyed allocator), so the fetch chain is: +/// `node name → Surrogate (via `csr`) → StorageKey → sparse.get`. /// /// The CSR/graph is keyed per `(database_id, tenant_id)` only, so the collection /// holding the document comes from the MATCH query's `IN ''` @@ -237,7 +237,7 @@ fn coerce_literal(expected: &str) -> nodedb_types::Value { /// The document is keyed by the node's GLOBAL SURROGATE, not by the node-id /// string. We resolve `node_id → Surrogate` through the CSR (a graph node and /// its same-pk document share one surrogate), derive the redb storage key via -/// `surrogate_to_doc_id`, then fetch. A node that is unknown to the partition +/// `StorageKey::for_surrogate`, then fetch. A node that is unknown to the partition /// or has no surrogate set (the ZERO sentinel) is treated as having no /// document → `Ok(None)`. /// diff --git a/nodedb/src/engine/sparse/btree/chain_head.rs b/nodedb/src/engine/sparse/btree/chain_head.rs index 95c49c90c..7ddcd96d1 100644 --- a/nodedb/src/engine/sparse/btree/chain_head.rs +++ b/nodedb/src/engine/sparse/btree/chain_head.rs @@ -22,7 +22,7 @@ //! `"{database_id}:{tenant_id}:{collection}"`. A separate table is what makes //! collision with a document row structurally impossible — document rows live //! in `DOCUMENTS` under `"{database_id}:{tenant_id}:{collection}:{document_id}"` -//! where `document_id` is the 8-hex surrogate (`surrogate_to_doc_id`). Storing +//! where `document_id` is the 8-hex surrogate rendered via `StorageKey`. Storing //! the head as a sentinel row inside `DOCUMENTS` would be worse than a //! collision risk: every document scan is a prefix range over that table and //! would return the head as if it were a row. diff --git a/nodedb/tests/inproc/cases/cross_engine_three_way_fts_vector_doc.rs b/nodedb/tests/inproc/cases/cross_engine_three_way_fts_vector_doc.rs index e8c9d7c87..27e9cd469 100644 --- a/nodedb/tests/inproc/cases/cross_engine_three_way_fts_vector_doc.rs +++ b/nodedb/tests/inproc/cases/cross_engine_three_way_fts_vector_doc.rs @@ -125,7 +125,7 @@ fn extract_vector_surrogates(payload: &[u8]) -> Vec { /// Extract surrogate u32 values from a `TextOp::Search` response. /// FTS hits share the document-scan envelope `{id, data}`; `id` is the -/// 8-char hex surrogate produced by `surrogate_to_doc_id`. +/// 8-char hex surrogate produced by `StorageKey`'s `Display` impl. fn extract_fts_surrogates(payload: &[u8]) -> Vec { parse_json(payload) .as_array() diff --git a/nodedb/tests/inproc/cases/executor_tests/test_conditional_update.rs b/nodedb/tests/inproc/cases/executor_tests/test_conditional_update.rs index db96b8534..ed824abee 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_conditional_update.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_conditional_update.rs @@ -27,7 +27,7 @@ fn filter(field: &str, op: &str, value: nodedb_types::Value) -> ScanFilter { /// Hash a string PK to a deterministic non-zero surrogate so each test /// row lands on its own substrate key. The data plane keys redb rows by -/// `surrogate_to_doc_id(surrogate)`; with a wired catalog the assigner +/// `StorageKey::for_surrogate(surrogate)`; with a wired catalog the assigner /// guarantees a stable injection, but executor-direct fixtures bypass /// the catalog and have to thread their own bindings. fn surrogate_for(id: &str) -> nodedb_types::Surrogate { diff --git a/nodedb/tests/inproc/cases/executor_tests/test_range_scan_bitemporal.rs b/nodedb/tests/inproc/cases/executor_tests/test_range_scan_bitemporal.rs index b3790df6d..981a75167 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_range_scan_bitemporal.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_range_scan_bitemporal.rs @@ -224,7 +224,7 @@ fn schemaless_bitemporal_update_moves_row_in_range() { seed_four(&mut ctx, "events"); // The versioned row identity is the SURROGATE (a PointPut keys its - // version by `surrogate_to_doc_id(surrogate)`), so an UPDATE must reuse + // version by `StorageKey::for_surrogate(surrogate)`), so an UPDATE must reuse // the row's original surrogate to append a new version of the SAME row — // a fresh surrogate would create a distinct row instead of superseding. // Seed surrogates are 1..4 for d1..d4. diff --git a/nodedb/tests/wire/cases/sql_hybrid_search.rs b/nodedb/tests/wire/cases/sql_hybrid_search.rs index 3bdf288ac..80a51f8ee 100644 --- a/nodedb/tests/wire/cases/sql_hybrid_search.rs +++ b/nodedb/tests/wire/cases/sql_hybrid_search.rs @@ -238,7 +238,7 @@ async fn rrf_score_in_select_without_order_by_returns_score() { // ── 5. `id` must be the user's primary key, not the internal surrogate ───── /// Pre-fix, `TextOp::HybridSearch` hits carried only `doc_id` (the DP-side -/// `surrogate_to_doc_id(surrogate)` hex string) with no surrogate->PK +/// `StorageKey::for_surrogate(surrogate)` hex string) with no surrogate->PK /// translation on the Text/Hybrid plan path (only the Vector plan path had /// one) — so `SELECT id` against a hybrid query had no `id` field to read at /// all and the cell came back empty. `create_hybrid_collection` inserts rows