diff --git a/nodedb-client/src/native/connection/response.rs b/nodedb-client/src/native/connection/response.rs index a1cd73131..894ae949e 100644 --- a/nodedb-client/src/native/connection/response.rs +++ b/nodedb-client/src/native/connection/response.rs @@ -54,6 +54,7 @@ pub(super) fn response_to_query_result(resp: NativeResponse) -> NodeDbResult Opti }) } -/// Find two collection names whose vShard ids differ. -fn two_distinct_vshard_collections() -> (String, String) { - let mut first: Option<(String, u32)> = None; - for i in 0u32..512 { - let name = format!("calvin_e2e_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); - if let Some((ref fname, fv)) = first { - if fv != vshard { - return (fname.clone(), name); - } - } else { - first = Some((name, vshard)); - } - } - panic!("could not find two distinct-vshard collections in 512 tries"); -} - /// Calvin multi-shard batch via pgwire `simple_query` COMMITS when sent as an /// interactive `BEGIN ... COMMIT` block. /// @@ -124,7 +107,7 @@ async fn calvin_multishard_write_in_explicit_block_commits() { ) .await; - let (col_a, col_b) = two_distinct_vshard_collections(); + let (col_a, col_b) = distinct_vshard_collections("calvin_e2e_0", "calvin_e2e"); // Create both collections on this (single) node. node.client diff --git a/nodedb-cluster-tests/tests/common_suite/cases/calvin_multi_shard_bitemporal_best_effort_restart.rs b/nodedb-cluster-tests/tests/common_suite/cases/calvin_multi_shard_bitemporal_best_effort_restart.rs index 3f2dbd5c9..650e59af1 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/calvin_multi_shard_bitemporal_best_effort_restart.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/calvin_multi_shard_bitemporal_best_effort_restart.rs @@ -11,8 +11,7 @@ //! //! 1. Two `document_schemaless` collections, each created `WITH //! (bitemporal=true)`, are placed on DIFFERENT vShards -//! (`distinct_vshard_bitemporal_collections`, same technique as -//! `calvin_multi_shard_redo_restart.rs`). +//! (`vshard_names::distinct_vshard_collections`). //! 2. `BEGIN; INSERT INTO ; INSERT INTO ; COMMIT` is sent as ONE //! `simple_query` call, so both writes are buffered inside the block and, //! on COMMIT, `classify_dispatch` sees writes on two vShards → MultiShard. @@ -32,9 +31,9 @@ use crate::common; use std::sync::atomic::Ordering; use std::time::Duration; -use nodedb::types::{DatabaseId, VShardId}; use tokio_postgres::SimpleQueryMessage; +use super::vshard_names::distinct_vshard_collections; use common::cluster_harness::{TestClusterNode, read_once_a_leader_exists, wait_for}; /// Observed sequencer-group leader id from a node's local Raft status, or `0` @@ -61,26 +60,6 @@ fn admitted_total(node: &TestClusterNode) -> u64 { .unwrap_or(0) } -/// A `(coll_a, coll_b)` pair of bitemporal collection names whose vShard ids -/// differ, so a transaction writing to both is genuinely multi-shard. -/// Deterministic: `VShardId::from_collection_in_database` is a pure function -/// of the database id + collection name bytes. Same technique as -/// `calvin_multi_shard_redo_restart.rs::distinct_vshard_collections`. -fn distinct_vshard_bitemporal_collections() -> (String, String) { - let a_name = "bt_a".to_string(); - let va = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &a_name).as_u32(); - for i in 0u32..512 { - let b_name = format!("bt_b_{i}"); - if VShardId::from_collection_in_database(DatabaseId::DEFAULT, &b_name).as_u32() != va { - return (a_name, b_name); - } - } - panic!( - "could not find a second bitemporal collection name on a distinct vShard \ - from the first in 512 tries" - ); -} - /// Current `value` for `id` in a bitemporal document collection, or `None` if /// not visible. /// @@ -160,7 +139,7 @@ async fn calvin_multi_shard_bitemporal_best_effort_commit_survives_wal_only_rest ) .await; - let (coll_a, coll_b) = distinct_vshard_bitemporal_collections(); + let (coll_a, coll_b) = distinct_vshard_collections("bt_a", "bt_b"); node.client .simple_query(&format!( diff --git a/nodedb-cluster-tests/tests/common_suite/cases/calvin_multi_shard_bitemporal_restart.rs b/nodedb-cluster-tests/tests/common_suite/cases/calvin_multi_shard_bitemporal_restart.rs index 780e24249..67d690f1c 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/calvin_multi_shard_bitemporal_restart.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/calvin_multi_shard_bitemporal_restart.rs @@ -9,8 +9,7 @@ //! //! 1. Two `document_schemaless` collections, each created `WITH //! (bitemporal=true)`, are placed on DIFFERENT vShards -//! (`distinct_vshard_bitemporal_collections`, same technique as -//! `calvin_multi_shard_redo_restart.rs`). +//! (`vshard_names::distinct_vshard_collections`). //! 2. `BEGIN; INSERT INTO ; INSERT INTO ; COMMIT` is sent as ONE //! `simple_query` call, so both writes are buffered inside the block and, //! on COMMIT, `classify_dispatch` sees writes on two vShards → MultiShard @@ -29,9 +28,9 @@ use crate::common; use std::sync::atomic::Ordering; use std::time::Duration; -use nodedb::types::{DatabaseId, VShardId}; use tokio_postgres::SimpleQueryMessage; +use super::vshard_names::distinct_vshard_collections; use common::cluster_harness::{TestClusterNode, read_once_a_leader_exists, wait_for}; /// Observed sequencer-group leader id from a node's local Raft status, or `0` @@ -58,26 +57,6 @@ fn admitted_total(node: &TestClusterNode) -> u64 { .unwrap_or(0) } -/// A `(coll_a, coll_b)` pair of bitemporal collection names whose vShard ids -/// differ, so a transaction writing to both is genuinely multi-shard. -/// Deterministic: `VShardId::from_collection_in_database` is a pure function -/// of the database id + collection name bytes. Same technique as -/// `calvin_multi_shard_redo_restart.rs::distinct_vshard_collections`. -fn distinct_vshard_bitemporal_collections() -> (String, String) { - let a_name = "bt_a".to_string(); - let va = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &a_name).as_u32(); - for i in 0u32..512 { - let b_name = format!("bt_b_{i}"); - if VShardId::from_collection_in_database(DatabaseId::DEFAULT, &b_name).as_u32() != va { - return (a_name, b_name); - } - } - panic!( - "could not find a second bitemporal collection name on a distinct vShard \ - from the first in 512 tries" - ); -} - /// Current `value` for `id` in a bitemporal document collection, or `None` if /// not visible. /// @@ -156,7 +135,7 @@ async fn calvin_multi_shard_bitemporal_commit_survives_wal_only_restart() { ) .await; - let (coll_a, coll_b) = distinct_vshard_bitemporal_collections(); + let (coll_a, coll_b) = distinct_vshard_collections("bt_a", "bt_b"); node.client .simple_query(&format!( diff --git a/nodedb-cluster-tests/tests/common_suite/cases/calvin_multi_shard_redo_restart.rs b/nodedb-cluster-tests/tests/common_suite/cases/calvin_multi_shard_redo_restart.rs index f3762cfbe..921409a28 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/calvin_multi_shard_redo_restart.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/calvin_multi_shard_redo_restart.rs @@ -6,8 +6,8 @@ //! `vector_index_txn_restart.rs` for the single-shard analogue). //! //! 1. Two collections — a KV collection and a vector-indexed document -//! collection — are created on DIFFERENT vShards (`distinct_vshard_ -//! collections`, same technique as `calvin_cluster_pgwire_e2e.rs`). +//! collection — are created on DIFFERENT vShards +//! (`vshard_names::distinct_vshard_collections`). //! 2. `BEGIN; INSERT INTO ; INSERT INTO ; COMMIT` is sent as ONE //! `simple_query` call. tokio-postgres ships this as a single wire //! message; the server buffers the two INSERTs during the transaction and, @@ -25,9 +25,9 @@ use crate::common; use std::sync::atomic::Ordering; use std::time::Duration; -use nodedb::types::{DatabaseId, VShardId}; use tokio_postgres::SimpleQueryMessage; +use super::vshard_names::distinct_vshard_collections; use common::cluster_harness::{TestClusterNode, read_once_a_leader_exists, wait_for}; /// Observed sequencer-group leader id from a node's local Raft status, or `0` @@ -54,26 +54,6 @@ fn admitted_total(node: &TestClusterNode) -> u64 { .unwrap_or(0) } -/// A `(kv_name, vec_name)` pair of collection names whose vShard ids differ, -/// so a transaction writing to both is genuinely multi-shard. Deterministic: -/// `VShardId::from_collection_in_database` is a pure function of the database -/// id + collection name bytes. Same technique as -/// `calvin_cluster_pgwire_e2e.rs::two_distinct_vshard_collections`. -fn distinct_vshard_collections() -> (String, String) { - let kv_name = "cmr_kv".to_string(); - let vkv = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &kv_name).as_u32(); - for i in 0u32..512 { - let vec_name = format!("cmr_vecdocs_{i}"); - if VShardId::from_collection_in_database(DatabaseId::DEFAULT, &vec_name).as_u32() != vkv { - return (kv_name, vec_name); - } - } - panic!( - "could not find a vector-doc collection name on a distinct vShard from \ - the KV collection in 512 tries" - ); -} - /// Single-row `col` value for `id` in a KV/document collection, or `None` if /// not visible. /// @@ -140,7 +120,7 @@ async fn calvin_multi_shard_write_in_explicit_block_commits_and_survives_restart ) .await; - let (kv, vecdocs) = distinct_vshard_collections(); + let (kv, vecdocs) = distinct_vshard_collections("cmr_kv", "cmr_vecdocs"); node.client .simple_query(&format!( diff --git a/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_dml_tag_fold.rs b/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_dml_tag_fold.rs new file mode 100644 index 000000000..bb0e08632 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_dml_tag_fold.rs @@ -0,0 +1,214 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Statement-level DML tag fold on the Calvin multi-shard path, from a +//! coordinator that is NOT the sequencer leader (routed submit). +//! +//! A statement whose tasks span vShards commits as ONE Calvin batch and +//! answers ONE command tag with the REAL affected count: never one tag per +//! task, never the task count, never a bare `OK`. `tokio_postgres` hides the +//! tag verb, so pgwire assertions read the wire through `RawPgConn`; the +//! native protocol reports the same fold as `(rows_affected, command)`. +//! +//! `MERGE INTO a USING b` across two vShards is NOT a Calvin batch: +//! `DocumentOp::Merge` is not a Calvin write (`calvin/write_class.rs`), so +//! `classify_dispatch` sees zero write vShards and the autocommit statement +//! reaches `merge_orchestrator::run_authorized_merge`, which scans the source +//! on its own core, resolves the arms, and proposes the resolved apply through +//! Raft to the target's owner and every replica. The MERGE test pins the tag +//! that path answers and that the rows land on every node. + +use super::calvin_multishard_fixture::{ + Fixture, edge_batch_sql, edge_doc_sql, keyed_ddl, native_outcome, schemaless_ddl, tags, +}; +use super::vshard_names::distinct_vshard_collections; + +/// An autocommit implicit-edge INSERT fans out to a document task and an +/// `EdgePut` task on another vShard. The statement answers ONE `INSERT 0 1`; +/// a three-row `VALUES` list answers ONE `INSERT 0 3`. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn cross_shard_single_statement_reports_one_folded_insert_tag_pgwire() { + let coll = "tagfold_edges_pg"; + let fx = Fixture::spawn(&[schemaless_ddl(coll)]).await; + let mut conn = fx.raw().await; + + assert_eq!( + tags(&mut conn, &edge_doc_sql(coll, "e1", "keep")).await, + vec!["INSERT 0 1"], + "document + cross-vShard edge task fold into one INSERT tag" + ); + assert_eq!( + tags(&mut conn, &edge_batch_sql(coll, ["e2", "e3", "e4"])).await, + vec!["INSERT 0 3"], + "three-row VALUES with cross-vShard edge tasks folds into one INSERT tag" + ); + + fx.converge().await; + assert_eq!( + fx.count_on_coordinator(&format!("SELECT id FROM {coll}")) + .await, + 4 + ); + + fx.cluster.shutdown().await; +} + +/// Native-protocol mirror of the pgwire fold: `(1, INSERT)` then `(3, INSERT)`. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn cross_shard_single_statement_reports_one_folded_insert_tag_native() { + let coll = "tagfold_edges_nat"; + let fx = Fixture::spawn(&[schemaless_ddl(coll)]).await; + let node = fx.coordinator(); + + assert_eq!( + native_outcome(node, &edge_doc_sql(coll, "e1", "keep")).await, + (1, Some("INSERT".to_owned())), + "native single-row implicit-edge insert reports the document count" + ); + assert_eq!( + native_outcome(node, &edge_batch_sql(coll, ["e2", "e3", "e4"])).await, + (3, Some("INSERT".to_owned())), + "native three-row implicit-edge insert reports the row count" + ); + + fx.converge().await; + assert_eq!( + fx.count_on_coordinator(&format!("SELECT id FROM {coll}")) + .await, + 4 + ); + + fx.cluster.shutdown().await; +} + +/// A predicate DELETE over implicit-edge documents runs through OLLP/Calvin +/// with one edge-delete task per matched row on other vShards. The tag counts +/// the two deleted documents, not the tasks. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn cross_shard_delete_reports_real_count_never_task_count() { + let pg_coll = "tagfold_del_pg"; + let nat_coll = "tagfold_del_nat"; + let fx = Fixture::spawn(&[schemaless_ddl(pg_coll), schemaless_ddl(nat_coll)]).await; + let mut conn = fx.raw().await; + let node = fx.coordinator(); + + assert_eq!( + tags(&mut conn, &edge_batch_sql(pg_coll, ["d1", "d2", "d3"])).await, + vec!["INSERT 0 3"] + ); + assert_eq!( + native_outcome(node, &edge_batch_sql(nat_coll, ["d1", "d2", "d3"])).await, + (3, Some("INSERT".to_owned())) + ); + fx.converge().await; + + assert_eq!( + tags( + &mut conn, + &format!("DELETE FROM {pg_coll} WHERE mark = 'del'") + ) + .await, + vec!["DELETE 2"], + "pgwire predicate delete counts deleted documents, not edge tasks" + ); + assert_eq!( + native_outcome(node, &format!("DELETE FROM {nat_coll} WHERE mark = 'del'")).await, + (2, Some("DELETE".to_owned())), + "native predicate delete counts deleted documents, not edge tasks" + ); + + fx.converge().await; + assert_eq!( + fx.count_on_coordinator(&format!("SELECT id FROM {pg_coll}")) + .await, + 1 + ); + assert_eq!( + fx.count_on_coordinator(&format!("SELECT id FROM {nat_coll}")) + .await, + 1 + ); + + fx.cluster.shutdown().await; +} + +/// `MERGE INTO target USING source` with the two collections on distinct +/// vShards answers `MERGE ` with `n` = matched updates + unmatched inserts, +/// on pgwire and native alike. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn cross_shard_merge_reports_merge_tag() { + let (target, source) = distinct_vshard_collections("tagfold_merge_tgt", "tagfold_merge_src"); + let fx = Fixture::spawn(&[keyed_ddl(&target), keyed_ddl(&source)]).await; + let mut conn = fx.raw().await; + let node = fx.coordinator(); + + assert_eq!( + tags( + &mut conn, + &format!("INSERT INTO {target} (id, v) VALUES ('k1', 'old')") + ) + .await, + vec!["INSERT 0 1"] + ); + assert_eq!( + tags( + &mut conn, + &format!("INSERT INTO {source} (id, v) VALUES ('k1', 'new'), ('k2', 'two')") + ) + .await, + vec!["INSERT 0 2"] + ); + fx.converge().await; + + let merge = format!( + "MERGE INTO {target} t USING {source} s ON t.id = s.id \ + WHEN MATCHED THEN UPDATE SET v = s.v \ + WHEN NOT MATCHED THEN INSERT (id, v) VALUES (s.id, s.v)" + ); + assert_eq!( + tags(&mut conn, &merge).await, + vec!["MERGE 2"], + "pgwire MERGE across vShards: one update + one insert" + ); + fx.converge().await; + assert_eq!( + fx.count_on_coordinator(&format!("SELECT id FROM {target} WHERE v = 'new'")) + .await, + 1 + ); + assert_eq!( + fx.count_on_coordinator(&format!("SELECT id FROM {target} WHERE id = 'k2'")) + .await, + 1 + ); + fx.wait_rows_on_every_node(&format!("SELECT id FROM {target} WHERE v = 'new'"), 1) + .await; + fx.wait_rows_on_every_node(&format!("SELECT id FROM {target} WHERE id = 'k2'"), 1) + .await; + + assert_eq!( + native_outcome( + node, + &format!("INSERT INTO {source} (id, v) VALUES ('k3', 'three')") + ) + .await, + (1, Some("INSERT".to_owned())) + ); + fx.converge().await; + assert_eq!( + native_outcome(node, &merge).await, + (3, Some("MERGE".to_owned())), + "native MERGE across vShards: two updates + one insert" + ); + fx.converge().await; + assert_eq!( + fx.count_on_coordinator(&format!("SELECT id FROM {target}")) + .await, + 3 + ); + fx.wait_rows_on_every_node(&format!("SELECT id FROM {target}"), 3) + .await; + fx.wait_rows_on_every_node(&format!("SELECT id FROM {target} WHERE id = 'k3'"), 1) + .await; + + fx.cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_fixture.rs b/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_fixture.rs new file mode 100644 index 000000000..7e3d1d965 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_fixture.rs @@ -0,0 +1,229 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Shared fixture for the Calvin multi-shard DML cases: a 3-node cluster with +//! a coordinator that is not the sequencer leader, raw pgwire in strict +//! cross-shard mode, and per-node row-count waits. +//! +//! Cross-shard fan-out comes from two sources: +//! - explicit transactions writing two collections on distinct vShards +//! (`vshard_names::distinct_vshard_collections`), +//! - implicit graph edges whose `_from` endpoint hashes (`VShardId::from_key`) +//! to a vShard other than the document's collection +//! (`vshard_names::key_on_other_vshard`). + +use std::time::Duration; + +use tokio_postgres::SimpleQueryMessage; + +use super::vshard_names::key_on_other_vshard; +use crate::common::cluster_harness::{ + TestCluster, TestClusterNode, is_no_serving_leader, wait_for, wait_for_async, +}; +use crate::common::pgwire_harness::raw_pgwire::{RawPgConn, command_tags}; + +const CONVERGENCE: Duration = Duration::from_secs(15); + +/// A 3-node cluster with a coordinator that is not the sequencer leader. +pub(super) struct Fixture { + pub(super) cluster: TestCluster, + pub(super) coordinator: usize, +} + +impl Fixture { + /// Spawn, run every `ddl` statement, and wait for the collections and a + /// stable sequencer leader to be visible on every node. + pub(super) async fn spawn(ddl: &[String]) -> Self { + let fx = + Self::from_cluster(TestCluster::spawn_three().await.expect("3-node cluster")).await; + fx.create(ddl).await; + fx + } + + /// Wait for a stable sequencer leader on every node and pick a + /// coordinator that is not it. + pub(super) async fn from_cluster(cluster: TestCluster) -> Self { + wait_for( + "sequencer-group leader elected and visible on every node", + CONVERGENCE, + Duration::from_millis(50), + || { + let leader = cluster.nodes[0].sequencer_leader(); + leader != 0 && cluster.nodes.iter().all(|n| n.sequencer_leader() == leader) + }, + ) + .await; + let leader = cluster.nodes[0].sequencer_leader(); + let coordinator = cluster + .nodes + .iter() + .position(|n| n.shared.node_id != leader) + .expect("a non-sequencer-leader coordinator exists in a 3-node cluster"); + Self { + cluster, + coordinator, + } + } + + /// Run every `ddl` statement and wait for the collections to be visible + /// on every node. + pub(super) async fn create(&self, ddl: &[String]) { + let before = self + .cluster + .nodes + .iter() + .map(|n| n.cached_collection_count()) + .min() + .unwrap_or(0); + for stmt in ddl { + self.cluster + .exec_ddl_on_any_leader(stmt) + .await + .unwrap_or_else(|e| panic!("{stmt}: {e}")); + } + let expected = before + ddl.len(); + wait_for( + "all 3 nodes see every collection", + CONVERGENCE, + Duration::from_millis(50), + || { + self.cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= expected) + }, + ) + .await; + } + + pub(super) fn coordinator(&self) -> &TestClusterNode { + &self.cluster.nodes[self.coordinator] + } + + /// Wait until `collection`'s data group is mounted cluster-wide: every + /// node resolves it to a group id, and at least one node actually hosts + /// that group locally (`hosts_data_group`, a live self-report, not a + /// placement prediction). + pub(super) async fn wait_group_mounted(&self, collection: &str) { + wait_for( + &format!("{collection}'s data group mounted"), + CONVERGENCE, + Duration::from_millis(50), + || { + self.cluster + .nodes + .iter() + .all(|n| n.group_id_for_collection(collection).is_some()) + && self.cluster.nodes.iter().any(|n| { + n.group_id_for_collection(collection) + .is_some_and(|gid| n.hosts_data_group(gid)) + }) + }, + ) + .await; + } + + /// A raw pgwire session on the coordinator in strict cross-shard mode. + pub(super) async fn raw(&self) -> RawPgConn { + let mut conn = self.coordinator().raw_pgwire().await; + conn.simple_query("SET cross_shard_txn = 'strict'").await; + conn + } + + pub(super) async fn converge(&self) { + self.cluster + .wait_for_full_apply_convergence(CONVERGENCE) + .await; + } + + /// Row count of `sql` on the coordinator. + pub(super) async fn count_on_coordinator(&self, sql: &str) -> usize { + row_count(&self.coordinator().client, sql).await + } + + /// Wait until `sql` returns exactly `expected` rows on every node. + pub(super) async fn wait_rows_on_every_node(&self, sql: &str, expected: usize) { + for idx in 0..self.cluster.nodes.len() { + self.wait_rows_on_node(idx, sql, expected).await; + } + } + + /// Wait until `sql` returns exactly `expected` rows on node `idx`. + pub(super) async fn wait_rows_on_node(&self, idx: usize, sql: &str, expected: usize) { + let node = &self.cluster.nodes[idx]; + wait_for_async( + &format!("node {idx}: `{sql}` returns {expected} rows"), + CONVERGENCE, + Duration::from_millis(100), + || async move { + match node.client.simple_query(sql).await { + Ok(msgs) => data_rows(&msgs) == expected, + Err(e) if is_no_serving_leader(&e) => false, + Err(e) => panic!("node {idx}: `{sql}`: {e}"), + } + }, + ) + .await; + } +} + +pub(super) fn data_rows(msgs: &[SimpleQueryMessage]) -> usize { + msgs.iter() + .filter(|m| matches!(m, SimpleQueryMessage::Row(_))) + .count() +} + +pub(super) async fn row_count(client: &tokio_postgres::Client, sql: &str) -> usize { + let msgs = client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("`{sql}`: {e}")); + data_rows(&msgs) +} + +/// Send one statement on the raw connection and return its command tags. +pub(super) async fn tags(conn: &mut RawPgConn, sql: &str) -> Vec { + command_tags(&conn.simple_query(sql).await) +} + +/// Native `(rows_affected, command)` for one statement on the coordinator. +pub(super) async fn native_outcome(node: &TestClusterNode, sql: &str) -> (u64, Option) { + let result = node + .native_client() + .query(sql) + .await + .unwrap_or_else(|e| panic!("native `{sql}`: {e}")); + (result.rows_affected, result.command) +} + +pub(super) fn schemaless_ddl(coll: &str) -> String { + format!("CREATE COLLECTION {coll} WITH (engine='document_schemaless')") +} + +pub(super) fn keyed_ddl(coll: &str) -> String { + format!("CREATE COLLECTION {coll} (id TEXT PRIMARY KEY, v TEXT)") +} + +/// One implicit-edge document whose `_from` hashes away from `coll`'s vShard. +pub(super) fn edge_doc_sql(coll: &str, id: &str, mark: &str) -> String { + let src = key_on_other_vshard(coll, &format!("src_{id}")); + format!( + "INSERT INTO {coll} \ + {{ id: '{id}', _from: '{src}', _to: 'hub', _type: 'l', mark: '{mark}' }}" + ) +} + +/// Three implicit-edge documents in one `VALUES` list; two carry `mark='del'`. +pub(super) fn edge_batch_sql(coll: &str, ids: [&str; 3]) -> String { + let rows: Vec = ids + .iter() + .zip(["del", "del", "keep"]) + .map(|(id, mark)| { + let src = key_on_other_vshard(coll, &format!("src_{id}")); + format!("('{id}', '{src}', 'hub', 'l', '{mark}')") + }) + .collect(); + format!( + "INSERT INTO {coll} (id, _from, _to, _type, mark) VALUES {}", + rows.join(", ") + ) +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_pk_read.rs b/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_pk_read.rs new file mode 100644 index 000000000..c49979c32 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_pk_read.rs @@ -0,0 +1,163 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Point reads by primary key after a Calvin write, on every node. +//! +//! The coordinator assigns each row's surrogate in its own catalog at plan +//! time. Every Calvin participant — leader and follower — installs that +//! `pk → surrogate` binding when it applies its slice, so a later +//! `WHERE id = ...` resolves on every node, including one that is neither the +//! coordinator nor a member of the vShard's group (it forwards to the owner, +//! which re-resolves the key against its own catalog). + +use super::calvin_multishard_fixture::{Fixture, keyed_ddl, tags}; +use super::vshard_names::distinct_vshard_collections; + +/// `BEGIN; INSERT a; INSERT b; COMMIT` from a non-leader coordinator, one wire +/// message per statement: each in-block INSERT answers `INSERT 0 1`, COMMIT +/// answers exactly `COMMIT`, and both rows land on every node. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn cross_shard_transaction_commit_reports_commit_and_rows_land() { + let (col_a, col_b) = distinct_vshard_collections("tagfold_txn_a", "tagfold_txn_b"); + let fx = Fixture::spawn(&[keyed_ddl(&col_a), keyed_ddl(&col_b)]).await; + let mut conn = fx.raw().await; + + assert_eq!(tags(&mut conn, "BEGIN").await, vec!["BEGIN"]); + assert_eq!( + tags( + &mut conn, + &format!("INSERT INTO {col_a} (id, v) VALUES ('k1', 'hello')") + ) + .await, + vec!["INSERT 0 1"], + "in-block INSERT into {col_a} answers its own tag" + ); + assert_eq!( + tags( + &mut conn, + &format!("INSERT INTO {col_b} (id, v) VALUES ('k2', 'world')") + ) + .await, + vec!["INSERT 0 1"], + "in-block INSERT into {col_b} answers its own tag" + ); + assert_eq!( + tags(&mut conn, "COMMIT").await, + vec!["COMMIT"], + "the Calvin flush answers exactly one COMMIT tag" + ); + + fx.converge().await; + fx.wait_rows_on_every_node(&format!("SELECT v FROM {col_a} WHERE id = 'k1'"), 1) + .await; + fx.wait_rows_on_every_node(&format!("SELECT v FROM {col_b} WHERE id = 'k2'"), 1) + .await; + + fx.cluster.shutdown().await; +} + +/// An autocommit single-collection INSERT from a coordinator that does not +/// own the collection's vShard answers `INSERT 0 1`, and the row resolves by +/// primary key on every node. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn cross_shard_autocommit_insert_is_readable_by_pk_on_every_node() { + let coll = "tagfold_autocommit_pk"; + let fx = Fixture::spawn(&[keyed_ddl(coll)]).await; + let mut conn = fx.raw().await; + + assert_eq!( + tags( + &mut conn, + &format!("INSERT INTO {coll} (id, v) VALUES ('k1', 'hello')") + ) + .await, + vec!["INSERT 0 1"] + ); + + fx.converge().await; + fx.wait_rows_on_every_node(&format!("SELECT v FROM {coll} WHERE id = 'k1'"), 1) + .await; + fx.wait_rows_on_every_node(&format!("SELECT v FROM {coll} WHERE id = 'absent'"), 0) + .await; + + fx.cluster.shutdown().await; +} + +/// A node added as a Raft learner after both data groups are mounted joins +/// them as a non-voting member: it applies the replicated log but never +/// coordinated the transaction, so its catalog holds no binding the +/// coordinator minted. +/// +/// After a Calvin commit from one of the original three, a point read by +/// primary key from the learner resolves: the owner it forwards to, and the +/// learner's own apply, must both install the coordinator's binding. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn cross_node_pk_read_from_learner_node_after_calvin_commit() { + let (col_a, col_b) = distinct_vshard_collections("tagfold_nm_a", "tagfold_nm_b"); + let mut fx = Fixture::spawn(&[keyed_ddl(&col_a), keyed_ddl(&col_b)]).await; + fx.wait_group_mounted(&col_a).await; + fx.wait_group_mounted(&col_b).await; + + let gid_a = fx.cluster.nodes[0] + .group_id_for_collection(&col_a) + .expect("col_a group resolved after mount"); + + // Add a 4th node as a learner now that both groups are already mounted + // by the original 3 nodes at the default replication factor (3). + let learner_id = fx + .cluster + .add_learner_node() + .await + .expect("add learner node") + .node_id; + let reader = fx + .cluster + .nodes + .iter() + .position(|n| n.node_id == learner_id) + .expect("learner present in cluster"); + + // The 4th node joins `col_a`'s group as a non-voting learner: it + // applies the replicated log but never coordinated this transaction, + // so its catalog holds no binding the coordinator minted. + let status = fx.cluster.nodes[reader].group_status_line(gid_a); + assert!( + status.contains("role=Learner"), + "node {learner_id} must join {col_a}'s group {gid_a} as a learner: {status}" + ); + + // Coordinator is one of the original 3, chosen by the fixture default + // (not the sequencer leader); it hosts both groups since it was present + // when they were created. + let mut conn = fx.raw().await; + + assert_eq!(tags(&mut conn, "BEGIN").await, vec!["BEGIN"]); + assert_eq!( + tags( + &mut conn, + &format!("INSERT INTO {col_a} (id, v) VALUES ('k1', 'hello')") + ) + .await, + vec!["INSERT 0 1"] + ); + assert_eq!( + tags( + &mut conn, + &format!("INSERT INTO {col_b} (id, v) VALUES ('k2', 'world')") + ) + .await, + vec!["INSERT 0 1"] + ); + assert_eq!(tags(&mut conn, "COMMIT").await, vec!["COMMIT"]); + + fx.converge().await; + fx.wait_rows_on_every_node(&format!("SELECT v FROM {col_a} WHERE id = 'k1'"), 1) + .await; + fx.wait_rows_on_node( + reader, + &format!("SELECT v FROM {col_a} WHERE id = 'absent'"), + 0, + ) + .await; + + fx.cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_txn_staging_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_txn_staging_cross_node.rs new file mode 100644 index 000000000..b76d94931 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_txn_staging_cross_node.rs @@ -0,0 +1,391 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Statements from a coordinator that does NOT own the target collection's +//! vShard, on pgwire and native. +//! +//! An in-block write from such a coordinator stages on the owner under the +//! session's transaction (the staging gate's leader forward), never applies +//! durably at statement time: ROLLBACK discards it on every node, and the +//! same session reads it back before COMMIT while every other session does +//! not. An autocommit write forwarded to the owner answers exactly what a +//! local dispatch answers: its own verb and count, one tag per statement, +//! and `RETURNING` rows as rows. + +use std::time::Duration; + +use nodedb_client::native::NativeClient; +use nodedb_client::native::pool::PoolConfig; +use tokio_postgres::SimpleQueryMessage; + +use crate::common::cluster_harness::{ + TestCluster, TestClusterNode, is_no_serving_leader, wait_for, wait_for_async, +}; +use crate::common::pgwire_harness::raw_pgwire::{RawPgConn, command_tags}; + +const CONVERGENCE: Duration = Duration::from_secs(15); + +/// A 3-node cluster, the node leading `coll`'s Raft group, and a coordinator +/// that does not. +struct Fixture { + cluster: TestCluster, + owner: usize, + coordinator: usize, +} + +/// The leader `node` observes for the group owning `coll`'s vShard, `0` +/// while unknown. +fn owner_of(node: &TestClusterNode, coll: &str) -> u64 { + let Some(group_id) = node.group_id_for_collection(coll) else { + return 0; + }; + node.all_group_leaders() + .into_iter() + .find(|(id, _)| *id == group_id) + .map(|(_, leader)| leader) + .unwrap_or(0) +} + +impl Fixture { + /// Spawn, create `coll` with `ddl`, and wait until every node agrees on + /// the collection, on `coll`'s group leader, and on the sequencer leader. + async fn spawn(coll: &str, ddl: &str) -> Self { + let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); + cluster + .exec_ddl_on_any_leader(ddl) + .await + .unwrap_or_else(|e| panic!("{ddl}: {e}")); + wait_for( + "all 3 nodes see the collection", + CONVERGENCE, + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 1) + }, + ) + .await; + wait_for( + "sequencer-group leader elected and visible on every node", + CONVERGENCE, + Duration::from_millis(50), + || { + let leader = cluster.nodes[0].sequencer_leader(); + leader != 0 && cluster.nodes.iter().all(|n| n.sequencer_leader() == leader) + }, + ) + .await; + wait_for( + "collection group leader elected and visible on every node", + CONVERGENCE, + Duration::from_millis(50), + || { + let leader = owner_of(&cluster.nodes[0], coll); + leader != 0 && cluster.nodes.iter().all(|n| owner_of(n, coll) == leader) + }, + ) + .await; + let leader = owner_of(&cluster.nodes[0], coll); + let owner = cluster + .nodes + .iter() + .position(|n| n.node_id == leader) + .expect("the collection's group leader is a cluster node"); + let coordinator = cluster + .nodes + .iter() + .position(|n| n.node_id != leader) + .expect("a non-owner coordinator exists in a 3-node cluster"); + Self { + cluster, + owner, + coordinator, + } + } + + fn owner(&self) -> &TestClusterNode { + &self.cluster.nodes[self.owner] + } + + fn coordinator(&self) -> &TestClusterNode { + &self.cluster.nodes[self.coordinator] + } + + async fn converge(&self) { + self.cluster + .wait_for_full_apply_convergence(CONVERGENCE) + .await; + } + + /// Row count of `sql` on every node, in node order. + async fn rows_on_every_node(&self, sql: &str) -> Vec { + let mut counts = Vec::with_capacity(self.cluster.nodes.len()); + for node in &self.cluster.nodes { + counts.push(row_count(&node.client, sql).await); + } + counts + } + + /// Wait until `sql` returns exactly `expected` rows on every node. + async fn wait_rows_on_every_node(&self, sql: &str, expected: usize) { + for (idx, node) in self.cluster.nodes.iter().enumerate() { + wait_for_async( + &format!("node {idx}: `{sql}` returns {expected} rows"), + CONVERGENCE, + Duration::from_millis(100), + || async move { + match node.client.simple_query(sql).await { + Ok(msgs) => data_rows(&msgs) == expected, + Err(e) if is_no_serving_leader(&e) => false, + Err(e) => panic!("node {idx}: `{sql}`: {e}"), + } + }, + ) + .await; + } + } +} + +fn data_rows(msgs: &[SimpleQueryMessage]) -> usize { + msgs.iter() + .filter(|m| matches!(m, SimpleQueryMessage::Row(_))) + .count() +} + +async fn row_count(client: &tokio_postgres::Client, sql: &str) -> usize { + let msgs = client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("`{sql}`: {e}")); + data_rows(&msgs) +} + +/// Send one statement on the raw connection and return its command tags. +async fn tags(conn: &mut RawPgConn, sql: &str) -> Vec { + command_tags(&conn.simple_query(sql).await) +} + +/// The first column of every `DataRow` (`D`) in `messages`, as text. +fn first_cells(messages: &[(u8, Vec)]) -> Vec { + messages + .iter() + .filter(|(tag, _)| *tag == b'D') + .map(|(_, body)| { + // i16 column count, then per column an i32 length and its bytes. + let len = i32::from_be_bytes([body[2], body[3], body[4], body[5]]); + let cell = &body[6..6 + usize::try_from(len).expect("a non-null first cell")]; + String::from_utf8_lossy(cell).into_owned() + }) + .collect() +} + +/// Native `(rows_affected, command)` for one statement on `client`. +async fn native_outcome(client: &NativeClient, sql: &str) -> (u64, Option) { + let result = client + .query(sql) + .await + .unwrap_or_else(|e| panic!("native `{sql}`: {e}")); + (result.rows_affected, result.command) +} + +/// A native client pinned to ONE connection, so `begin` / `query` / +/// `rollback` share one server session and one transaction. +fn pinned_native_client(node: &TestClusterNode) -> NativeClient { + node.native_client_with(|base| PoolConfig { + max_size: 1, + ..base + }) +} + +fn keyed_ddl(coll: &str) -> String { + format!("CREATE COLLECTION {coll} (id TEXT PRIMARY KEY, v TEXT)") +} + +fn insert_sql(coll: &str, id: &str, v: &str) -> String { + format!("INSERT INTO {coll} (id, v) VALUES ('{id}', '{v}')") +} + +fn select_by_v(coll: &str, v: &str) -> String { + format!("SELECT id FROM {coll} WHERE v = '{v}'") +} + +/// `BEGIN; INSERT; ROLLBACK` from a non-owner coordinator: the in-block +/// INSERT stages on the owner and answers its count, and ROLLBACK leaves no +/// row on any node. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn cross_node_transaction_rollback_discards_write_on_owner() { + let coll = "txstage_rb_pg"; + let fx = Fixture::spawn(coll, &keyed_ddl(coll)).await; + let mut conn = fx.coordinator().raw_pgwire().await; + + assert_eq!(tags(&mut conn, "BEGIN").await, vec!["BEGIN"]); + assert_eq!( + tags(&mut conn, &insert_sql(coll, "rb", "gone")).await, + vec!["INSERT 0 1"], + "in-block INSERT from a non-owner coordinator answers its staged count" + ); + assert_eq!(tags(&mut conn, "ROLLBACK").await, vec!["ROLLBACK"]); + + fx.converge().await; + assert_eq!( + fx.rows_on_every_node(&select_by_v(coll, "gone")).await, + vec![0; 3], + "a rolled-back in-block INSERT must not land on any node" + ); + + fx.cluster.shutdown().await; +} + +/// The session that staged a write on a remote owner reads it back before +/// COMMIT; another session on the owner does not until COMMIT lands. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn cross_node_transaction_reads_its_own_staged_write() { + let coll = "txstage_ryow_pg"; + let fx = Fixture::spawn(coll, &keyed_ddl(coll)).await; + let mut conn = fx.coordinator().raw_pgwire().await; + + assert_eq!(tags(&mut conn, "BEGIN").await, vec!["BEGIN"]); + assert_eq!( + tags(&mut conn, &insert_sql(coll, "ryow", "staged")).await, + vec!["INSERT 0 1"] + ); + + let own = conn.simple_query(&select_by_v(coll, "staged")).await; + assert_eq!( + first_cells(&own), + vec!["ryow".to_owned()], + "the staging session reads its own write before COMMIT" + ); + assert_eq!( + row_count(&fx.owner().client, &select_by_v(coll, "staged")).await, + 0, + "another session on the owner must not see the staged write" + ); + + assert_eq!(tags(&mut conn, "COMMIT").await, vec!["COMMIT"]); + fx.converge().await; + fx.wait_rows_on_every_node(&select_by_v(coll, "staged"), 1) + .await; + + fx.cluster.shutdown().await; +} + +/// Native twin of the rollback test. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn cross_node_transaction_rollback_discards_write_on_owner_native() { + let coll = "txstage_rb_nat"; + let fx = Fixture::spawn(coll, &keyed_ddl(coll)).await; + let driver = pinned_native_client(fx.coordinator()); + + driver.begin().await.expect("native BEGIN"); + assert_eq!( + native_outcome(&driver, &insert_sql(coll, "rb", "gone")).await, + (1, Some("INSERT".to_owned())), + "in-block native INSERT from a non-owner coordinator answers its staged count" + ); + driver.rollback().await.expect("native ROLLBACK"); + + fx.converge().await; + assert_eq!( + fx.rows_on_every_node(&select_by_v(coll, "gone")).await, + vec![0; 3], + "a rolled-back in-block native INSERT must not land on any node" + ); + + fx.cluster.shutdown().await; +} + +/// Autocommit `UPDATE` / `DELETE` forwarded to the owner answer their own +/// verb and count: `UPDATE 1` / `DELETE 1` on the wire, `(1, Some(verb))` +/// on native. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn cross_node_update_and_delete_report_their_own_tags() { + let coll = "txstage_upd_del"; + let fx = Fixture::spawn(coll, &keyed_ddl(coll)).await; + let mut conn = fx.coordinator().raw_pgwire().await; + let native = fx.coordinator().native_client(); + + for id in ["k1", "k2", "k3", "k4"] { + assert_eq!( + tags(&mut conn, &insert_sql(coll, id, "seed")).await, + vec!["INSERT 0 1"], + "autocommit INSERT of {id} from a non-owner coordinator" + ); + } + fx.converge().await; + + assert_eq!( + tags( + &mut conn, + &format!("UPDATE {coll} SET v = 'changed' WHERE id = 'k1'") + ) + .await, + vec!["UPDATE 1"], + "pgwire UPDATE forwarded to the owner answers its own tag" + ); + assert_eq!( + tags(&mut conn, &format!("DELETE FROM {coll} WHERE id = 'k2'")).await, + vec!["DELETE 1"], + "pgwire DELETE forwarded to the owner answers its own tag" + ); + assert_eq!( + native_outcome( + &native, + &format!("UPDATE {coll} SET v = 'changed' WHERE id = 'k3'") + ) + .await, + (1, Some("UPDATE".to_owned())), + "native UPDATE forwarded to the owner reports its verb and count" + ); + assert_eq!( + native_outcome(&native, &format!("DELETE FROM {coll} WHERE id = 'k4'")).await, + (1, Some("DELETE".to_owned())), + "native DELETE forwarded to the owner reports its verb and count" + ); + + fx.converge().await; + fx.wait_rows_on_every_node(&select_by_v(coll, "changed"), 2) + .await; + fx.wait_rows_on_every_node(&format!("SELECT id FROM {coll}"), 2) + .await; + + fx.cluster.shutdown().await; +} + +/// `INSERT ... RETURNING id` forwarded to the owner answers the row plus the +/// one command tag a local dispatch answers with — never a folded tag in +/// place of the row, never one tag per forwarded payload. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn cross_node_insert_returning_emits_rows_not_a_folded_tag() { + let coll = "txstage_returning"; + let fx = Fixture::spawn(coll, &keyed_ddl(coll)).await; + let mut local = fx.owner().raw_pgwire().await; + let mut forwarded = fx.coordinator().raw_pgwire().await; + + let on_owner = local + .simple_query(&format!("{} RETURNING id", insert_sql(coll, "own", "x"))) + .await; + let on_coordinator = forwarded + .simple_query(&format!("{} RETURNING id", insert_sql(coll, "fwd", "x"))) + .await; + + assert_eq!(first_cells(&on_owner), vec!["own".to_owned()]); + assert_eq!( + first_cells(&on_coordinator), + vec!["fwd".to_owned()], + "the forwarded INSERT ... RETURNING answers its RETURNING row" + ); + let local_tags = command_tags(&on_owner); + assert_eq!(local_tags.len(), 1, "a local RETURNING answers one tag"); + assert_eq!( + command_tags(&on_coordinator), + local_tags, + "the forwarded RETURNING answers the same one tag a local dispatch answers" + ); + + fx.converge().await; + fx.wait_rows_on_every_node(&select_by_v(coll, "x"), 2).await; + + fx.cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/calvin_submit_routed_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/calvin_submit_routed_cross_node.rs index 745b6a373..c6570c1fe 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/calvin_submit_routed_cross_node.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/calvin_submit_routed_cross_node.rs @@ -25,30 +25,13 @@ use std::time::Duration; -use nodedb::types::{DatabaseId, VShardId}; use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; use crate::common; +use super::vshard_names::distinct_vshard_collections; use crate::common::cluster_harness::{TestCluster, wait_for}; -/// Find two collection names whose vShard ids differ. -fn two_distinct_vshard_collections() -> (String, String) { - let mut first: Option<(String, u32)> = None; - for i in 0u32..512 { - let name = format!("calvin_routed_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); - if let Some((ref fname, fv)) = first { - if fv != vshard { - return (fname.clone(), name); - } - } else { - first = Some((name, vshard)); - } - } - panic!("could not find two distinct-vshard collections in 512 tries"); -} - /// Observed sequencer-group leader id from a node's local Raft status, or `0` if /// no leader is known yet. fn sequencer_leader(node: &common::cluster_harness::TestClusterNode) -> u64 { @@ -73,7 +56,7 @@ fn sequencer_leader(node: &common::cluster_harness::TestClusterNode) -> u64 { async fn cross_shard_write_from_non_sequencer_leader_completes() { let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); - let (col_a, col_b) = two_distinct_vshard_collections(); + let (col_a, col_b) = distinct_vshard_collections("calvin_routed_0", "calvin_routed"); cluster .exec_ddl_on_any_leader(&format!( diff --git a/nodedb-cluster-tests/tests/common_suite/cases/mod.rs b/nodedb-cluster-tests/tests/common_suite/cases/mod.rs index 63bad9da2..bd7e362ef 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/mod.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/mod.rs @@ -13,6 +13,10 @@ mod calvin_cluster_pgwire_e2e; mod calvin_multi_shard_bitemporal_best_effort_restart; mod calvin_multi_shard_bitemporal_restart; mod calvin_multi_shard_redo_restart; +mod calvin_multishard_dml_tag_fold; +mod calvin_multishard_fixture; +mod calvin_multishard_pk_read; +mod calvin_multishard_txn_staging_cross_node; mod calvin_ollp_cross_node; mod calvin_ollp_pk_delete; mod calvin_ollp_update; @@ -99,4 +103,5 @@ mod sync_retryable_delta_refusal; mod topic_consumer_group_cross_node; mod vector_alter_params_cross_node; mod vector_index_dispatch_cross_node; +mod vshard_names; mod write_admission_concurrent_same_key_replay; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/vshard_names.rs b/nodedb-cluster-tests/tests/common_suite/cases/vshard_names.rs new file mode 100644 index 000000000..155e91073 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/vshard_names.rs @@ -0,0 +1,47 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Deterministic name picking for cross-vShard tests. +//! +//! vShards are per collection (`VShardId::from_collection_in_database`) and +//! per graph endpoint key (`VShardId::from_key`). Both are pure functions of +//! their input bytes, so a test can pick names that land on distinct vShards +//! without probing the cluster. + +use nodedb::types::{DatabaseId, VShardId}; + +/// Upper bound on candidate names tried before giving up. +const MAX_TRIES: u32 = 512; + +/// `(first, second)` collection names whose vShard ids differ. `first` is +/// used verbatim; `second` is `{second_prefix}_{i}` for the lowest `i` that +/// hashes away from `first`. +pub fn distinct_vshard_collections(first: &str, second_prefix: &str) -> (String, String) { + let first_vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, first); + for i in 0..MAX_TRIES { + let second = format!("{second_prefix}_{i}"); + if VShardId::from_collection_in_database(DatabaseId::DEFAULT, &second) != first_vshard { + return (first.to_owned(), second); + } + } + panic!( + "no collection name under prefix {second_prefix} hashes away from {first} \ + in {MAX_TRIES} tries" + ); +} + +/// A graph endpoint key `{prefix}_{i}` whose `from_key` vShard differs from +/// `collection`'s own vShard, so an implicit edge task homed on the key is +/// dispatched to a different vShard than the document write. +pub fn key_on_other_vshard(collection: &str, prefix: &str) -> String { + let coll_vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, collection); + for i in 0..MAX_TRIES { + let key = format!("{prefix}_{i}"); + if VShardId::from_key(key.as_bytes()) != coll_vshard { + return key; + } + } + panic!( + "no key under prefix {prefix} hashes away from collection {collection} \ + in {MAX_TRIES} tries" + ); +} diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/ddl_objects.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/ddl_objects.rs index 520dfcc05..1b0ab9c58 100644 --- a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/ddl_objects.rs +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/ddl_objects.rs @@ -85,15 +85,17 @@ async fn sequence_create_visible_on_every_node() { .await .expect("alter sequence restart"); + // A restart marks the sequence not yet called, so its stored counter + // sits one increment below 500; the next `nextval` still returns 500. wait_for( - "all 3 nodes see sequence counter == 500", + "all 3 nodes: next order_id nextval == 500", Duration::from_secs(10), Duration::from_millis(50), || { cluster .nodes .iter() - .all(|n| n.sequence_current_value(1, "order_id") == Some(500)) + .all(|n| n.sequence_next_value(1, "order_id") == Some(500)) }, ) .await; diff --git a/nodedb-cluster/src/distributed_array/coordinator/read.rs b/nodedb-cluster/src/distributed_array/coordinator/read.rs index a392f7b90..fa9ba1597 100644 --- a/nodedb-cluster/src/distributed_array/coordinator/read.rs +++ b/nodedb-cluster/src/distributed_array/coordinator/read.rs @@ -428,6 +428,7 @@ mod tests { shard_hilbert_range: None, system_time: nodedb_types::SystemTimeScope::Current, valid_at_ms: None, + txn_id: None, }; // 3 shards × 2 rows each = 6 merged rows. @@ -458,6 +459,7 @@ mod tests { shard_hilbert_range: None, system_time: nodedb_types::SystemTimeScope::Current, valid_at_ms: None, + txn_id: None, }; let result = coord @@ -478,6 +480,7 @@ mod tests { shard_hilbert_range: None, system_as_of: None, valid_at_ms: None, + txn_id: None, } } @@ -635,6 +638,7 @@ mod tests { shard_hilbert_range: None, system_time: nodedb_types::SystemTimeScope::Current, valid_at_ms: None, + txn_id: None, }; // coordinator_limit = 0 → no cutoff → 20 rows. diff --git a/nodedb-cluster/src/distributed_array/coordinator/write.rs b/nodedb-cluster/src/distributed_array/coordinator/write.rs index b3d7f4b08..80c61d54f 100644 --- a/nodedb-cluster/src/distributed_array/coordinator/write.rs +++ b/nodedb-cluster/src/distributed_array/coordinator/write.rs @@ -193,6 +193,11 @@ mod tests { let resp = ArrayShardPutResp { shard_id: req.vshard_id, applied_lsn: shard_req.wal_lsn, + // The shard's cell count, as the Data Plane's `{"inserted": n}` + // would report for a batch with no conflicts. + affected: zerompk::from_msgpack::>>(&shard_req.cells_msgpack) + .unwrap() + .len() as u64, }; let payload = zerompk::to_msgpack_vec(&resp).unwrap(); Ok(VShardEnvelope::new( @@ -228,6 +233,11 @@ mod tests { let resp = ArrayShardDeleteResp { shard_id: req.vshard_id, applied_lsn: shard_req.wal_lsn, + // The shard's coordinate count, as the Data Plane's + // `{"deleted": n}` would report when every cell exists. + affected: zerompk::from_msgpack::>>(&shard_req.coords_msgpack) + .unwrap() + .len() as u64, }; let payload = zerompk::to_msgpack_vec(&resp).unwrap(); Ok(VShardEnvelope::new( @@ -285,6 +295,27 @@ mod tests { } } + #[tokio::test] + async fn coord_put_reports_real_affected_per_shard_for_summing() { + // Two shards with one and two cells. The coordinator hands back each + // shard's own `affected` (what the Data Plane reported), so the caller + // sums shard reports rather than counting the cells it sent. + let p0 = 0x0000_0000_0000_0000u64; + let p1 = 0x0040_0000_0000_0000u64; + let cells = vec![(p0, vec![0x01u8]), (p1, vec![0x02u8]), (p1, vec![0x03u8])]; + + let dispatch: Arc = Arc::new(PutEchoDispatch); + let mut resps = coord_put(&write_params(), vec![], 10, 7, &cells, &dispatch, &cb()) + .await + .expect("coord_put should succeed"); + resps.sort_by_key(|r| r.affected); + + let per_shard: Vec = resps.iter().map(|r| r.affected).collect(); + assert_eq!(per_shard, vec![1, 2], "each shard reports its own count"); + let total_affected: u64 = per_shard.iter().sum(); + assert_eq!(total_affected, 3); + } + #[tokio::test] async fn coord_put_aggregates_partial_failures() { // A failing dispatch must surface as an error, not silent partial success. diff --git a/nodedb-cluster/src/distributed_array/handler.rs b/nodedb-cluster/src/distributed_array/handler.rs index 50557c512..2f8ae0acc 100644 --- a/nodedb-cluster/src/distributed_array/handler.rs +++ b/nodedb-cluster/src/distributed_array/handler.rs @@ -119,10 +119,11 @@ async fn handle_put( // wrong shard. validate_put_routing(&req, local_vshard_id)?; - let applied_lsn = executor.exec_put(local_vshard_id, &req).await?; + let outcome = executor.exec_put(local_vshard_id, &req).await?; let resp = ArrayShardPutResp { shard_id: local_vshard_id, - applied_lsn, + applied_lsn: outcome.applied_lsn, + affected: outcome.affected, }; serialise(resp) } @@ -139,10 +140,11 @@ async fn handle_delete( validate_delete_routing(&req, local_vshard_id)?; - let applied_lsn = executor.exec_delete(local_vshard_id, &req).await?; + let outcome = executor.exec_delete(local_vshard_id, &req).await?; let resp = ArrayShardDeleteResp { shard_id: local_vshard_id, - applied_lsn, + applied_lsn: outcome.applied_lsn, + affected: outcome.affected, }; serialise(resp) } @@ -248,7 +250,9 @@ mod tests { use crate::distributed_array::wire::{ArrayShardAggReq, ArrayShardDeleteReq, ArrayShardPutReq}; use crate::error::Result; - use super::super::local_executor::{ArrayAggExec, ArrayLocalExecutor, ArraySliceExec}; + use super::super::local_executor::{ + ArrayAggExec, ArrayLocalExecutor, ArrayShardWriteOutcome, ArraySliceExec, + }; use super::super::opcodes::{ ARRAY_SHARD_AGG_REQ, ARRAY_SHARD_DELETE_REQ, ARRAY_SHARD_PUT_REQ, ARRAY_SHARD_SLICE_REQ, ARRAY_SHARD_SURROGATE_BITMAP_REQ, @@ -267,6 +271,7 @@ mod tests { bitmap: Vec, partials: Vec, truncated_before_horizon: bool, + affected: u64, } #[async_trait] @@ -302,16 +307,26 @@ mod tests { }) } - async fn exec_put(&self, _local_vshard_id: u32, req: &ArrayShardPutReq) -> Result { - Ok(req.wal_lsn) + async fn exec_put( + &self, + _local_vshard_id: u32, + req: &ArrayShardPutReq, + ) -> Result { + Ok(ArrayShardWriteOutcome { + applied_lsn: req.wal_lsn, + affected: self.affected, + }) } async fn exec_delete( &self, _local_vshard_id: u32, req: &ArrayShardDeleteReq, - ) -> Result { - Ok(req.wal_lsn) + ) -> Result { + Ok(ArrayShardWriteOutcome { + applied_lsn: req.wal_lsn, + affected: self.affected, + }) } } @@ -328,6 +343,7 @@ mod tests { shard_hilbert_range: None, system_time: nodedb_types::SystemTimeScope::Current, valid_at_ms: None, + txn_id: None, }; zerompk::to_msgpack_vec(&req).unwrap() } @@ -347,6 +363,7 @@ mod tests { shard_hilbert_range: None, system_as_of: None, valid_at_ms: None, + txn_id: None, }; zerompk::to_msgpack_vec(&req).unwrap() } @@ -367,6 +384,7 @@ mod tests { bitmap: vec![], partials: vec![], truncated_before_horizon: false, + affected: 0, }); let payload = make_slice_req_bytes(); let resp_bytes = handle_array_shard_rpc(ARRAY_SHARD_SLICE_REQ, 0, &payload, &executor) @@ -390,6 +408,7 @@ mod tests { bitmap: vec![], partials: vec![], truncated_before_horizon: true, + affected: 0, }); let payload = make_slice_req_bytes(); let resp_bytes = handle_array_shard_rpc(ARRAY_SHARD_SLICE_REQ, 0, &payload, &executor) @@ -410,6 +429,7 @@ mod tests { bitmap: vec![], partials: vec![], truncated_before_horizon: true, + affected: 0, }); let payload = make_agg_req_bytes(); let resp_bytes = handle_array_shard_rpc(ARRAY_SHARD_AGG_REQ, 0, &payload, &executor) @@ -431,6 +451,7 @@ mod tests { bitmap: bitmap.clone(), partials: vec![], truncated_before_horizon: false, + affected: 0, }); let payload = make_bitmap_req_bytes(); let resp_bytes = @@ -452,6 +473,7 @@ mod tests { bitmap: vec![], partials: vec![partial.clone()], truncated_before_horizon: false, + affected: 0, }); let payload = make_agg_req_bytes(); let resp_bytes = handle_array_shard_rpc(ARRAY_SHARD_AGG_REQ, 3, &payload, &executor) @@ -508,6 +530,7 @@ mod tests { bitmap: vec![], partials: vec![], truncated_before_horizon: false, + affected: 3, }); // prefix_bits=0 disables routing validation. let payload = make_put_req_bytes(0, 0); @@ -519,15 +542,22 @@ mod tests { zerompk::from_msgpack(&resp_bytes).expect("response should deserialise"); assert_eq!(resp.shard_id, 0); assert_eq!(resp.applied_lsn, 77); + assert_eq!( + resp.affected, 3, + "handler must forward the executor's real affected count, not cells.len()" + ); } #[tokio::test] async fn handle_delete_delegates_to_executor_and_echoes_lsn() { + // The executor reports 0 affected even though the request named a + // coordinate: DELETE of an absent coordinate must answer DELETE 0. let executor: Arc = Arc::new(StubExecutor { rows: vec![], bitmap: vec![], partials: vec![], truncated_before_horizon: false, + affected: 0, }); let payload = make_delete_req_bytes(); let resp_bytes = handle_array_shard_rpc(ARRAY_SHARD_DELETE_REQ, 2, &payload, &executor) @@ -538,6 +568,10 @@ mod tests { zerompk::from_msgpack(&resp_bytes).expect("response should deserialise"); assert_eq!(resp.shard_id, 2); assert_eq!(resp.applied_lsn, 88); + assert_eq!( + resp.affected, 0, + "deleting an absent coordinate must report 0 affected, not coords.len()" + ); } #[tokio::test] @@ -547,6 +581,7 @@ mod tests { bitmap: vec![], partials: vec![], truncated_before_horizon: false, + affected: 0, }); // prefix_bits=10, stride=1 → bucket = top 10 bits of hilbert_prefix. // hilbert_prefix = 0 → bucket 0 → expected vshard 0. @@ -588,6 +623,7 @@ mod tests { shard_hilbert_range: None, system_time: nodedb_types::SystemTimeScope::Current, valid_at_ms: None, + txn_id: None, }; let err = super::validate_slice_routing(&req, 5) .expect_err("disjoint Hilbert range should reject"); @@ -614,6 +650,7 @@ mod tests { shard_hilbert_range: None, system_time: nodedb_types::SystemTimeScope::Current, valid_at_ms: None, + txn_id: None, }; super::validate_slice_routing(&req, 1).expect("overlapping range should accept"); } @@ -625,6 +662,7 @@ mod tests { bitmap: vec![], partials: vec![], truncated_before_horizon: false, + affected: 0, }); let err = handle_array_shard_rpc(0xFF, 0, &[], &executor) .await diff --git a/nodedb-cluster/src/distributed_array/local_executor.rs b/nodedb-cluster/src/distributed_array/local_executor.rs index 922547f09..13dcf319b 100644 --- a/nodedb-cluster/src/distributed_array/local_executor.rs +++ b/nodedb-cluster/src/distributed_array/local_executor.rs @@ -38,6 +38,17 @@ pub struct ArrayAggExec { pub truncated_before_horizon: bool, } +/// Result of a local shard write (put or delete): the LSN the coordinator +/// acks with, plus the real cell count the Data Plane handler reported. +/// +/// `affected` is read from the handler's `{"inserted": n}` / `{"deleted": n}` +/// response payload — never the number of cells/coords named in the request, +/// which for delete over-counts absent coordinates. +pub struct ArrayShardWriteOutcome { + pub applied_lsn: u64, + pub affected: u64, +} + /// Execute array operations against the local Data Plane. /// /// `local_vshard_id` is the destination vShard from the validated RPC envelope. @@ -61,6 +72,9 @@ pub trait ArrayLocalExecutor: Send + Sync + 'static { /// set only tiles whose prefix falls in this range are returned, preventing /// duplicate rows in single-node harnesses where all vShards share one /// Data Plane. `None` = no Hilbert filter. + /// `txn_id` — the reading transaction's id; the executor stamps it on the + /// Data Plane request so the shard folds that transaction's staged + /// cells into the result. `None` = autocommit read. /// /// Returns the per-row bytes (one element per matching row, each the /// native-msgpack encoding of that row) plus the `truncated_before_horizon` @@ -88,7 +102,9 @@ pub trait ArrayLocalExecutor: Send + Sync + 'static { /// The Data Plane computes the aggregate with `return_partial = true`, so it /// returns partial states (plus the `truncated_before_horizon` signal) /// rather than finalized scalars. The coordinator merges partials from all - /// shards before finalizing. + /// shards before finalizing. `req.txn_id` is stamped on the Data Plane + /// request so the partial folds in that transaction's staged cells on + /// this shard. async fn exec_agg(&self, local_vshard_id: u32, req: &ArrayShardAggReq) -> Result; /// Apply a cell-batch write to the local array engine. @@ -98,7 +114,14 @@ pub trait ArrayLocalExecutor: Send + Sync + 'static { /// the same Hilbert-prefix tile. The shard handler has already validated /// that this shard owns the tile; the executor dispatches directly to the /// Data Plane without further routing checks. - async fn exec_put(&self, local_vshard_id: u32, req: &ArrayShardPutReq) -> Result; + /// + /// Returns the applied LSN plus the real cell count the Data Plane + /// handler reported, read from its `{"inserted": n}` response. + async fn exec_put( + &self, + local_vshard_id: u32, + req: &ArrayShardPutReq, + ) -> Result; /// Delete cells by exact coordinates from the local array engine. /// @@ -106,6 +129,12 @@ pub trait ArrayLocalExecutor: Send + Sync + 'static { /// local executor can apply the original delete payload on the validated /// `local_vshard_id`, mirroring [`Self::exec_put`]. /// - /// Returns the `applied_lsn` (equal to `req.wal_lsn` on success). - async fn exec_delete(&self, local_vshard_id: u32, req: &ArrayShardDeleteReq) -> Result; + /// Returns the applied LSN (equal to `req.wal_lsn` on success) plus the + /// number of coords that existed and were removed, read from the Data + /// Plane handler's `{"deleted": n}` response. + async fn exec_delete( + &self, + local_vshard_id: u32, + req: &ArrayShardDeleteReq, + ) -> Result; } diff --git a/nodedb-cluster/src/distributed_array/mod.rs b/nodedb-cluster/src/distributed_array/mod.rs index c609b7d07..acf71a1cd 100644 --- a/nodedb-cluster/src/distributed_array/mod.rs +++ b/nodedb-cluster/src/distributed_array/mod.rs @@ -16,7 +16,9 @@ pub use coordinator::{ coord_put, coord_put_partitioned, }; pub use handler::handle_array_shard_rpc; -pub use local_executor::{ArrayAggExec, ArrayLocalExecutor, ArraySliceExec}; +pub use local_executor::{ + ArrayAggExec, ArrayLocalExecutor, ArrayShardWriteOutcome, ArraySliceExec, +}; pub use merge::{ ArrayAggPartial, any_truncated_before_horizon_agg, any_truncated_before_horizon_slice, merge_slice_rows, reduce_agg_partials, diff --git a/nodedb-cluster/src/distributed_array/wire.rs b/nodedb-cluster/src/distributed_array/wire.rs index 68fbebdb6..0a3611830 100644 --- a/nodedb-cluster/src/distributed_array/wire.rs +++ b/nodedb-cluster/src/distributed_array/wire.rs @@ -51,6 +51,14 @@ pub struct ArrayShardSliceReq { /// Bitemporal valid-time point forwarded from `ArrayOp::Slice::valid_at_ms`. /// `None` = no valid-time filter. pub valid_at_ms: Option, + /// The reading transaction's id, for read-your-own-writes against the + /// shard's per-transaction staging overlay. `None` = autocommit read. + /// + /// The id is minted on the coordinator and has meaning on a shard only + /// where that transaction staged cells. Staging resolves each shard's + /// own leader and stages there, and this read resolves the same leader, + /// so the overlay keyed by this id is on the node that serves the read. + pub txn_id: Option, } /// Gather response: shard returns matching rows as opaque msgpack row bytes. @@ -89,6 +97,16 @@ pub struct ArrayShardAggReq { /// Bitemporal valid-time point forwarded from `ArrayOp::Aggregate::valid_at_ms`. /// `None` = no valid-time filter. pub valid_at_ms: Option, + /// The reading transaction's id, for read-your-own-writes against the + /// shard's per-transaction staging overlay. `None` = autocommit read. + /// + /// The id is minted on the coordinator and has meaning on a shard only + /// where that transaction staged cells. Staging resolves each shard's + /// own leader and stages there, and this read resolves the same leader, + /// so the overlay keyed by this id is on the node that serves the read. + /// Each shard folds only its own overlay cells into its partial, so the + /// coordinator's merge counts every staged cell exactly once. + pub txn_id: Option, } /// Gather response: shard returns partial aggregate(s) for merge. @@ -125,6 +143,11 @@ pub struct ArrayShardPutReq { pub struct ArrayShardPutResp { pub shard_id: u32, pub applied_lsn: u64, + /// Cells this shard actually wrote, read from the Data Plane handler's + /// `{"inserted": n}` count. The coordinator sums this across shards for + /// the client-facing `INSERT n` count — never `cells.len()`, which counts + /// coordinates named, not cells written. + pub affected: u64, } /// Scatter request: coordinator asks a shard to delete cells by exact coords. @@ -148,6 +171,12 @@ pub struct ArrayShardDeleteReq { pub struct ArrayShardDeleteResp { pub shard_id: u32, pub applied_lsn: u64, + /// Cells this shard actually removed, read from the Data Plane handler's + /// `{"deleted": n}` count (the number of coords that existed, not the + /// number named). A delete of an absent coordinate contributes 0. The + /// coordinator sums this across shards for the client-facing `DELETE n` + /// count — never `coords.len()`. + pub affected: u64, } /// Scatter request: coordinator asks a shard to run a surrogate bitmap scan. diff --git a/nodedb-columnar/src/lib.rs b/nodedb-columnar/src/lib.rs index 5648d9f2d..07e205e3b 100644 --- a/nodedb-columnar/src/lib.rs +++ b/nodedb-columnar/src/lib.rs @@ -46,7 +46,7 @@ pub use format::{ }; pub use materialize_rows::materialize_segment_live_rows; pub use memtable::{ColumnarMemtable, IngestValue, MemtableRowIter}; -pub use mutation::{ColumnDataSnapshot, ColumnarEngineSnapshot, MutationEngine}; +pub use mutation::{ColumnDataSnapshot, ColumnarEngineSnapshot, MutationEngine, TruncatedRows}; pub use pk_index::PkIndex; pub use predicate::{ BLOOM_BITS_DEFAULT, BLOOM_BYTES, BLOOM_K_DEFAULT, PredicateOp, PredicateValue, ScanPredicate, diff --git a/nodedb-columnar/src/memtable/core.rs b/nodedb-columnar/src/memtable/core.rs index 1526dab95..3ff243db0 100644 --- a/nodedb-columnar/src/memtable/core.rs +++ b/nodedb-columnar/src/memtable/core.rs @@ -51,6 +51,11 @@ impl ColumnarMemtable { self.row_count >= self.flush_threshold } + /// Row count at which [`Self::should_flush`] turns true. + pub fn flush_threshold(&self) -> usize { + self.flush_threshold + } + /// Whether the memtable is empty. pub fn is_empty(&self) -> bool { self.row_count == 0 diff --git a/nodedb-columnar/src/mutation/flush.rs b/nodedb-columnar/src/mutation/flush.rs index 14e799d7d..3e44c81a3 100644 --- a/nodedb-columnar/src/mutation/flush.rs +++ b/nodedb-columnar/src/mutation/flush.rs @@ -11,7 +11,13 @@ use super::engine::{MutationEngine, MutationResult}; impl MutationEngine { /// Notify the engine that the memtable was flushed to a new segment. /// - /// Updates the PK index to remap memtable entries to the new segment. + /// Remaps the PK index entries of the memtable's virtual segment to + /// `new_segment_id` and moves the memtable's delete bitmap with them, so + /// a row tombstoned while in the memtable stays tombstoned in the + /// segment. The virtual segment id itself never changes: it names the + /// memtable, not a flushed segment, so no flushed segment can ever share + /// it and a later remap can never sweep flushed rows along. + /// /// Returns the WAL record for the flush event, or `SegmentIdExhausted` /// if the u64 segment ID counter has wrapped past its maximum. pub fn on_memtable_flushed( @@ -28,6 +34,9 @@ impl MutationEngine { row_index: old_row, }) }); + if let Some(bitmap) = self.delete_bitmaps.remove(&self.memtable_segment_id) { + self.delete_bitmaps.insert(new_segment_id, bitmap); + } // Advance the segment ID counter with overflow protection. let next = self @@ -36,7 +45,6 @@ impl MutationEngine { .ok_or(ColumnarError::SegmentIdExhausted)?; // Reset memtable tracking. - self.memtable_segment_id = self.next_segment_id; self.next_segment_id = next; self.memtable_row_counter = 0; self.memtable_surrogates.clear(); @@ -52,3 +60,67 @@ impl MutationEngine { }) } } + +#[cfg(test)] +mod tests { + use nodedb_types::columnar::{ColumnDef, ColumnType, ColumnarSchema}; + use nodedb_types::value::Value; + + use super::*; + use crate::pk_index::encode_pk; + + fn engine() -> MutationEngine { + let schema = ColumnarSchema::new(vec![ + ColumnDef::required("id", ColumnType::Int64).with_primary_key(), + ]) + .expect("valid"); + MutationEngine::new("t".into(), schema) + } + + /// Two flushes must land in two distinct segments: the memtable keeps its + /// virtual id, so the second remap touches only the rows that were still + /// in the memtable. + #[test] + fn second_flush_leaves_first_segment_rows_in_place() { + let mut e = engine(); + e.insert(&[Value::Integer(1)]).expect("insert"); + let first = e.next_segment_id(); + e.on_memtable_flushed(first).expect("flush 1"); + e.insert(&[Value::Integer(2)]).expect("insert"); + let second = e.next_segment_id(); + e.on_memtable_flushed(second).expect("flush 2"); + + assert_ne!(first, second); + assert_eq!(e.memtable_segment_id(), 0); + assert_eq!( + e.pk_index() + .get(&encode_pk(&Value::Integer(1))) + .map(|l| l.segment_id), + Some(first) + ); + assert_eq!( + e.pk_index() + .get(&encode_pk(&Value::Integer(2))) + .map(|l| l.segment_id), + Some(second) + ); + } + + /// A row tombstoned in the memtable stays tombstoned once flushed. + #[test] + fn memtable_tombstones_follow_the_flushed_segment() { + let mut e = engine(); + e.insert(&[Value::Integer(1)]).expect("insert"); + e.insert(&[Value::Integer(2)]).expect("insert"); + e.delete(&Value::Integer(1)).expect("delete"); + let seg = e.next_segment_id(); + e.on_memtable_flushed(seg).expect("flush"); + + assert!(e.delete_bitmap(0).is_none(), "memtable bitmap moved out"); + assert!( + e.delete_bitmap(seg).is_some_and(|bm| bm.is_deleted(0)), + "row 0 of the flushed segment is the tombstoned row" + ); + assert!(!e.delete_bitmap(seg).is_some_and(|bm| bm.is_deleted(1))); + } +} diff --git a/nodedb-columnar/src/mutation/mod.rs b/nodedb-columnar/src/mutation/mod.rs index cbec28675..9ab451f01 100644 --- a/nodedb-columnar/src/mutation/mod.rs +++ b/nodedb-columnar/src/mutation/mod.rs @@ -10,7 +10,9 @@ pub mod engine; pub mod flush; pub mod snapshot; +pub mod truncate; pub mod write; pub use engine::{MutationEngine, MutationResult}; pub use snapshot::{ColumnDataSnapshot, ColumnarEngineSnapshot}; +pub use truncate::TruncatedRows; diff --git a/nodedb-columnar/src/mutation/truncate.rs b/nodedb-columnar/src/mutation/truncate.rs new file mode 100644 index 000000000..30584b981 --- /dev/null +++ b/nodedb-columnar/src/mutation/truncate.rs @@ -0,0 +1,159 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Whole-collection `TRUNCATE` on a `MutationEngine`, reversible for a +//! transaction batch that aborts after it. + +use std::collections::HashMap; + +use nodedb_types::surrogate::Surrogate; + +use crate::delete_bitmap::DeleteBitmap; +use crate::memtable::ColumnarMemtable; +use crate::pk_index::PkIndex; + +use super::engine::MutationEngine; + +/// Every row-bearing part of a `MutationEngine`, taken out by +/// [`MutationEngine::truncate`] and put back by +/// [`MutationEngine::restore_truncated`]. Holds every bitemporal version and +/// every tombstone exactly as they were, so a restore is exact. +pub struct TruncatedRows { + memtable: ColumnarMemtable, + pk_index: PkIndex, + delete_bitmaps: HashMap, + memtable_row_counter: u32, + memtable_surrogates: Vec>, +} + +impl MutationEngine { + /// Number of live rows: one per bound primary key. Historical bitemporal + /// versions and tombstoned rows are not counted. + pub fn live_row_count(&self) -> usize { + self.pk_index.len() + } + + /// Remove every row, every bitemporal version, and every tombstone, and + /// hand the pre-image back. Segment ids stay monotonic: the caller drops + /// its flushed segment bytes for the same collection, and the next flush + /// takes the next unused id rather than reusing one a stale reader could + /// still name. + pub fn truncate(&mut self) -> TruncatedRows { + let threshold = self.memtable.flush_threshold(); + let memtable = std::mem::replace( + &mut self.memtable, + ColumnarMemtable::with_threshold(&self.schema, threshold), + ); + TruncatedRows { + memtable, + pk_index: std::mem::take(&mut self.pk_index), + delete_bitmaps: std::mem::take(&mut self.delete_bitmaps), + memtable_row_counter: std::mem::replace(&mut self.memtable_row_counter, 0), + memtable_surrogates: std::mem::take(&mut self.memtable_surrogates), + } + } + + /// Put back what [`Self::truncate`] took out. Used exclusively by the + /// transaction undo log; never called on the normal write path. + pub fn restore_truncated(&mut self, rows: TruncatedRows) { + let TruncatedRows { + memtable, + pk_index, + delete_bitmaps, + memtable_row_counter, + memtable_surrogates, + } = rows; + self.memtable = memtable; + self.pk_index = pk_index; + self.delete_bitmaps = delete_bitmaps; + self.memtable_row_counter = memtable_row_counter; + self.memtable_surrogates = memtable_surrogates; + } +} + +#[cfg(test)] +mod tests { + use nodedb_types::columnar::{ColumnDef, ColumnType, ColumnarSchema}; + use nodedb_types::value::Value; + + use super::*; + + fn schema() -> ColumnarSchema { + ColumnarSchema::new(vec![ + ColumnDef::required("id", ColumnType::Int64).with_primary_key(), + ColumnDef::required("name", ColumnType::String), + ]) + .expect("valid") + } + + fn seeded() -> MutationEngine { + let mut engine = MutationEngine::with_flush_threshold("t".into(), schema(), 7); + for i in 0..4 { + engine + .insert_with_surrogate( + &[Value::Integer(i), Value::String(format!("r{i}"))], + Surrogate(100 + i as u32), + ) + .expect("insert"); + } + engine.delete(&Value::Integer(1)).expect("delete"); + engine + } + + #[test] + fn truncate_empties_every_row_bearing_part_and_keeps_the_threshold() { + let mut engine = seeded(); + assert_eq!(engine.live_row_count(), 3); + + let _pre = engine.truncate(); + + assert_eq!(engine.live_row_count(), 0); + assert_eq!(engine.memtable().row_count(), 0); + assert!(engine.delete_bitmaps().is_empty()); + assert!(engine.memtable_surrogates().is_empty()); + assert_eq!(engine.scan_memtable_rows().count(), 0); + assert_eq!(engine.memtable().flush_threshold(), 7); + assert_eq!(engine.next_segment_id(), 1); + + engine + .insert(&[Value::Integer(9), Value::String("z".into())]) + .expect("insert after truncate"); + assert_eq!(engine.live_row_count(), 1); + assert_eq!(engine.scan_memtable_rows().count(), 1); + } + + #[test] + fn restore_truncated_brings_back_rows_and_tombstones_exactly() { + let mut engine = seeded(); + let before: Vec> = engine.scan_memtable_rows().collect(); + + let pre = engine.truncate(); + engine + .insert(&[Value::Integer(9), Value::String("z".into())]) + .expect("insert after truncate"); + engine.restore_truncated(pre); + + let after: Vec> = engine.scan_memtable_rows().collect(); + assert_eq!( + after, before, + "restore must reproduce the pre-truncate rows" + ); + assert_eq!(engine.live_row_count(), 3); + assert!( + engine + .pk_index() + .get(&crate::pk_index::encode_pk(&Value::Integer(1))) + .is_none(), + "the tombstone deleted before the truncate stays deleted" + ); + assert_eq!(engine.memtable_surrogates().len(), 4); + } + + #[test] + fn truncate_keeps_segment_ids_monotonic_after_a_flush() { + let mut engine = seeded(); + engine.on_memtable_flushed(1).expect("flush"); + assert_eq!(engine.next_segment_id(), 2); + let _pre = engine.truncate(); + assert_eq!(engine.next_segment_id(), 2); + } +} diff --git a/nodedb-physical/src/physical_plan/cluster_array.rs b/nodedb-physical/src/physical_plan/cluster_array.rs index 6e9a1d604..add4e26d0 100644 --- a/nodedb-physical/src/physical_plan/cluster_array.rs +++ b/nodedb-physical/src/physical_plan/cluster_array.rs @@ -110,3 +110,17 @@ pub enum ClusterArrayOp { prefix_bits: u8, }, } + +impl ClusterArrayOp { + /// The array this op targets. Total over every variant, so read-set + /// tracking and the commit validator key on the same name a shard-local + /// `ArrayOp` reports. + pub fn array_id(&self) -> &ArrayId { + match self { + ClusterArrayOp::Slice { array_id, .. } + | ClusterArrayOp::Agg { array_id, .. } + | ClusterArrayOp::Put { array_id, .. } + | ClusterArrayOp::Delete { array_id, .. } => array_id, + } + } +} diff --git a/nodedb-physical/src/physical_plan/collection.rs b/nodedb-physical/src/physical_plan/collection.rs index 90001efca..5ab0855c4 100644 --- a/nodedb-physical/src/physical_plan/collection.rs +++ b/nodedb-physical/src/physical_plan/collection.rs @@ -29,6 +29,12 @@ impl PhysicalPlan { // A vector-primary row lives here only; `None` left it with no // collection to key a redaction policy on. | PhysicalPlan::Vector(VectorOp::DirectUpsert { collection, .. }) + | PhysicalPlan::Vector(VectorOp::DirectInsert { collection, .. }) + | PhysicalPlan::Vector(VectorOp::DirectInsertIfAbsent { collection, .. }) + | PhysicalPlan::Vector(VectorOp::DirectDelete { collection, .. }) + | PhysicalPlan::Vector(VectorOp::DirectTruncate { collection, .. }) + | PhysicalPlan::Vector(VectorOp::DirectUpdate { collection, .. }) + | PhysicalPlan::Vector(VectorOp::ResolvedDirectWrite { collection, .. }) | PhysicalPlan::Vector(VectorOp::Delete { collection, .. }) | PhysicalPlan::Document(DocumentOp::BatchInsert { collection, .. }) | PhysicalPlan::Document(DocumentOp::PointPut { collection, .. }) @@ -72,8 +78,10 @@ impl PhysicalPlan { | PhysicalPlan::Columnar(ColumnarOp::Delete { collection, .. }) | PhysicalPlan::Columnar(ColumnarOp::ResolvedUpdate { collection, .. }) | PhysicalPlan::Columnar(ColumnarOp::ResolvedDelete { collection, .. }) + | PhysicalPlan::Columnar(ColumnarOp::Truncate { collection, .. }) | PhysicalPlan::Timeseries(TimeseriesOp::Scan { collection, .. }) | PhysicalPlan::Timeseries(TimeseriesOp::Ingest { collection, .. }) + | PhysicalPlan::Timeseries(TimeseriesOp::Truncate { collection, .. }) | PhysicalPlan::Spatial(SpatialOp::Scan { collection, .. }) | PhysicalPlan::Document(DocumentOp::Register { collection, .. }) | PhysicalPlan::Document(DocumentOp::IndexLookup { collection, .. }) @@ -85,7 +93,8 @@ impl PhysicalPlan { // collection, which is what the propose step routes on. PhysicalPlan::Timeseries(TimeseriesOp::ResolveIngest(inner)) => match inner.as_ref() { TimeseriesOp::Scan { collection, .. } - | TimeseriesOp::Ingest { collection, .. } => Some(collection.as_str()), + | TimeseriesOp::Ingest { collection, .. } + | TimeseriesOp::Truncate { collection, .. } => Some(collection.as_str()), TimeseriesOp::ResolveIngest(_) => None, }, // Same shape on the graph side, and `EdgeDelete` itself reports @@ -109,6 +118,10 @@ impl PhysicalPlan { | PhysicalPlan::Graph(GraphOp::MatchVarLenResume { .. }) | PhysicalPlan::Graph(GraphOp::BspSuperstep(_)) | PhysicalPlan::Graph(GraphOp::WccSuperstep(_)) => None, + // Read-only resolve wrapper: it reports the wrapped write's collection. + PhysicalPlan::Vector(VectorOp::ResolveDirectWrite(inner)) => { + inner.direct_write_collection() + } // Exchange: recurse into the child plan to extract the collection. PhysicalPlan::Query(QueryOp::Exchange(op)) => op.child.collection(), // PostProcess: recurse into the materialized input plan. @@ -133,9 +146,12 @@ impl PhysicalPlan { | PhysicalPlan::Spatial(_) | PhysicalPlan::Query(_) | PhysicalPlan::Meta(_) - | PhysicalPlan::Array(_) - | PhysicalPlan::ClusterArray(_) | PhysicalPlan::ClusterEvent(_) => None, + // An array is a collection for read-set tracking and the commit + // validator's own-write exclusion: a same-transaction slice after a + // staged put must key on the name the put's write floor records. + PhysicalPlan::Array(op) => Some(op.primary_array().name.as_str()), + PhysicalPlan::ClusterArray(op) => Some(op.array_id().name.as_str()), } } } diff --git a/nodedb-physical/src/physical_plan/columnar.rs b/nodedb-physical/src/physical_plan/columnar.rs index 4ac6ce611..2cce68d98 100644 --- a/nodedb-physical/src/physical_plan/columnar.rs +++ b/nodedb-physical/src/physical_plan/columnar.rs @@ -257,4 +257,16 @@ pub enum ColumnarOp { count: usize, system_as_of_ms: Option, }, + + /// `TRUNCATE` of a columnar or spatial collection: every row, every + /// bitemporal version, every flushed segment, and every R-tree entry + /// the collection's geometry columns produced. Reports the number of + /// live rows that existed. + Truncate { + collection: QualifiedCollection, + /// `TRUNCATE ... RESTART IDENTITY`: the Control Plane resets the + /// collection's sequences after the Data Plane clears the rows. + #[serde(default)] + restart_identity: bool, + }, } diff --git a/nodedb-physical/src/physical_plan/crdt/collection.rs b/nodedb-physical/src/physical_plan/crdt/collection.rs index 1230620cf..fdf4165b4 100644 --- a/nodedb-physical/src/physical_plan/crdt/collection.rs +++ b/nodedb-physical/src/physical_plan/crdt/collection.rs @@ -179,6 +179,7 @@ mod tests { fields_json: "{}".to_string(), surrogate: Surrogate::ZERO, partial: false, + verb: crate::physical_plan::CrdtWriteVerb::Insert, returning: None, rls_filters: Vec::new(), }, diff --git a/nodedb-physical/src/physical_plan/crdt/mod.rs b/nodedb-physical/src/physical_plan/crdt/mod.rs index 4c61a12e6..770af2fd1 100644 --- a/nodedb-physical/src/physical_plan/crdt/mod.rs +++ b/nodedb-physical/src/physical_plan/crdt/mod.rs @@ -4,5 +4,7 @@ pub mod collection; pub mod op; +pub mod write_verb; pub use op::CrdtOp; +pub use write_verb::CrdtWriteVerb; diff --git a/nodedb-physical/src/physical_plan/crdt/op.rs b/nodedb-physical/src/physical_plan/crdt/op.rs index 80c5a597f..202c287d9 100644 --- a/nodedb-physical/src/physical_plan/crdt/op.rs +++ b/nodedb-physical/src/physical_plan/crdt/op.rs @@ -6,6 +6,8 @@ use nodedb_types::{QualifiedCollection, Surrogate}; use crate::physical_plan::document::ReturningSpec; +use super::write_verb::CrdtWriteVerb; + /// CRDT engine physical operations. #[derive( Debug, @@ -224,6 +226,8 @@ pub enum CrdtOp { fields_json: String, surrogate: Surrogate, partial: bool, + /// The SQL statement that produced this write; decides the command tag. + verb: CrdtWriteVerb, /// When `Some`, return the STORED post-image of the upserted row — /// projected per spec. Carried across replication so a replay /// re-executes this write for the originating request, not just for diff --git a/nodedb-physical/src/physical_plan/crdt/write_verb.rs b/nodedb-physical/src/physical_plan/crdt/write_verb.rs new file mode 100644 index 000000000..da885c067 --- /dev/null +++ b/nodedb-physical/src/physical_plan/crdt/write_verb.rs @@ -0,0 +1,40 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The SQL verb behind a `CrdtOp::DocUpsert`. +//! +//! `INSERT`, `UPSERT` and `UPDATE` against a CRDT collection all lower to the +//! same Loro map write. The verb rides on the op so the response layer can +//! render the command tag the client's statement expects. + +/// The SQL statement that produced a `CrdtOp::DocUpsert`. +#[derive( + Clone, + Copy, + Debug, + PartialEq, + Eq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum CrdtWriteVerb { + /// Plain `INSERT`: full-row replace, tagged `INSERT 0 n`. + Insert, + /// `UPSERT` / `INSERT ... ON CONFLICT DO UPDATE`: full-row replace, + /// tagged `UPSERT n`. + Upsert, + /// `UPDATE ... SET`: partial field write, tagged `UPDATE n`. + Update, +} + +impl CrdtWriteVerb { + /// The pgwire command-tag word for this verb. + pub fn command_tag(self) -> &'static str { + match self { + Self::Insert => "INSERT", + Self::Upsert => "UPSERT", + Self::Update => "UPDATE", + } + } +} diff --git a/nodedb-physical/src/physical_plan/document/op.rs b/nodedb-physical/src/physical_plan/document/op.rs index cf978f848..c59eaa0fe 100644 --- a/nodedb-physical/src/physical_plan/document/op.rs +++ b/nodedb-physical/src/physical_plan/document/op.rs @@ -506,6 +506,13 @@ pub enum DocumentOp { /// intercepts — never reaches the Data Plane. #[serde(default)] resolved_inserts: Option>, + /// The same NOT-MATCHED surrogates keyed by the target document id + /// each row stores under, index-aligned with `resolved_inserts`, so + /// every applying node installs the `(collection, pk) → surrogate` + /// binding a later point read by primary key resolves through. Empty + /// when `resolved_inserts` is `None`. The handler never reads it. + #[serde(default)] + resolved_insert_identities: Vec<(String, u32)>, /// See `UpdateFromJoin::source_rows`. #[serde(default)] source_rows: Option)>>, diff --git a/nodedb-physical/src/physical_plan/kv/op.rs b/nodedb-physical/src/physical_plan/kv/op.rs index 822cf4b32..1ccd43fee 100644 --- a/nodedb-physical/src/physical_plan/kv/op.rs +++ b/nodedb-physical/src/physical_plan/kv/op.rs @@ -297,7 +297,13 @@ pub enum KvOp { }, /// Truncate: delete ALL entries in a KV collection. - Truncate { collection: QualifiedCollection }, + Truncate { + collection: QualifiedCollection, + /// `TRUNCATE ... RESTART IDENTITY`: the Control Plane resets the + /// collection's sequences after the Data Plane clears the entries. + #[serde(default)] + restart_identity: bool, + }, /// Atomic increment: init 0 if absent, `TypeMismatch` if not i64, /// `OverflowError` on wrap. `ttl_ms > 0` sets/resets TTL; `0` preserves it. diff --git a/nodedb-physical/src/physical_plan/meta.rs b/nodedb-physical/src/physical_plan/meta.rs index 297de18ab..265f23c4a 100644 --- a/nodedb-physical/src/physical_plan/meta.rs +++ b/nodedb-physical/src/physical_plan/meta.rs @@ -10,6 +10,13 @@ use nodedb_types::{QualifiedCollection, TenantId, Value}; pub use super::meta_calvin::PassiveReadKeyId; +/// Byte length of the [`MetaOp::MarkSavepoint`] response payload / +/// [`MetaOp::RollbackToSavepoint`] marker set: three little-endian `u64`s +/// (value/TTL overlay, GRAPH overlay, ARRAY overlay journal lengths). +/// Encoder and decoder both key off this constant so the layout can only +/// change in one place. +pub const SAVEPOINT_MARKER_BYTES: usize = 24; + /// Meta / maintenance physical operations. #[derive( Debug, @@ -461,27 +468,29 @@ pub enum MetaOp { /// Mark a savepoint in the per-transaction staging overlays. /// - /// A single savepoint spans BOTH the value/TTL overlay and the parallel - /// GRAPH overlay, which keep independent undo journals. The Data Plane - /// returns a 16-byte composite marker — two little-endian `u64`s: the - /// value overlay's journal length followed by the GRAPH overlay's — so the - /// Control Plane can record both as the savepoint's rollback markers. + /// A single savepoint spans the value/TTL overlay and the parallel GRAPH + /// and ARRAY overlays, which keep independent undo journals. The Data + /// Plane returns a 24-byte composite marker — three little-endian `u64`s: + /// the value overlay's journal length, then the GRAPH overlay's, then the + /// ARRAY overlay's — so the Control Plane can record all three as the + /// savepoint's rollback markers. /// In-memory only — savepoints append no WAL. Keyed by the request's /// `txn_id`. MarkSavepoint { txn_id: nodedb_types::id::TxnId }, /// Roll the per-transaction staging overlays back to a savepoint. /// - /// Replays BOTH overlays' undo journals from their ends down to their - /// respective markers in reverse — restoring each recorded prior slot (or - /// removing it when absent) in the value/TTL overlay to `value_marker` and - /// in the GRAPH overlay to `graph_marker` — then truncates each journal to - /// its marker. The transaction stays open. In-memory only. Keyed by - /// `txn_id`. + /// Replays every overlay's undo journal from its end down to its marker + /// in reverse — restoring each recorded prior slot (or removing it when + /// absent) in the value/TTL overlay to `value_marker`, in the GRAPH + /// overlay to `graph_marker`, and in the ARRAY overlay to `array_marker` + /// — then truncates each journal to its marker. The transaction stays + /// open. In-memory only. Keyed by `txn_id`. RollbackToSavepoint { txn_id: nodedb_types::id::TxnId, value_marker: u64, graph_marker: u64, + array_marker: u64, }, /// Record the per-key / per-collection write versions of a committed diff --git a/nodedb-physical/src/physical_plan/mod.rs b/nodedb-physical/src/physical_plan/mod.rs index 0df5f1b35..04e8a3ce3 100644 --- a/nodedb-physical/src/physical_plan/mod.rs +++ b/nodedb-physical/src/physical_plan/mod.rs @@ -28,6 +28,7 @@ pub mod spatial; pub mod streaming; pub mod text; pub mod timeseries; +pub mod truncate_target; pub mod vector; pub mod wire; @@ -35,7 +36,7 @@ pub use array::{ArrayBinaryOp, ArrayOp, ArrayReducer}; pub use cluster_array::ClusterArrayOp; pub use cluster_event::{ClusterEventOp, MAX_REMOTE_CDC_COMMITTED_OFFSETS}; pub use columnar::{ColumnarInsertIntent, ColumnarOp}; -pub use crdt::CrdtOp; +pub use crdt::{CrdtOp, CrdtWriteVerb}; pub use document::{ BalancedDef, DocumentOp, DocumentResolveOutcome, DocumentResolvedMutation, EnforcementOptions, GeneratedColumnSpec, MaterializedSumBinding, OllpPredictedEdge, PeriodLockConfig, @@ -48,7 +49,7 @@ pub use graph::{ BatchEdge, BspSuperstepPlan, BspSuperstepResult, GraphOp, WccSuperstepPlan, WccSuperstepResult, }; pub use kv::{KvOp, KvResolveOutcome, KvResolvedMutation}; -pub use meta::MetaOp; +pub use meta::{MetaOp, SAVEPOINT_MARKER_BYTES}; pub use plan::PhysicalPlan; pub use query::{AggregateSpec, GroupKeySpec, JoinProjection, QueryOp}; pub use routing::plan_contains_cluster_partitioned_leaf; @@ -57,5 +58,8 @@ pub use sort_key::SortKeySpec; pub use spatial::{SpatialOp, SpatialPredicate}; pub use text::TextOp; pub use timeseries::{TimeseriesOp, UNBOUNDED_TIME_RANGE}; -pub use vector::VectorOp; +pub use vector::{ + VectorDirectWriteIntent, VectorOp, VectorResolveOutcome, VectorResolvedMutation, + VectorWriteTargets, +}; pub use wire::{decode, encode}; diff --git a/nodedb-physical/src/physical_plan/rls_write_check_accessor.rs b/nodedb-physical/src/physical_plan/rls_write_check_accessor.rs index e7ee99654..729b85e59 100644 --- a/nodedb-physical/src/physical_plan/rls_write_check_accessor.rs +++ b/nodedb-physical/src/physical_plan/rls_write_check_accessor.rs @@ -15,6 +15,7 @@ use super::document::DocumentOp; use super::graph::GraphOp; use super::kv::KvOp; use super::timeseries::TimeseriesOp; +use super::vector::VectorOp; impl PhysicalPlan { /// The single RLS write check this plan carries, or `None` if it carries @@ -108,6 +109,18 @@ impl PhysicalPlan { }) | PhysicalPlan::Kv(KvOp::ResolvedWrite { rls_write_check, .. + }) + | PhysicalPlan::Vector(VectorOp::DirectUpsert { + rls_write_check, .. + }) + | PhysicalPlan::Vector(VectorOp::DirectDelete { + rls_write_check, .. + }) + | PhysicalPlan::Vector(VectorOp::DirectUpdate { + rls_write_check, .. + }) + | PhysicalPlan::Vector(VectorOp::ResolvedDirectWrite { + rls_write_check, .. }) => Some(rls_write_check), // Carries two checks, not one — see `rls_write_checks`. @@ -119,6 +132,9 @@ impl PhysicalPlan { // Same shape on the document side. PhysicalPlan::Document(DocumentOp::ResolveWrite(_)) => None, + // And on the vector-primary side. + PhysicalPlan::Vector(VectorOp::ResolveDirectWrite(_)) => None, + // Not write-class: no `rls_write_check` field at all. PhysicalPlan::Vector(_) | PhysicalPlan::Graph(_) diff --git a/nodedb-physical/src/physical_plan/timeseries.rs b/nodedb-physical/src/physical_plan/timeseries.rs index e7451d60c..70867a666 100644 --- a/nodedb-physical/src/physical_plan/timeseries.rs +++ b/nodedb-physical/src/physical_plan/timeseries.rs @@ -105,4 +105,15 @@ pub enum TimeseriesOp { /// into stamped ILP lines (memtable schema is Data-Plane-only, so /// normalization must happen here) and decides the policy without writing. ResolveIngest(Box), + + /// `TRUNCATE` of a timeseries collection: the memtable, every on-disk + /// partition, the series catalog, and the last-value cache. Reports the + /// number of rows that existed across memtable and partitions. + Truncate { + collection: QualifiedCollection, + /// `TRUNCATE ... RESTART IDENTITY`: the Control Plane resets the + /// collection's sequences after the Data Plane clears the rows. + #[serde(default)] + restart_identity: bool, + }, } diff --git a/nodedb-physical/src/physical_plan/truncate_target.rs b/nodedb-physical/src/physical_plan/truncate_target.rs new file mode 100644 index 000000000..f0d4a5367 --- /dev/null +++ b/nodedb-physical/src/physical_plan/truncate_target.rs @@ -0,0 +1,118 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Which collection a whole-collection `TRUNCATE` plan clears. +//! +//! Kept beside the plan enum rather than inside it so `plan.rs` stays the +//! single declaration of the wire shape and nothing else. + +use nodedb_types::QualifiedCollection; + +use super::{ColumnarOp, DocumentOp, KvOp, PhysicalPlan, TimeseriesOp, VectorOp}; + +impl PhysicalPlan { + /// The collection this plan truncates and its `RESTART IDENTITY` flag, or + /// `None` for every plan that is not a whole-collection truncate. The + /// match is exhaustive so a new truncate-shaped op forces a decision + /// here, where sequence restart is applied engine-neutrally. + pub fn truncate_target(&self) -> Option<(&QualifiedCollection, bool)> { + match self { + PhysicalPlan::Document(DocumentOp::Truncate { + collection, + restart_identity, + .. + }) + | PhysicalPlan::Kv(KvOp::Truncate { + collection, + restart_identity, + }) + | PhysicalPlan::Vector(VectorOp::DirectTruncate { + collection, + restart_identity, + .. + }) + | PhysicalPlan::Columnar(ColumnarOp::Truncate { + collection, + restart_identity, + }) + | PhysicalPlan::Timeseries(TimeseriesOp::Truncate { + collection, + restart_identity, + }) => Some((collection, *restart_identity)), + PhysicalPlan::Document(_) + | PhysicalPlan::Kv(_) + | PhysicalPlan::Vector(_) + | PhysicalPlan::Graph(_) + | PhysicalPlan::Text(_) + | PhysicalPlan::Columnar(_) + | PhysicalPlan::Timeseries(_) + | PhysicalPlan::Spatial(_) + | PhysicalPlan::Crdt(_) + | PhysicalPlan::Query(_) + | PhysicalPlan::Meta(_) + | PhysicalPlan::Array(_) + | PhysicalPlan::ClusterArray(_) + | PhysicalPlan::ClusterEvent(_) => None, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_types::DatabaseId; + + fn coll() -> QualifiedCollection { + QualifiedCollection::new(DatabaseId::DEFAULT, "t") + } + + #[test] + fn kv_truncate_reports_collection_and_flag() { + let plan = PhysicalPlan::Kv(KvOp::Truncate { + collection: coll(), + restart_identity: true, + }); + let (c, restart) = plan.truncate_target().expect("kv truncate target"); + assert_eq!(c.as_str(), "t"); + assert!(restart); + } + + #[test] + fn vector_direct_truncate_reports_collection_and_flag() { + let plan = PhysicalPlan::Vector(VectorOp::DirectTruncate { + collection: coll(), + field: "vec".into(), + restart_identity: false, + }); + let (c, restart) = plan.truncate_target().expect("vector truncate target"); + assert_eq!(c.as_str(), "t"); + assert!(!restart); + } + + #[test] + fn columnar_truncate_reports_collection_and_flag() { + let plan = PhysicalPlan::Columnar(ColumnarOp::Truncate { + collection: coll(), + restart_identity: true, + }); + let (c, restart) = plan.truncate_target().expect("columnar truncate target"); + assert_eq!(c.as_str(), "t"); + assert!(restart); + } + + #[test] + fn timeseries_truncate_reports_collection_and_flag() { + let plan = PhysicalPlan::Timeseries(TimeseriesOp::Truncate { + collection: coll(), + restart_identity: false, + }); + let (c, restart) = plan.truncate_target().expect("timeseries truncate target"); + assert_eq!(c.as_str(), "t"); + assert!(!restart); + } + + #[test] + fn non_truncate_plan_reports_none() { + let plan = PhysicalPlan::Meta(super::super::MetaOp::Checkpoint); + assert!(plan.truncate_target().is_none()); + } +} diff --git a/nodedb-physical/src/physical_plan/vector/collection.rs b/nodedb-physical/src/physical_plan/vector/collection.rs new file mode 100644 index 000000000..600238f46 --- /dev/null +++ b/nodedb-physical/src/physical_plan/vector/collection.rs @@ -0,0 +1,45 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Which user collection a vector-primary direct write targets. +//! +//! Kept beside the operation enum rather than inside it so `op.rs` stays the +//! single declaration of the wire shape and nothing else. + +use super::op::VectorOp; + +impl VectorOp { + /// The vector-primary collection a direct write targets, or `None` for + /// every op outside that family. `ResolveDirectWrite` reports the wrapped + /// write's collection so the resolve dispatch routes to the core that + /// owns exactly the rows the write reads. + pub fn direct_write_collection(&self) -> Option<&str> { + match self { + VectorOp::DirectUpsert { collection, .. } + | VectorOp::DirectInsert { collection, .. } + | VectorOp::DirectInsertIfAbsent { collection, .. } + | VectorOp::DirectDelete { collection, .. } + | VectorOp::DirectTruncate { collection, .. } + | VectorOp::DirectUpdate { collection, .. } + | VectorOp::ResolvedDirectWrite { collection, .. } => Some(collection.as_str()), + VectorOp::ResolveDirectWrite(inner) => inner.direct_write_collection(), + VectorOp::Search { .. } + | VectorOp::Insert { .. } + | VectorOp::BatchInsert { .. } + | VectorOp::MultiSearch { .. } + | VectorOp::Delete { .. } + | VectorOp::DeleteBySurrogate { .. } + | VectorOp::SetParams { .. } + | VectorOp::DropIndex { .. } + | VectorOp::QueryStats { .. } + | VectorOp::Seal { .. } + | VectorOp::CompactIndex { .. } + | VectorOp::Rebuild { .. } + | VectorOp::SparseInsert { .. } + | VectorOp::SparseSearch { .. } + | VectorOp::SparseDelete { .. } + | VectorOp::MultiVectorInsert { .. } + | VectorOp::MultiVectorDelete { .. } + | VectorOp::MultiVectorScoreSearch { .. } => None, + } + } +} diff --git a/nodedb-physical/src/physical_plan/vector/mod.rs b/nodedb-physical/src/physical_plan/vector/mod.rs new file mode 100644 index 000000000..fd851e2bc --- /dev/null +++ b/nodedb-physical/src/physical_plan/vector/mod.rs @@ -0,0 +1,12 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Vector engine operations dispatched to the Data Plane. + +pub mod collection; +pub mod op; +pub mod resolved_mutation; +pub mod write; + +pub use op::VectorOp; +pub use resolved_mutation::{VectorResolveOutcome, VectorResolvedMutation}; +pub use write::{VectorDirectWriteIntent, VectorWriteTargets}; diff --git a/nodedb-physical/src/physical_plan/vector.rs b/nodedb-physical/src/physical_plan/vector/op.rs similarity index 64% rename from nodedb-physical/src/physical_plan/vector.rs rename to nodedb-physical/src/physical_plan/vector/op.rs index 47e0211d8..b7d10a2e4 100644 --- a/nodedb-physical/src/physical_plan/vector.rs +++ b/nodedb-physical/src/physical_plan/vector/op.rs @@ -7,7 +7,10 @@ use nodedb_types::{ }; use crate::physical_plan::PhysicalPlan; -use crate::physical_plan::document::ReturningSpec; +use crate::physical_plan::document::{ReturningSpec, UpdateValue}; + +use super::resolved_mutation::VectorResolvedMutation; +use super::write::VectorWriteTargets; /// Vector engine physical operations. #[derive( @@ -276,29 +279,31 @@ pub enum VectorOp { mode: String, }, - /// Direct vector upsert for vector-primary collections. + /// Direct vector upsert for vector-primary collections: an existing + /// primary key is replaced whole, or patched by `on_conflict_updates`. /// - /// Bypasses MessagePack document encoding — the Data Plane inserts the - /// vector into HNSW directly and updates payload bitmap indexes from - /// `payload`. No full-document blob is written. + /// Bypasses MessagePack document encoding — the Data Plane binds the + /// vector to an HNSW node, updates the payload bitmap indexes from + /// `payload`, and stores `payload` as the row's sidecar. No + /// full-document blob is written. /// /// Ordering invariant (enforced by the handler): - /// 1. Validate dim. - /// 2. Decode `payload` bytes. - /// 3. Insert into HNSW (surrogate bound). - /// 4. Update payload bitmap indexes. - /// - /// If step 3 fails, step 4 is not reached — no partial state. - /// If step 4 fails (should not happen — pure in-memory), the handler - /// attempts to delete the just-inserted HNSW node and returns an error. + /// 1. Validate dim and storage dtype. + /// 2. Decode `payload` bytes and probe the surrogate. + /// 3. Remove the old node, bitmap entries, and sidecar when one exists. + /// 4. Insert into HNSW (surrogate bound) and update the bitmap indexes. + /// 5. Store the sidecar; a failed store rolls step 4 back. DirectUpsert { collection: QualifiedCollection, /// Vector column name. Used to compute the vector index key so the /// SELECT path (which keys by `(tid, collection, field)`) finds the /// same index this insert wrote into. field: String, - /// Global surrogate allocated by the Control Plane. + /// Global surrogate the Control Plane bound to `pk_bytes`. surrogate: Surrogate, + /// UTF-8 of the declared primary-key value. Followers bind the + /// leader-assigned surrogate to this exact key. + pk_bytes: Vec, /// FP32 vector values. vector: Vec, /// Pre-encoded MessagePack of every non-vector column the statement @@ -327,8 +332,7 @@ pub enum VectorOp { /// the sidecar is `zerompk` TAGGED bytes and echoing the request would /// report a shape no read of the collection ever produces. /// - /// This is the only insert op a vector-primary collection plans to — - /// the vector goes to HNSW, every other column to the sidecar, and + /// The vector goes to HNSW, every other column to the sidecar, and /// there is no companion document write to carry the clause instead. #[serde(default)] returning: Option, @@ -340,5 +344,143 @@ pub enum VectorOp { /// same principal. #[serde(default)] rls_filters: Vec, + /// `ON CONFLICT (pk) DO UPDATE SET` assignments applied to the stored + /// payload when the key exists. Empty means whole-row replace. + #[serde(default)] + on_conflict_updates: Vec<(String, UpdateValue)>, + /// Write policy for the merged post-image an `on_conflict_updates` + /// patch produces; the image exists only on the Data Plane. Decided + /// Control-Plane-side when the patch is empty. + rls_write_check: nodedb_types::RlsWriteCheck, + }, + + /// Vector-primary `INSERT`: an existing primary key raises + /// `unique_violation`. Same fields as [`VectorOp::DirectUpsert`] minus + /// the conflict patch. + DirectInsert { + collection: QualifiedCollection, + field: String, + surrogate: Surrogate, + pk_bytes: Vec, + vector: Vec, + payload: Vec, + quantization: nodedb_types::VectorQuantization, + storage_dtype: nodedb_types::VectorStorageDtype, + payload_indexes: Vec<(String, nodedb_types::PayloadIndexKind)>, + #[serde(default)] + returning: Option, + #[serde(default)] + rls_filters: Vec, + }, + + /// Vector-primary `INSERT ... ON CONFLICT DO NOTHING`: an existing + /// primary key leaves the stored row alone and reports zero rows. + DirectInsertIfAbsent { + collection: QualifiedCollection, + field: String, + surrogate: Surrogate, + pk_bytes: Vec, + vector: Vec, + payload: Vec, + quantization: nodedb_types::VectorQuantization, + storage_dtype: nodedb_types::VectorStorageDtype, + payload_indexes: Vec<(String, nodedb_types::PayloadIndexKind)>, + #[serde(default)] + returning: Option, + #[serde(default)] + rls_filters: Vec, + }, + + /// Vector-primary `DELETE`. Removes the HNSW node, its payload bitmap + /// entries, and the payload sidecar row of every targeted surrogate. + /// Reports the number of rows that existed. + DirectDelete { + collection: QualifiedCollection, + /// Vector column name; keys the HNSW index. + field: String, + targets: VectorWriteTargets, + /// When `Some`, return the removed rows' sidecar images. + #[serde(default)] + returning: Option, + /// Read filters gating the rows `returning` emits. + #[serde(default)] + rls_filters: Vec, + /// Write policy decided against each removed row's sidecar image. + rls_write_check: nodedb_types::RlsWriteCheck, + }, + + /// Vector-primary `TRUNCATE`. Removes every live row of the collection's + /// primary index: each HNSW node, its payload bitmap entries, and its + /// payload sidecar row. Reports the number of rows that existed. + DirectTruncate { + collection: QualifiedCollection, + /// Vector column name; keys the HNSW index. + field: String, + /// `TRUNCATE ... RESTART IDENTITY`: the Control Plane resets the + /// collection's sequences after the Data Plane clears the rows. + restart_identity: bool, + }, + + /// Vector-primary `UPDATE`. A `new_vector` rebuilds the HNSW node under + /// the same surrogate; `payload_patch` merges into the sidecar and moves + /// the payload bitmap entries. Reports the number of rows that existed. + DirectUpdate { + collection: QualifiedCollection, + /// Vector column name; keys the HNSW index. + field: String, + targets: VectorWriteTargets, + /// Replacement vector, when the statement assigns the vector column. + new_vector: Option>, + /// Assignments to non-vector columns, applied to the stored payload. + payload_patch: Vec<(String, UpdateValue)>, + quantization: nodedb_types::VectorQuantization, + storage_dtype: nodedb_types::VectorStorageDtype, + payload_indexes: Vec<(String, nodedb_types::PayloadIndexKind)>, + /// When `Some`, return the rows' stored post-images. + #[serde(default)] + returning: Option, + /// Read filters gating the rows `returning` emits. + #[serde(default)] + rls_filters: Vec, + /// Write policy decided against each row's post-image. + rls_write_check: nodedb_types::RlsWriteCheck, + }, + + // ── Resolve-before-propose (governed vector-primary writes) ───────── + /// Read-only: report every row mutation the wrapped vector-primary write + /// would apply, and the response payload it would return, without + /// applying anything. + /// + /// Wraps a [`VectorOp::DirectDelete`], [`VectorOp::DirectUpdate`], or + /// [`VectorOp::DirectUpsert`] verbatim, live write predicate included — + /// the handler resolves the targets, reads each row's sidecar, computes + /// the post-image, and decides the policy here, where the writing + /// identity is still available. A follower has none, so the predicate + /// can never cross the Raft wire. + ResolveDirectWrite(Box), + + /// Apply exactly the mutations a [`VectorOp::ResolveDirectWrite`] + /// reported, then return `response_payload` verbatim. + /// + /// No predicate is evaluated and no image is recomputed: the Control + /// Plane already decided both, and `rls_write_check` carries the verdict + /// (`RlsWriteCheck::DecidedEarlierInRequest`). Every mutation's pre-image + /// is checked against the stored sidecar before the first one applies, + /// so a resolution that drifted under a concurrent write applies nothing + /// and asks for a retry. + ResolvedDirectWrite { + collection: QualifiedCollection, + /// Vector column name; keys the HNSW index. + field: String, + /// Index settings a first write into a new index registers; an + /// `Upsert` mutation into an empty collection creates the index. + quantization: nodedb_types::VectorQuantization, + storage_dtype: nodedb_types::VectorStorageDtype, + payload_indexes: Vec<(String, nodedb_types::PayloadIndexKind)>, + mutations: Vec, + /// The statement's reply, decided while resolving. Applying nodes + /// return it unchanged, so leader and follower report the same thing. + response_payload: Vec, + rls_write_check: nodedb_types::RlsWriteCheck, }, } diff --git a/nodedb-physical/src/physical_plan/vector/resolved_mutation.rs b/nodedb-physical/src/physical_plan/vector/resolved_mutation.rs new file mode 100644 index 000000000..9f6bb9089 --- /dev/null +++ b/nodedb-physical/src/physical_plan/vector/resolved_mutation.rs @@ -0,0 +1,103 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The decided mutation set a governed vector-primary write resolved to. +//! +//! A vector-primary `DELETE` / `UPDATE` / conflict-patching `UPSERT` on a +//! collection with a row-level write policy cannot cross the Raft wire +//! carrying the live predicate: a follower has no writing identity to decide +//! it against. The Control Plane resolves the write against the rows the +//! Data Plane holds, decides the policy there, and ships the row mutations +//! themselves. + +use nodedb_types::Surrogate; + +/// One row mutation a resolved vector-primary write applies. +/// +/// Every variant carries the sidecar bytes the resolve read for its row — +/// the apply's drift check. A row's sidecar is its `zerompk` TAGGED payload +/// map and is never empty once written, so `Vec::new()` on `Delete` / +/// `Update` names a bound node with no sidecar row behind it, and the apply +/// compares it against an absent sidecar. +#[derive( + Debug, + Clone, + PartialEq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum VectorResolvedMutation { + /// Remove `surrogate`'s HNSW node, bitmap entries, and sidecar. + Delete { + surrogate: Surrogate, + /// Sidecar bytes the resolve read; the node must still be bound and + /// hold exactly these bytes. + old_payload: Vec, + }, + /// Rewrite `surrogate`'s row: a `new_vector` rebuilds the HNSW node + /// under the same surrogate, `merged_payload` replaces the sidecar and + /// moves the bitmap entries. + Update { + surrogate: Surrogate, + new_vector: Option>, + /// The full post-image sidecar, already merged and lower-cased, + /// in the sidecar's own `zerompk` TAGGED encoding. + merged_payload: Vec, + /// Sidecar bytes the resolve read; see `Delete::old_payload`. + old_payload: Vec, + }, + /// Store a whole row under `surrogate`, replacing any existing one. + Upsert { + surrogate: Surrogate, + /// UTF-8 of the declared primary-key value. Followers bind the + /// leader-assigned surrogate to this exact key. + pk_bytes: Vec, + vector: Vec, + /// The sidecar to store, in its own `zerompk` TAGGED encoding. + payload: Vec, + /// `None` requires the surrogate to still be unbound; `Some(bytes)` + /// requires a bound node whose sidecar holds exactly `bytes`. + old_payload: Option>, + }, +} + +impl VectorResolvedMutation { + /// The surrogate this one mutation targets. + pub fn surrogate(&self) -> Surrogate { + match self { + VectorResolvedMutation::Delete { surrogate, .. } + | VectorResolvedMutation::Update { surrogate, .. } + | VectorResolvedMutation::Upsert { surrogate, .. } => *surrogate, + } + } + + /// The vector this mutation stores, when it stores one. + pub fn stored_vector(&self) -> Option<&[f32]> { + match self { + VectorResolvedMutation::Delete { .. } => None, + VectorResolvedMutation::Update { new_vector, .. } => new_vector.as_deref(), + VectorResolvedMutation::Upsert { vector, .. } => Some(vector.as_slice()), + } + } +} + +/// What `VectorOp::ResolveDirectWrite` reports back: every row mutation the +/// intercepted write applies, plus the exact response payload the statement +/// returns once they all apply cleanly. +/// +/// An empty `mutations` list is a legitimate outcome — a predicate that +/// matches no row writes nothing and still owes its `{"affected": 0}` reply. +#[derive( + Debug, + Clone, + PartialEq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct VectorResolveOutcome { + pub mutations: Vec, + pub response_payload: Vec, +} diff --git a/nodedb-physical/src/physical_plan/vector/write.rs b/nodedb-physical/src/physical_plan/vector/write.rs new file mode 100644 index 000000000..31927d64b --- /dev/null +++ b/nodedb-physical/src/physical_plan/vector/write.rs @@ -0,0 +1,50 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Row-existence and row-targeting shapes of the vector-primary direct +//! writes. + +use nodedb_types::Surrogate; + +/// Row-existence semantics of a vector-primary direct write, as the three +/// insert-family ops carry them on the durable record. +#[derive( + Debug, + Clone, + Copy, + PartialEq, + Eq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum VectorDirectWriteIntent { + /// [`VectorOp::DirectInsert`](super::VectorOp::DirectInsert): an existing + /// key is a `unique_violation`. + Insert, + /// [`VectorOp::DirectInsertIfAbsent`](super::VectorOp::DirectInsertIfAbsent): + /// an existing key is left alone. + InsertIfAbsent, + /// [`VectorOp::DirectUpsert`](super::VectorOp::DirectUpsert): an existing + /// key is replaced or patched. + Upsert, +} + +/// The rows a vector-primary `DELETE` / `UPDATE` targets. +#[derive( + Debug, + Clone, + PartialEq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum VectorWriteTargets { + /// Surrogates the Control Plane resolved from primary-key equalities. + /// `Surrogate::ZERO` is a key with no binding; it matches no row. + Surrogates(Vec), + /// Serialized `Vec` the Data Plane evaluates against every + /// payload sidecar row. Empty bytes match every row. + Predicate(Vec), +} diff --git a/nodedb-query/src/msgpack_scan/kv_body.rs b/nodedb-query/src/msgpack_scan/kv_body.rs new file mode 100644 index 000000000..0e746142c --- /dev/null +++ b/nodedb-query/src/msgpack_scan/kv_body.rs @@ -0,0 +1,240 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Shape-preserving decode / encode of a KV row body for read-modify-write. +//! +//! A KV body is stored in one of two shapes (see [`super::kv_row_msgpack`]): +//! a msgpack map for typed columns, or raw scalar bytes for the single-`value` +//! SQL form and RESP `SET`. A merge decodes the body into the same row object +//! reads present, mutates it, and re-encodes into the shape it came from. +//! Decoding raw bytes as msgpack instead either fails (multi-byte value) or +//! reads the first byte as a fixint and discards the rest. + +use nodedb_types::{MsgpackError, NotScalar, Value, scalar_to_raw_bytes}; + +use crate::msgpack_scan::{map_header, write_map_header, write_str}; + +/// The on-disk shape of a KV row body. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum KvBodyShape { + /// Scalar bytes, not msgpack: the single-`value` SQL form and RESP `SET`. + Raw, + /// A msgpack map of typed columns. + Map, +} + +/// A merged row cannot be encoded back into its body shape. +#[derive(Debug, thiserror::Error)] +pub enum KvBodyError { + /// The stored map body is not well-formed msgpack. + #[error("KV map body: {0}")] + Decode(#[from] MsgpackError), + /// The merged row is not an object. + #[error("KV row must be an object, got {kind}")] + RowNotObject { + /// `Value::type_name()` of the merged row. + kind: &'static str, + }, + /// A raw body holds only `value`; the merged row lost it. + #[error("KV raw body has no `value` field after merge")] + RawMissingValue, + /// A raw body holds only `value`; the merged row carries other keys. + #[error("KV row holds a bare `value`, not typed columns; cannot set {keys}")] + RawExtraKeys { + /// The offending keys, sorted, comma-separated. + keys: String, + }, + /// A raw body holds a scalar; the merged `value` is not one. + #[error("KV raw body `value`: {0}")] + RawNotScalar(#[from] NotScalar), + /// The merged map row does not encode. + #[error("KV map body encode: {0}")] + Encode(zerompk::Error), +} + +/// The shape of a stored body: a msgpack map header marks a map, anything +/// else (an empty body included) is raw scalar bytes. +pub fn kv_body_shape(body: &[u8]) -> KvBodyShape { + if map_header(body, 0).is_some() { + KvBodyShape::Map + } else { + KvBodyShape::Raw + } +} + +/// Decode a stored body into the row object a merge operates on. +/// +/// A map body decodes as is. A raw body becomes `{"value": }`, the +/// exact row `kv_row_msgpack` presents to reads (non-UTF-8 bytes take the +/// lossy view there too). An empty body is a raw empty string. Returns the +/// shape so the writer re-encodes in the same one. +pub fn kv_body_to_row(body: &[u8]) -> Result<(Value, KvBodyShape), MsgpackError> { + if kv_body_shape(body) == KvBodyShape::Map { + return Ok((nodedb_types::value_from_msgpack(body)?, KvBodyShape::Map)); + } + let mut row = std::collections::HashMap::with_capacity(1); + row.insert( + "value".to_string(), + Value::String(String::from_utf8_lossy(body).into_owned()), + ); + Ok((Value::Object(row), KvBodyShape::Raw)) +} + +/// Encode a merged row back into `shape`. +/// +/// `Map` writes the fields in key order, so the same logical row encodes to +/// the same bytes on every node and on WAL replay. `Raw` accepts only an +/// object whose single key is `value` holding a scalar; the scalar encodes +/// through [`scalar_to_raw_bytes`], the same rule the SQL lowering uses for +/// a fresh insert. Any other object is an error naming the extra keys: a raw +/// row never silently turns into a map. +pub fn row_to_kv_body(row: &Value, shape: KvBodyShape) -> Result, KvBodyError> { + let Value::Object(map) = row else { + return Err(KvBodyError::RowNotObject { + kind: row.type_name(), + }); + }; + match shape { + KvBodyShape::Map => { + let mut fields: Vec<(&String, &Value)> = map.iter().collect(); + fields.sort_unstable_by(|a, b| a.0.cmp(b.0)); + let mut buf = Vec::with_capacity(map.len() * 16); + write_map_header(&mut buf, fields.len()); + for (key, value) in fields { + write_str(&mut buf, key); + let encoded = nodedb_types::value_to_msgpack(value).map_err(KvBodyError::Encode)?; + buf.extend_from_slice(&encoded); + } + Ok(buf) + } + KvBodyShape::Raw => { + let mut extra: Vec<&str> = map + .keys() + .map(String::as_str) + .filter(|k| *k != "value") + .collect(); + if !extra.is_empty() { + extra.sort_unstable(); + return Err(KvBodyError::RawExtraKeys { + keys: extra.join(", "), + }); + } + let value = map.get("value").ok_or(KvBodyError::RawMissingValue)?; + Ok(scalar_to_raw_bytes(value)?) + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::collections::HashMap; + + fn value_of(row: &Value) -> &Value { + row.get("value").expect("row carries `value`") + } + + #[test] + fn raw_string_round_trips_in_raw_shape() { + let (row, shape) = kv_body_to_row(b"second-longer-value").unwrap(); + assert_eq!(shape, KvBodyShape::Raw); + assert_eq!(value_of(&row), &Value::String("second-longer-value".into())); + assert_eq!( + row_to_kv_body(&row, shape).unwrap(), + b"second-longer-value".to_vec() + ); + } + + #[test] + fn raw_single_byte_value_is_a_string_not_a_fixint() { + // `b"1"` is 0x31, a valid msgpack fixint. It must still read as the + // string "1" and write back as the byte 0x31. + let (row, shape) = kv_body_to_row(b"1").unwrap(); + assert_eq!(shape, KvBodyShape::Raw); + assert_eq!(value_of(&row), &Value::String("1".into())); + assert_eq!(row_to_kv_body(&row, shape).unwrap(), b"1".to_vec()); + } + + #[test] + fn raw_integer_value_encodes_as_decimal_text() { + let mut row = HashMap::new(); + row.insert("value".to_string(), Value::Integer(42)); + assert_eq!( + row_to_kv_body(&Value::Object(row), KvBodyShape::Raw).unwrap(), + b"42".to_vec() + ); + } + + #[test] + fn map_body_round_trips_in_map_shape() { + let mut fields = HashMap::new(); + fields.insert("n".to_string(), Value::Integer(7)); + let body = nodedb_types::value_to_msgpack(&Value::Object(fields.clone())).unwrap(); + + let (row, shape) = kv_body_to_row(&body).unwrap(); + assert_eq!(shape, KvBodyShape::Map); + assert_eq!(row, Value::Object(fields.clone())); + + let out = row_to_kv_body(&row, shape).unwrap(); + assert_eq!( + nodedb_types::value_from_msgpack(&out).unwrap(), + Value::Object(fields) + ); + } + + #[test] + fn map_body_encodes_fields_in_key_order() { + let mut fields = HashMap::new(); + fields.insert("b".to_string(), Value::Integer(2)); + fields.insert("a".to_string(), Value::Integer(1)); + let out = row_to_kv_body(&Value::Object(fields), KvBodyShape::Map).unwrap(); + assert_eq!(out[0], 0x82, "fixmap of two entries"); + let key_at = |name: u8| out.windows(2).position(|w| w == [0xa1, name]).unwrap(); + assert!( + key_at(b'a') < key_at(b'b'), + "keys must be written in sorted order" + ); + } + + #[test] + fn raw_shape_with_extra_keys_is_an_error_naming_them() { + let mut row = HashMap::new(); + row.insert("value".to_string(), Value::String("x".into())); + row.insert("n".to_string(), Value::Integer(1)); + row.insert("a".to_string(), Value::Integer(2)); + let err = row_to_kv_body(&Value::Object(row), KvBodyShape::Raw).unwrap_err(); + match err { + KvBodyError::RawExtraKeys { keys } => assert_eq!(keys, "a, n"), + other => panic!("expected RawExtraKeys, got {other:?}"), + } + } + + #[test] + fn raw_shape_with_non_scalar_value_is_an_error() { + let mut row = HashMap::new(); + row.insert("value".to_string(), Value::Array(vec![Value::Integer(1)])); + let err = row_to_kv_body(&Value::Object(row), KvBodyShape::Raw).unwrap_err(); + assert!(matches!(err, KvBodyError::RawNotScalar(_)), "{err:?}"); + } + + #[test] + fn non_object_row_is_an_error_in_either_shape() { + for shape in [KvBodyShape::Raw, KvBodyShape::Map] { + let err = row_to_kv_body(&Value::Integer(1), shape).unwrap_err(); + assert!(matches!(err, KvBodyError::RowNotObject { kind: "int" })); + } + } + + #[test] + fn empty_body_is_a_raw_empty_string() { + let (row, shape) = kv_body_to_row(b"").unwrap(); + assert_eq!(shape, KvBodyShape::Raw); + assert_eq!(value_of(&row), &Value::String(String::new())); + assert!(row_to_kv_body(&row, shape).unwrap().is_empty()); + } + + #[test] + fn corrupt_map_body_is_a_decode_error() { + // fixmap header claiming one entry, then nothing. + assert!(kv_body_to_row(&[0x81]).is_err()); + } +} diff --git a/nodedb-query/src/msgpack_scan/kv_row.rs b/nodedb-query/src/msgpack_scan/kv_row.rs index 8f5b7d9a3..368cb18be 100644 --- a/nodedb-query/src/msgpack_scan/kv_row.rs +++ b/nodedb-query/src/msgpack_scan/kv_row.rs @@ -13,7 +13,9 @@ //! silently diverged: one of them appended raw bytes verbatim into a msgpack //! map, and every scan path served `value: 118` for a stored `"v1"`. -use crate::msgpack_scan::{inject_str_field, map_header, write_map_header, write_str}; +use crate::msgpack_scan::{ + KvBodyShape, inject_str_field, kv_body_shape, write_map_header, write_str, +}; /// Shape a KV entry into the canonical `{key, value…}` msgpack row. /// @@ -36,8 +38,12 @@ use crate::msgpack_scan::{inject_str_field, map_header, write_map_header, write_ /// lossy view is taken: a SQL `SELECT` over binary values is already degraded /// by the pgwire text protocol, and this keeps the output well-formed msgpack /// rather than merely different-but-still-broken. +/// +/// A read-modify-write decodes and re-encodes the body through +/// [`super::kv_body_to_row`] / [`super::row_to_kv_body`], which keep the +/// stored shape. pub fn kv_row_msgpack(key: &str, value: &[u8]) -> Vec { - if map_header(value, 0).is_some() { + if kv_body_shape(value) == KvBodyShape::Map { return inject_str_field(value, "key", key); } let mut buf = Vec::with_capacity(value.len() + key.len() + 16); diff --git a/nodedb-query/src/msgpack_scan/mod.rs b/nodedb-query/src/msgpack_scan/mod.rs index 62c227ee0..af657f92b 100644 --- a/nodedb-query/src/msgpack_scan/mod.rs +++ b/nodedb-query/src/msgpack_scan/mod.rs @@ -13,6 +13,7 @@ pub mod field; pub mod filter; pub mod group_key; pub mod index; +pub mod kv_body; pub mod kv_row; pub mod reader; pub mod sidecar; @@ -23,6 +24,7 @@ pub use compare::{compare_field_bytes, hash_field_bytes}; pub use field::{extract_field, extract_path}; pub use group_key::build_group_key; pub use index::FieldIndex; +pub use kv_body::{KvBodyError, KvBodyShape, kv_body_shape, kv_body_to_row, row_to_kv_body}; pub use kv_row::kv_row_msgpack; pub use reader::{ array_header, map_header, read_bin_advance, read_bool, read_f64, read_i64, read_null, read_str, diff --git a/nodedb-sql/src/engine_rules/array.rs b/nodedb-sql/src/engine_rules/array.rs index 523a9c3cd..9b5f18112 100644 --- a/nodedb-sql/src/engine_rules/array.rs +++ b/nodedb-sql/src/engine_rules/array.rs @@ -61,6 +61,14 @@ impl EngineRules for ArrayRules { )) } + fn plan_truncate(&self, _p: TruncateParams) -> Result> { + Err(unsupported( + "TRUNCATE", + "use DROP ARRAY to remove the array, or \ + DELETE FROM ARRAY WHERE COORDS IN (...) to remove cells", + )) + } + fn plan_aggregate(&self, _p: AggregateParams) -> Result { Err(unsupported( "GROUP BY", diff --git a/nodedb-sql/src/engine_rules/columnar.rs b/nodedb-sql/src/engine_rules/columnar.rs index e6939dcb6..2567192f6 100644 --- a/nodedb-sql/src/engine_rules/columnar.rs +++ b/nodedb-sql/src/engine_rules/columnar.rs @@ -108,6 +108,14 @@ impl EngineRules for ColumnarRules { }]) } + fn plan_truncate(&self, p: TruncateParams) -> Result> { + Ok(vec![SqlPlan::Truncate { + collection: p.collection, + engine: EngineType::Columnar, + restart_identity: p.restart_identity, + }]) + } + fn plan_aggregate(&self, p: AggregateParams) -> Result { if p.temporal.is_temporal() && !p.bitemporal { return Err(SqlError::Unsupported { diff --git a/nodedb-sql/src/engine_rules/document_schemaless.rs b/nodedb-sql/src/engine_rules/document_schemaless.rs index bdad8b51f..7f7f1d315 100644 --- a/nodedb-sql/src/engine_rules/document_schemaless.rs +++ b/nodedb-sql/src/engine_rules/document_schemaless.rs @@ -122,6 +122,14 @@ impl EngineRules for SchemalessRules { }]) } + fn plan_truncate(&self, p: TruncateParams) -> Result> { + Ok(vec![SqlPlan::Truncate { + collection: p.collection, + engine: EngineType::DocumentSchemaless, + restart_identity: p.restart_identity, + }]) + } + fn plan_aggregate(&self, p: AggregateParams) -> Result { let base_scan = SqlPlan::Scan { collection: p.collection, diff --git a/nodedb-sql/src/engine_rules/document_strict.rs b/nodedb-sql/src/engine_rules/document_strict.rs index 3e8d25327..ad447fe07 100644 --- a/nodedb-sql/src/engine_rules/document_strict.rs +++ b/nodedb-sql/src/engine_rules/document_strict.rs @@ -122,6 +122,14 @@ impl EngineRules for StrictRules { }]) } + fn plan_truncate(&self, p: TruncateParams) -> Result> { + Ok(vec![SqlPlan::Truncate { + collection: p.collection, + engine: EngineType::DocumentStrict, + restart_identity: p.restart_identity, + }]) + } + fn plan_aggregate(&self, p: AggregateParams) -> Result { let base_scan = SqlPlan::Scan { collection: p.collection, diff --git a/nodedb-sql/src/engine_rules/kv.rs b/nodedb-sql/src/engine_rules/kv.rs index 20d4ac5f3..434fc6b9c 100644 --- a/nodedb-sql/src/engine_rules/kv.rs +++ b/nodedb-sql/src/engine_rules/kv.rs @@ -97,6 +97,14 @@ impl EngineRules for KvRules { }]) } + fn plan_truncate(&self, p: TruncateParams) -> Result> { + Ok(vec![SqlPlan::Truncate { + collection: p.collection, + engine: EngineType::KeyValue, + restart_identity: p.restart_identity, + }]) + } + fn plan_aggregate(&self, p: AggregateParams) -> Result { let base_scan = SqlPlan::Scan { collection: p.collection, diff --git a/nodedb-sql/src/engine_rules/mod.rs b/nodedb-sql/src/engine_rules/mod.rs index a208d2637..2de2f276c 100644 --- a/nodedb-sql/src/engine_rules/mod.rs +++ b/nodedb-sql/src/engine_rules/mod.rs @@ -15,7 +15,7 @@ pub mod timeseries; pub(crate) use index_lookup::try_document_index_lookup; pub use params::{ AggregateParams, DeleteParams, InsertParams, MergeParams, PointGetParams, ScanParams, - UpdateFromParams, UpdateParams, UpsertParams, + TruncateParams, UpdateFromParams, UpdateParams, UpsertParams, }; pub use resolve::resolve_engine_rules; pub use rules::EngineRules; diff --git a/nodedb-sql/src/engine_rules/params.rs b/nodedb-sql/src/engine_rules/params.rs index e4e98a890..a7454d6f3 100644 --- a/nodedb-sql/src/engine_rules/params.rs +++ b/nodedb-sql/src/engine_rules/params.rs @@ -99,6 +99,12 @@ pub struct DeleteParams { pub target_keys: Vec, } +/// Parameters for planning a TRUNCATE operation. +pub struct TruncateParams { + pub collection: String, + pub restart_identity: bool, +} + /// Parameters for planning a MERGE operation. pub struct MergeParams { pub collection: String, diff --git a/nodedb-sql/src/engine_rules/rules.rs b/nodedb-sql/src/engine_rules/rules.rs index 1409cf150..496520f1e 100644 --- a/nodedb-sql/src/engine_rules/rules.rs +++ b/nodedb-sql/src/engine_rules/rules.rs @@ -7,7 +7,7 @@ use crate::types::SqlPlan; use super::params::{ AggregateParams, DeleteParams, InsertParams, MergeParams, PointGetParams, ScanParams, - UpdateFromParams, UpdateParams, UpsertParams, + TruncateParams, UpdateFromParams, UpdateParams, UpsertParams, }; /// Engine-specific planning rules. @@ -37,6 +37,9 @@ pub trait EngineRules { fn plan_update_from(&self, params: UpdateFromParams) -> Result>; /// Plan a DELETE (point or bulk). fn plan_delete(&self, params: DeleteParams) -> Result>; + /// Plan a TRUNCATE. Returns `Err(SqlError::Unsupported)` for engines + /// that carry no whole-collection clear on the SQL surface (array). + fn plan_truncate(&self, params: TruncateParams) -> Result>; /// Plan a GROUP BY / aggregate query. fn plan_aggregate(&self, params: AggregateParams) -> Result; /// Plan a MERGE statement. diff --git a/nodedb-sql/src/engine_rules/spatial.rs b/nodedb-sql/src/engine_rules/spatial.rs index 6373f11e9..59bf058fe 100644 --- a/nodedb-sql/src/engine_rules/spatial.rs +++ b/nodedb-sql/src/engine_rules/spatial.rs @@ -109,6 +109,14 @@ impl EngineRules for SpatialRules { }]) } + fn plan_truncate(&self, p: TruncateParams) -> Result> { + Ok(vec![SqlPlan::Truncate { + collection: p.collection, + engine: EngineType::Spatial, + restart_identity: p.restart_identity, + }]) + } + fn plan_aggregate(&self, p: AggregateParams) -> Result { let base_scan = SqlPlan::Scan { collection: p.collection, diff --git a/nodedb-sql/src/engine_rules/timeseries.rs b/nodedb-sql/src/engine_rules/timeseries.rs index 4c4122bed..f5ea41f9a 100644 --- a/nodedb-sql/src/engine_rules/timeseries.rs +++ b/nodedb-sql/src/engine_rules/timeseries.rs @@ -93,6 +93,14 @@ impl EngineRules for TimeseriesRules { }) } + fn plan_truncate(&self, p: TruncateParams) -> Result> { + Ok(vec![SqlPlan::Truncate { + collection: p.collection, + engine: EngineType::Timeseries, + restart_identity: p.restart_identity, + }]) + } + fn plan_aggregate(&self, p: AggregateParams) -> Result { if p.temporal.is_temporal() && !p.bitemporal { return Err(SqlError::Unsupported { diff --git a/nodedb-sql/src/lib.rs b/nodedb-sql/src/lib.rs index 41e3e7d1b..62fbf1692 100644 --- a/nodedb-sql/src/lib.rs +++ b/nodedb-sql/src/lib.rs @@ -161,7 +161,7 @@ fn plan_statements( plans.append(&mut delete_plans); } StatementKind::Truncate(stmt) => { - let mut trunc_plans = planner::dml::plan_truncate_stmt(stmt)?; + let mut trunc_plans = planner::dml::plan_truncate_stmt(stmt, catalog)?; plans.append(&mut trunc_plans); } StatementKind::Merge(stmt) => { diff --git a/nodedb-sql/src/planner/catalog_fold.rs b/nodedb-sql/src/planner/catalog_fold.rs index 5aca0895d..2ba9abb95 100644 --- a/nodedb-sql/src/planner/catalog_fold.rs +++ b/nodedb-sql/src/planner/catalog_fold.rs @@ -175,7 +175,9 @@ fn walk_plan( mut plan @ (SqlPlan::DocumentIndexLookup { .. } | SqlPlan::Update { .. } - | SqlPlan::Delete { .. }) => { + | SqlPlan::Delete { .. } + | SqlPlan::VectorPrimaryUpdate { .. } + | SqlPlan::VectorPrimaryDelete { .. }) => { match &mut plan { SqlPlan::DocumentIndexLookup { filters, @@ -191,7 +193,7 @@ fn walk_plan( fold_sort_keys(sort_keys, catalog, database_id, tenant_id); fold_windows(window_functions, catalog, database_id, tenant_id); } - SqlPlan::Delete { filters, .. } => { + SqlPlan::Delete { filters, .. } | SqlPlan::VectorPrimaryDelete { filters, .. } => { for filter in filters { fold_filter(filter, catalog, database_id, tenant_id); } @@ -200,6 +202,11 @@ fn walk_plan( assignments, filters, .. + } + | SqlPlan::VectorPrimaryUpdate { + assignments, + filters, + .. } => { for (_, expr) in assignments { let owned = std::mem::replace(expr, SqlExpr::Wildcard); diff --git a/nodedb-sql/src/planner/catalog_plan_validate.rs b/nodedb-sql/src/planner/catalog_plan_validate.rs index fb0a24f18..6230bd833 100644 --- a/nodedb-sql/src/planner/catalog_plan_validate.rs +++ b/nodedb-sql/src/planner/catalog_plan_validate.rs @@ -45,13 +45,18 @@ pub(super) fn validate_catalog_exprs( assignments, filters, .. + } + | SqlPlan::VectorPrimaryUpdate { + assignments, + filters, + .. } => { for (_, expr) in assignments { validate_expr(expr, catalog, database_id, tenant_id)?; } validate_filters(filters, catalog, database_id, tenant_id)?; } - SqlPlan::Delete { filters, .. } => { + SqlPlan::Delete { filters, .. } | SqlPlan::VectorPrimaryDelete { filters, .. } => { validate_filters(filters, catalog, database_id, tenant_id)? } SqlPlan::UpdateFrom { diff --git a/nodedb-sql/src/planner/dml/insert.rs b/nodedb-sql/src/planner/dml/insert.rs index b7bfb4293..57a54166c 100644 --- a/nodedb-sql/src/planner/dml/insert.rs +++ b/nodedb-sql/src/planner/dml/insert.rs @@ -5,8 +5,8 @@ use sqlparser::ast; use super::super::dml_helpers::{ - KvInsertParams, bind_insert_select_columns, build_kv_insert_plan, - build_vector_primary_insert_plan, resolve_insert_columns, + KvInsertParams, VectorPrimaryInsertParams, bind_insert_select_columns, build_kv_insert_plan, + build_vector_primary_insert_plan, is_vector_primary, resolve_insert_columns, }; use super::target::{ OnConflict, classify_on_conflict, column_schema, insert_columns, resolve_target, target_scope, @@ -88,17 +88,25 @@ pub fn plan_insert( // then every cell coerced and range-checked against its declared type. let typed = typed_rows(&info, &columns, rows_ast, catalog)?; - // Vector-primary collection: bypass document encoding. - if info.primary == nodedb_types::PrimaryEngine::Vector + // Vector-primary collection: bypass document encoding. The row's + // existence intent travels with the plan, as it does for KV. + if is_vector_primary(&info) && let Some(ref vpc) = info.vector_primary { - return build_vector_primary_insert_plan( - &table_name, + let intent = if if_absent { + VectorPrimaryInsertIntent::InsertIfAbsent + } else { + VectorPrimaryInsertIntent::Insert + }; + return build_vector_primary_insert_plan(VectorPrimaryInsertParams { + collection: &table_name, vpc, - &columns, - typed.rows, - typed.volatile_defaults, - ); + rows: typed.rows, + volatile_defaults: typed.volatile_defaults, + intent, + on_conflict_updates: Vec::new(), + primary_key: info.primary_key.clone(), + }); } // All other engines: delegate to engine rules. diff --git a/nodedb-sql/src/planner/dml/update_delete.rs b/nodedb-sql/src/planner/dml/update_delete.rs index b49470263..4d80382cd 100644 --- a/nodedb-sql/src/planner/dml/update_delete.rs +++ b/nodedb-sql/src/planner/dml/update_delete.rs @@ -9,10 +9,13 @@ use super::super::ast_helpers::{ flatten_and_expr, qualified_ident_pair, strip_and_convert_filters, }; use super::super::dml_helpers::{ + VectorPrimaryUpdateParams, build_vector_primary_delete_plan, + build_vector_primary_truncate_plan, build_vector_primary_update_plan, check_declared_float_ranges_in_assignments, check_declared_int_ranges_in_assignments, - extract_point_keys, extract_table_name_from_table_with_joins, + extract_point_keys, extract_table_name_from_table_with_joins, is_vector_primary, + refuse_vector_primary_shape, }; -use crate::engine_rules::{self, DeleteParams, UpdateFromParams, UpdateParams}; +use crate::engine_rules::{self, DeleteParams, TruncateParams, UpdateFromParams, UpdateParams}; use crate::error::{Result, SqlError}; use crate::parser::normalize::{ SCHEMA_QUALIFIED_MSG, normalize_ident, normalize_object_name_checked, @@ -77,6 +80,22 @@ pub fn plan_update(stmt: &ast::Statement, catalog: &dyn SqlCatalog) -> Result Result Result Result> { +/// +/// Each named collection resolves through the catalog: a vector-primary +/// collection lowers to `SqlPlan::VectorPrimaryTruncate`, every other +/// engine routes through its `EngineRules::plan_truncate`. +pub fn plan_truncate_stmt(stmt: &ast::Statement, catalog: &dyn SqlCatalog) -> Result> { let ast::Statement::Truncate(truncate) = stmt else { return Err(SqlError::Parse { detail: "expected TRUNCATE statement".into(), @@ -430,14 +467,43 @@ pub fn plan_truncate_stmt(stmt: &ast::Statement) -> Result> { truncate.identity, Some(sqlparser::ast::TruncateIdentityOption::Restart) ); - truncate - .table_names - .iter() - .map(|t| { - Ok(SqlPlan::Truncate { - collection: normalize_object_name_checked(&t.name)?, + let mut plans = Vec::with_capacity(truncate.table_names.len()); + for target in &truncate.table_names { + let table_name = normalize_object_name_checked(&target.name)?; + // An array lives in its own catalog namespace, so its refusal comes + // from `ArrayRules` before the collection lookup can miss it. + if catalog.array_exists(&table_name) { + plans.append( + &mut engine_rules::resolve_engine_rules(EngineType::Array).plan_truncate( + TruncateParams { + collection: table_name, + restart_identity, + }, + )?, + ); + continue; + } + let info = catalog + .get_collection(DatabaseId::DEFAULT, &table_name)? + .ok_or_else(|| SqlError::UnknownTable { + name: table_name.clone(), + })?; + // Vector-primary collection: see `plan_update`. + if is_vector_primary(&info) + && let Some(ref vpc) = info.vector_primary + { + plans.push(build_vector_primary_truncate_plan( + &table_name, + vpc, restart_identity, - }) - }) - .collect() + )); + continue; + } + let rules = engine_rules::resolve_engine_rules(info.engine); + plans.append(&mut rules.plan_truncate(TruncateParams { + collection: table_name, + restart_identity, + })?); + } + Ok(plans) } diff --git a/nodedb-sql/src/planner/dml/upsert.rs b/nodedb-sql/src/planner/dml/upsert.rs index f8d75d3ce..1ccfcf0bf 100644 --- a/nodedb-sql/src/planner/dml/upsert.rs +++ b/nodedb-sql/src/planner/dml/upsert.rs @@ -5,8 +5,9 @@ use sqlparser::ast; use super::super::dml_helpers::{ - KvInsertParams, build_kv_insert_plan, check_declared_float_ranges_in_assignments, - check_declared_int_ranges_in_assignments, resolve_insert_columns, + KvInsertParams, VectorPrimaryInsertParams, build_kv_insert_plan, + build_vector_primary_insert_plan, check_declared_float_ranges_in_assignments, + check_declared_int_ranges_in_assignments, is_vector_primary, resolve_insert_columns, }; use super::target::{ column_schema, insert_columns, resolve_target, target_scope, typed_rows, values_rows, @@ -78,6 +79,24 @@ fn plan_upsert_rows( )?; check_declared_int_ranges_in_assignments(&info.columns, &on_conflict_updates)?; check_declared_float_ranges_in_assignments(&info.columns, &on_conflict_updates)?; + + // Vector-primary collection: an existing key is replaced, or patched by + // the carried assignments. The document upsert path never reaches the + // HNSW index, so it cannot serve this collection. + if is_vector_primary(&info) + && let Some(ref vpc) = info.vector_primary + { + return build_vector_primary_insert_plan(VectorPrimaryInsertParams { + collection: &table_name, + vpc, + rows: typed.rows, + volatile_defaults: typed.volatile_defaults, + intent: VectorPrimaryInsertIntent::Upsert, + on_conflict_updates, + primary_key: info.primary_key.clone(), + }); + } + let column_schema = column_schema(&info); let rules = engine_rules::resolve_engine_rules(info.engine); rules.plan_upsert(UpsertParams { diff --git a/nodedb-sql/src/planner/dml_helpers/mod.rs b/nodedb-sql/src/planner/dml_helpers/mod.rs index 23a2d3446..853b27b09 100644 --- a/nodedb-sql/src/planner/dml_helpers/mod.rs +++ b/nodedb-sql/src/planner/dml_helpers/mod.rs @@ -7,6 +7,7 @@ //! - [`ast_extract`] — table-name / primary-key point-lookup extraction //! - [`declared_defaults`] — declared column DEFAULT materialization //! - [`vector_primary_insert`] — vector-primary collection insert plans +//! - [`vector_primary_dml`] — vector-primary collection update / delete / truncate plans //! - [`kv_insert`] — KV engine insert plans //! - [`insert_select_bind`] — `INSERT ... SELECT` target-column binding //! - [`params`] — parameter structs for the helpers above @@ -19,6 +20,7 @@ mod kv_insert; mod params; mod range_check; mod value_convert; +mod vector_primary_dml; mod vector_primary_insert; pub use ast_extract::extract_point_keys; @@ -33,4 +35,11 @@ pub(super) use range_check::{ check_declared_int_ranges, check_declared_int_ranges_in_assignments, coerce_and_check_rows, }; pub(super) use value_convert::convert_value_rows; -pub(super) use vector_primary_insert::build_vector_primary_insert_plan; +pub(super) use vector_primary_dml::{ + VectorPrimaryUpdateParams, build_vector_primary_delete_plan, + build_vector_primary_truncate_plan, build_vector_primary_update_plan, is_vector_primary, + refuse_vector_primary_shape, +}; +pub(super) use vector_primary_insert::{ + VectorPrimaryInsertParams, build_vector_primary_insert_plan, +}; diff --git a/nodedb-sql/src/planner/dml_helpers/vector_primary_dml.rs b/nodedb-sql/src/planner/dml_helpers/vector_primary_dml.rs new file mode 100644 index 000000000..67bb6dab8 --- /dev/null +++ b/nodedb-sql/src/planner/dml_helpers/vector_primary_dml.rs @@ -0,0 +1,152 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Plan construction for `UPDATE` / `DELETE` / `TRUNCATE` against a vector-primary +//! collection, plus the refusals for the write shapes it does not carry. + +use super::vector_primary_insert::sql_values_to_vector; +use crate::error::{Result, SqlError}; +use crate::types::*; + +/// Whether `info` names a vector-primary collection. +pub(crate) fn is_vector_primary(info: &CollectionInfo) -> bool { + info.primary == nodedb_types::PrimaryEngine::Vector && info.vector_primary.is_some() +} + +/// Refuse a write shape a vector-primary collection cannot execute. +/// +/// `UPDATE ... FROM` and `MERGE` expand into document point writes on the +/// Control Plane. A vector-primary row has no document store behind it, so +/// those expansions would write nothing the collection can read back. +pub(crate) fn refuse_vector_primary_shape(info: &CollectionInfo, shape: &str) -> Result<()> { + if !is_vector_primary(info) { + return Ok(()); + } + Err(SqlError::Unsupported { + detail: format!( + "{shape} is not supported on vector-primary collection '{}'; use UPDATE ... WHERE, \ + DELETE ... WHERE, or UPSERT INTO", + info.name + ), + }) +} + +/// Inputs to [`build_vector_primary_update_plan`]. +pub(crate) struct VectorPrimaryUpdateParams<'a> { + pub collection: &'a str, + pub info: &'a CollectionInfo, + pub vpc: &'a nodedb_types::VectorPrimaryConfig, + pub assignments: Vec<(String, SqlExpr)>, + pub filters: Vec, + pub target_keys: Vec, + pub returning: bool, +} + +/// Build a `SqlPlan::VectorPrimaryUpdate`. +/// +/// The vector-column assignment is peeled into `new_vector`. It must be an +/// array literal: the HNSW node is rebuilt from it, and the Data Plane has +/// no row expression evaluator for a vector. An assignment to the primary +/// key is refused: the key is the row's surrogate identity. +pub(crate) fn build_vector_primary_update_plan( + params: VectorPrimaryUpdateParams<'_>, +) -> Result> { + let VectorPrimaryUpdateParams { + collection, + info, + vpc, + assignments, + filters, + target_keys, + returning, + } = params; + let mut new_vector: Option> = None; + let mut payload_assignments = Vec::with_capacity(assignments.len()); + for (column, expr) in assignments { + if info.primary_key.as_deref() == Some(column.as_str()) { + return Err(SqlError::Unsupported { + detail: format!( + "UPDATE of primary key '{column}' on vector-primary collection \ + '{collection}' is not supported; DELETE the row and INSERT it under \ + the new key" + ), + }); + } + if column == vpc.vector_field { + new_vector = Some(vector_literal(&column, &expr)?); + continue; + } + payload_assignments.push((column, expr)); + } + Ok(vec![SqlPlan::VectorPrimaryUpdate { + collection: collection.to_string(), + field: vpc.vector_field.clone(), + quantization: vpc.quantization, + storage_dtype: vpc.storage_dtype, + payload_indexes: vpc.payload_indexes.clone(), + new_vector, + assignments: payload_assignments, + filters, + target_keys, + returning, + primary_key: info.primary_key.clone(), + }]) +} + +/// Build a `SqlPlan::VectorPrimaryDelete`. +pub(crate) fn build_vector_primary_delete_plan( + collection: &str, + info: &CollectionInfo, + vpc: &nodedb_types::VectorPrimaryConfig, + filters: Vec, + target_keys: Vec, +) -> Vec { + vec![SqlPlan::VectorPrimaryDelete { + collection: collection.to_string(), + field: vpc.vector_field.clone(), + filters, + target_keys, + primary_key: info.primary_key.clone(), + }] +} + +/// Build a `SqlPlan::VectorPrimaryTruncate`. +pub(crate) fn build_vector_primary_truncate_plan( + collection: &str, + vpc: &nodedb_types::VectorPrimaryConfig, + restart_identity: bool, +) -> SqlPlan { + SqlPlan::VectorPrimaryTruncate { + collection: collection.to_string(), + field: vpc.vector_field.clone(), + restart_identity, + } +} + +/// The `f32` components of an assigned vector expression. +/// +/// Accepts `ARRAY[...]` of numeric literals and a pre-folded array literal. +fn vector_literal(field: &str, expr: &SqlExpr) -> Result> { + match expr { + SqlExpr::Literal(SqlValue::Array(items)) => sql_values_to_vector(field, items), + SqlExpr::ArrayLiteral(elems) => { + let items = elems + .iter() + .map(|e| match e { + SqlExpr::Literal(v) => Ok(v.clone()), + other => Err(SqlError::Unsupported { + detail: format!( + "vector field '{field}' must be assigned an array of numeric \ + literals, got element {other:?}" + ), + }), + }) + .collect::>>()?; + sql_values_to_vector(field, &items) + } + other => Err(SqlError::Unsupported { + detail: format!( + "vector field '{field}' must be assigned an array literal, got {other:?}" + ), + }), + } +} diff --git a/nodedb-sql/src/planner/dml_helpers/vector_primary_insert.rs b/nodedb-sql/src/planner/dml_helpers/vector_primary_insert.rs index 1b48ba1ad..08d88e256 100644 --- a/nodedb-sql/src/planner/dml_helpers/vector_primary_insert.rs +++ b/nodedb-sql/src/planner/dml_helpers/vector_primary_insert.rs @@ -1,10 +1,57 @@ // SPDX-License-Identifier: Apache-2.0 -//! Plan construction for `INSERT` against a vector-primary collection. +//! Plan construction for `INSERT` / `UPSERT` against a vector-primary +//! collection. use crate::error::{Result, SqlError}; use crate::types::*; +/// Inputs to [`build_vector_primary_insert_plan`]. +pub(crate) struct VectorPrimaryInsertParams<'a> { + pub collection: &'a str, + pub vpc: &'a nodedb_types::VectorPrimaryConfig, + pub rows: Vec>, + pub volatile_defaults: bool, + pub intent: VectorPrimaryInsertIntent, + pub on_conflict_updates: Vec<(String, SqlExpr)>, + pub primary_key: Option, +} + +/// Convert the elements of a vector literal to `f32`. +/// +/// Integers and decimals are accepted alongside floats. Anything else is +/// refused by name. +pub(crate) fn sql_values_to_vector(field: &str, items: &[SqlValue]) -> Result> { + items + .iter() + .map(|v| match v { + SqlValue::Float(f) => Ok(*f as f32), + SqlValue::Int(i) => Ok(*i as f32), + SqlValue::Decimal(d) => { + use rust_decimal::prelude::ToPrimitive; + d.to_f32().ok_or_else(|| SqlError::Parse { + detail: format!("vector element decimal '{d}' is out of f32 range"), + }) + } + other => Err(SqlError::Parse { + detail: format!("vector field '{field}' must contain numbers, got {other:?}"), + }), + }) + .collect() +} + +/// Convert an assigned vector-column value to `f32` components. +/// +/// The value must be an array literal. Anything else is refused by name. +pub(crate) fn sql_value_to_vector(field: &str, value: &SqlValue) -> Result> { + match value { + SqlValue::Array(items) => sql_values_to_vector(field, items), + other => Err(SqlError::Parse { + detail: format!("vector field '{field}' must be an array literal, got {other:?}"), + }), + } +} + /// Build a `SqlPlan::VectorPrimaryInsert` from parsed rows. /// /// Extracts the vector-field column into `vector: Vec` and collects @@ -16,12 +63,17 @@ use crate::types::*; /// one. `volatile_defaults` reports whether any of those defaults was volatile, /// which keeps the plan out of the physical-plan cache. pub(crate) fn build_vector_primary_insert_plan( - collection: &str, - vpc: &nodedb_types::VectorPrimaryConfig, - _columns: &[String], - rows: Vec>, - volatile_defaults: bool, + params: VectorPrimaryInsertParams<'_>, ) -> Result> { + let VectorPrimaryInsertParams { + collection, + vpc, + rows, + volatile_defaults, + intent, + on_conflict_updates, + primary_key, + } = params; let mut result_rows = Vec::with_capacity(rows.len()); for row in rows { let mut vector: Option> = None; @@ -29,39 +81,7 @@ pub(crate) fn build_vector_primary_insert_plan( for (col, val) in row { if col == vpc.vector_field { - match val { - SqlValue::Array(items) => { - let floats: Result> = items - .iter() - .map(|v| match v { - SqlValue::Float(f) => Ok(*f as f32), - SqlValue::Int(i) => Ok(*i as f32), - SqlValue::Decimal(d) => { - use rust_decimal::prelude::ToPrimitive; - d.to_f32().ok_or_else(|| SqlError::Parse { - detail: format!( - "vector element decimal '{d}' is out of f32 range" - ), - }) - } - other => Err(SqlError::Parse { - detail: format!( - "vector field must contain numbers, got {other:?}" - ), - }), - }) - .collect(); - vector = Some(floats?); - } - other => { - return Err(SqlError::Parse { - detail: format!( - "vector field '{}' must be an array literal, got {other:?}", - vpc.vector_field - ), - }); - } - } + vector = Some(sql_value_to_vector(&vpc.vector_field, &val)?); } else { payload_fields.insert(col, val); } @@ -89,5 +109,8 @@ pub(crate) fn build_vector_primary_insert_plan( payload_indexes: vpc.payload_indexes.clone(), rows: result_rows, volatile_defaults, + intent, + on_conflict_updates, + primary_key, }]) } diff --git a/nodedb-sql/src/planner/merge/plan.rs b/nodedb-sql/src/planner/merge/plan.rs index 1886ffe38..27a2bf29b 100644 --- a/nodedb-sql/src/planner/merge/plan.rs +++ b/nodedb-sql/src/planner/merge/plan.rs @@ -42,6 +42,7 @@ pub fn plan_merge(stmt: &ast::Statement, catalog: &dyn SqlCatalog) -> Result + { + DataDependent + } Self::InsertSelect { source, .. } | Self::UpdateFrom { source, .. } | Self::Aggregate { input: source, .. } @@ -166,6 +174,9 @@ impl SqlPlan { | Self::ArrayFlush { .. } | Self::ArrayCompact { .. } | Self::VectorPrimaryInsert { .. } + | Self::VectorPrimaryDelete { .. } + | Self::VectorPrimaryTruncate { .. } + | Self::VectorPrimaryUpdate { .. } | Self::CreateIndex { .. } | Self::DropIndex { .. } => Cacheable, } diff --git a/nodedb-sql/src/types/plan/mod.rs b/nodedb-sql/src/types/plan/mod.rs index 56ac8aa6e..fc48d02aa 100644 --- a/nodedb-sql/src/types/plan/mod.rs +++ b/nodedb-sql/src/types/plan/mod.rs @@ -18,7 +18,7 @@ pub use expr_scan::{ referenced_columns, }; pub use merge_types::{MergeClauseKind, MergePlanAction, MergePlanClause}; -pub use row_types::{KvInsertIntent, VectorPrimaryRow, WriteRoute}; +pub use row_types::{KvInsertIntent, VectorPrimaryInsertIntent, VectorPrimaryRow, WriteRoute}; pub use variants::{DistanceMetric, SqlPlan}; pub use vector_opts::{ArrayPrefilter, VectorAnnOptions, VectorQuantization}; pub use volatility_scan::expr_is_volatile; diff --git a/nodedb-sql/src/types/plan/row_types.rs b/nodedb-sql/src/types/plan/row_types.rs index b002fc657..6d0cb2dc8 100644 --- a/nodedb-sql/src/types/plan/row_types.rs +++ b/nodedb-sql/src/types/plan/row_types.rs @@ -36,6 +36,23 @@ pub enum KvInsertIntent { Put, } +/// Row-existence intent carried on `SqlPlan::VectorPrimaryInsert`. +/// +/// A vector-primary row is keyed by its declared primary key. The Data +/// Plane probes the HNSW surrogate map for that key and applies the +/// statement's intent against the result, the same way `KvInsertIntent` +/// drives the key-value hash-index probe. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum VectorPrimaryInsertIntent { + /// Plain `INSERT`: an existing key raises `SQLSTATE 23505`. + Insert, + /// `INSERT ... ON CONFLICT DO NOTHING`: an existing key is a no-op. + InsertIfAbsent, + /// `UPSERT` / `INSERT ... ON CONFLICT (pk) DO UPDATE`: an existing key + /// is replaced (whole row) or patched (`on_conflict_updates`). + Upsert, +} + /// The lowering a row-shaped write takes, carried on `SqlPlan::Insert` and /// `SqlPlan::Upsert`. /// diff --git a/nodedb-sql/src/types/plan/variant_name.rs b/nodedb-sql/src/types/plan/variant_name.rs index 7e516ba88..9e0781ec9 100644 --- a/nodedb-sql/src/types/plan/variant_name.rs +++ b/nodedb-sql/src/types/plan/variant_name.rs @@ -57,6 +57,9 @@ impl SqlPlan { SqlPlan::LateralTopK { .. } => "LateralTopK", SqlPlan::LateralLoop { .. } => "LateralLoop", SqlPlan::VectorPrimaryInsert { .. } => "VectorPrimaryInsert", + SqlPlan::VectorPrimaryDelete { .. } => "VectorPrimaryDelete", + SqlPlan::VectorPrimaryTruncate { .. } => "VectorPrimaryTruncate", + SqlPlan::VectorPrimaryUpdate { .. } => "VectorPrimaryUpdate", SqlPlan::CreateIndex { .. } => "CreateIndex", SqlPlan::DropIndex { .. } => "DropIndex", } diff --git a/nodedb-sql/src/types/plan/variants.rs b/nodedb-sql/src/types/plan/variants.rs index 87ed75533..63a157a2a 100644 --- a/nodedb-sql/src/types/plan/variants.rs +++ b/nodedb-sql/src/types/plan/variants.rs @@ -15,7 +15,7 @@ use crate::types::query::{ }; use super::merge_types::MergePlanClause; -use super::row_types::{KvInsertIntent, VectorPrimaryRow, WriteRoute}; +use super::row_types::{KvInsertIntent, VectorPrimaryInsertIntent, VectorPrimaryRow, WriteRoute}; use super::vector_opts::{ArrayPrefilter, VectorAnnOptions}; /// The top-level plan produced by the SQL planner. @@ -228,6 +228,7 @@ pub enum SqlPlan { }, Truncate { collection: String, + engine: EngineType, restart_identity: bool, }, @@ -722,17 +723,18 @@ pub enum SqlPlan { }, // ── Vector-primary ────────────────────────────────────────────────── - /// INSERT into a vector-primary collection. + /// INSERT / UPSERT into a vector-primary collection. /// - /// Emitted by the planner instead of the generic `Insert` variant when the - /// target collection has `primary = PrimaryEngine::Vector`. The Data Plane - /// routes each row through `VectorOp::DirectUpsert`, bypassing full-document - /// MessagePack encoding. + /// Emitted by the planner instead of the generic `Insert` / `Upsert` + /// variants when the target collection has `primary = + /// PrimaryEngine::Vector`. Each row lowers to one of + /// `VectorOp::DirectInsert` / `DirectInsertIfAbsent` / `DirectUpsert` + /// per `intent`, bypassing full-document MessagePack encoding. VectorPrimaryInsert { collection: String, /// Vector column name (matches `VectorPrimaryConfig::vector_field`). - /// Plumbed to `VectorOp::DirectUpsert` so the Data Plane keys its - /// HNSW index by `(tid, collection, field)` — the same key the SELECT + /// Plumbed to the direct write op so the Data Plane keys its HNSW + /// index by `(tid, collection, field)` — the same key the SELECT /// path uses. field: String, /// Collection-level quantization. Applied via `set_quantization` on @@ -750,6 +752,57 @@ pub enum SqlPlan { /// was built, so caching the lowered tasks would replay one /// execution's value into every later one. volatile_defaults: bool, + /// What an existing primary key means for each row. Mirrors + /// `KvInsert::intent`. + intent: VectorPrimaryInsertIntent, + /// `ON CONFLICT (pk) DO UPDATE SET field = expr` assignments, carried + /// when `intent == Upsert`. Empty means whole-row replace. + on_conflict_updates: Vec<(String, SqlExpr)>, + /// Resolved primary-key column name. See `Insert::primary_key`. + primary_key: Option, + }, + /// DELETE on a vector-primary collection. + /// + /// `target_keys` carries the primary keys when the WHERE clause is a + /// pure primary-key equality (or IN / OR of equalities). Otherwise + /// `filters` is evaluated against every sidecar row on the Data Plane. + VectorPrimaryDelete { + collection: String, + /// Vector column name; keys the HNSW index the rows live in. + field: String, + filters: Vec, + target_keys: Vec, + /// Resolved primary-key column name. See `Insert::primary_key`. + primary_key: Option, + }, + /// TRUNCATE on a vector-primary collection. + /// + /// Removes every row from the HNSW index and its payload sidecar. + VectorPrimaryTruncate { + collection: String, + /// Vector column name; keys the HNSW index the rows live in. + field: String, + restart_identity: bool, + }, + /// UPDATE on a vector-primary collection. + /// + /// `new_vector` is the literal the statement assigns to the vector + /// column, when it assigns one. Every other assignment stays in + /// `assignments` and patches the payload sidecar. + VectorPrimaryUpdate { + collection: String, + /// Vector column name; keys the HNSW index the rows live in. + field: String, + quantization: nodedb_types::VectorQuantization, + storage_dtype: nodedb_types::VectorStorageDtype, + payload_indexes: Vec<(String, nodedb_types::PayloadIndexKind)>, + new_vector: Option>, + assignments: Vec<(String, SqlExpr)>, + filters: Vec, + target_keys: Vec, + returning: bool, + /// Resolved primary-key column name. See `Insert::primary_key`. + primary_key: Option, }, // ── Index DDL ─────────────────────────────────────────────────────── diff --git a/nodedb-sql/src/visitor/plan_visitor/args.rs b/nodedb-sql/src/visitor/plan_visitor/args.rs index 2c89436e7..8fb1846fa 100644 --- a/nodedb-sql/src/visitor/plan_visitor/args.rs +++ b/nodedb-sql/src/visitor/plan_visitor/args.rs @@ -9,7 +9,10 @@ use crate::temporal::TemporalScope; use crate::types::SqlPlan; use crate::types::filter::Filter; -use crate::types::plan::{ArrayPrefilter, MergePlanClause, VectorAnnOptions, WriteRoute}; +use crate::types::plan::{ + ArrayPrefilter, MergePlanClause, VectorAnnOptions, VectorPrimaryInsertIntent, VectorPrimaryRow, + WriteRoute, +}; use crate::types::query::{ AggregateExpr, EngineType, JoinType, Projection, SortKey, SpatialPredicate, WindowSpec, }; @@ -263,3 +266,40 @@ pub struct LateralLoopVisitArgs<'a> { pub outer_row_cap: usize, pub left_join: bool, } + +/// Parameters for [`super::trait_def::PlanVisitor::vector_primary_insert`]. +pub struct VectorPrimaryInsertVisitArgs<'a> { + pub collection: &'a str, + pub field: &'a str, + pub quantization: nodedb_types::VectorQuantization, + pub storage_dtype: nodedb_types::VectorStorageDtype, + pub payload_indexes: &'a [(String, nodedb_types::PayloadIndexKind)], + pub rows: &'a [VectorPrimaryRow], + pub intent: VectorPrimaryInsertIntent, + pub on_conflict_updates: &'a [(String, SqlExpr)], + pub primary_key: Option<&'a str>, +} + +/// Parameters for [`super::trait_def::PlanVisitor::vector_primary_delete`]. +pub struct VectorPrimaryDeleteVisitArgs<'a> { + pub collection: &'a str, + pub field: &'a str, + pub filters: &'a [Filter], + pub target_keys: &'a [SqlValue], + pub primary_key: Option<&'a str>, +} + +/// Parameters for [`super::trait_def::PlanVisitor::vector_primary_update`]. +pub struct VectorPrimaryUpdateVisitArgs<'a> { + pub collection: &'a str, + pub field: &'a str, + pub quantization: nodedb_types::VectorQuantization, + pub storage_dtype: nodedb_types::VectorStorageDtype, + pub payload_indexes: &'a [(String, nodedb_types::PayloadIndexKind)], + pub new_vector: Option<&'a [f32]>, + pub assignments: &'a [(String, SqlExpr)], + pub filters: &'a [Filter], + pub target_keys: &'a [SqlValue], + pub returning: bool, + pub primary_key: Option<&'a str>, +} diff --git a/nodedb-sql/src/visitor/plan_visitor/dispatch.rs b/nodedb-sql/src/visitor/plan_visitor/dispatch.rs index 641295e96..b4fa81e56 100644 --- a/nodedb-sql/src/visitor/plan_visitor/dispatch.rs +++ b/nodedb-sql/src/visitor/plan_visitor/dispatch.rs @@ -184,8 +184,9 @@ pub fn dispatch(visitor: &mut V, plan: &SqlPlan) -> Result visitor.delete(collection, *engine, filters, target_keys), SqlPlan::Truncate { collection, + engine, restart_identity, - } => visitor.truncate(collection, *restart_identity), + } => visitor.truncate(collection, *engine, *restart_identity), SqlPlan::Join { left, right, diff --git a/nodedb-sql/src/visitor/plan_visitor/dispatch_rest.rs b/nodedb-sql/src/visitor/plan_visitor/dispatch_rest.rs index 1a24ee07b..e27b08462 100644 --- a/nodedb-sql/src/visitor/plan_visitor/dispatch_rest.rs +++ b/nodedb-sql/src/visitor/plan_visitor/dispatch_rest.rs @@ -9,6 +9,7 @@ //! `SqlPlan` enum without repeating their field lists. use super::args::{ CreateArrayVisitArgs, LateralLoopVisitArgs, LateralTopKVisitArgs, MergeVisitArgs, + VectorPrimaryDeleteVisitArgs, VectorPrimaryInsertVisitArgs, VectorPrimaryUpdateVisitArgs, }; use super::trait_def::PlanVisitor; use crate::types::SqlPlan; @@ -166,14 +167,63 @@ pub(super) fn dispatch_rest( rows, // Cache eligibility only; lowering does not read it. volatile_defaults: _, - } => visitor.vector_primary_insert( + intent, + on_conflict_updates, + primary_key, + } => visitor.vector_primary_insert(VectorPrimaryInsertVisitArgs { + collection, + field, + quantization: *quantization, + storage_dtype: *storage_dtype, + payload_indexes, + rows, + intent: *intent, + on_conflict_updates, + primary_key: primary_key.as_deref(), + }), + SqlPlan::VectorPrimaryDelete { + collection, + field, + filters, + target_keys, + primary_key, + } => visitor.vector_primary_delete(VectorPrimaryDeleteVisitArgs { + collection, + field, + filters, + target_keys, + primary_key: primary_key.as_deref(), + }), + SqlPlan::VectorPrimaryTruncate { + collection, + field, + restart_identity, + } => visitor.vector_primary_truncate(collection, field, *restart_identity), + SqlPlan::VectorPrimaryUpdate { collection, field, quantization, storage_dtype, payload_indexes, - rows, - ), + new_vector, + assignments, + filters, + target_keys, + returning, + primary_key, + } => visitor.vector_primary_update(VectorPrimaryUpdateVisitArgs { + collection, + field, + quantization: *quantization, + storage_dtype: *storage_dtype, + payload_indexes, + new_vector: new_vector.as_deref(), + assignments, + filters, + target_keys, + returning: *returning, + primary_key: primary_key.as_deref(), + }), SqlPlan::CreateIndex { index_name, collection, diff --git a/nodedb-sql/src/visitor/plan_visitor/trait_def.rs b/nodedb-sql/src/visitor/plan_visitor/trait_def.rs index 6a4ce0b0b..420e7a292 100644 --- a/nodedb-sql/src/visitor/plan_visitor/trait_def.rs +++ b/nodedb-sql/src/visitor/plan_visitor/trait_def.rs @@ -10,20 +10,19 @@ use super::args::{ HybridSearchTripleVisitArgs, HybridSearchVisitArgs, InsertVisitArgs, JoinVisitArgs, LateralLoopVisitArgs, LateralTopKVisitArgs, MergeVisitArgs, RecursiveScanVisitArgs, RecursiveValueVisitArgs, ScanVisitArgs, SpatialScanVisitArgs, SubqueryVisitArgs, - TimeseriesScanVisitArgs, UpdateFromVisitArgs, UpsertVisitArgs, VectorSearchVisitArgs, + TimeseriesScanVisitArgs, UpdateFromVisitArgs, UpsertVisitArgs, VectorPrimaryDeleteVisitArgs, + VectorPrimaryInsertVisitArgs, VectorPrimaryUpdateVisitArgs, VectorSearchVisitArgs, }; use crate::fts_types::FtsQuery; use crate::temporal::TemporalScope; use crate::types::SqlPlan; use crate::types::filter::Filter; -use crate::types::plan::{KvInsertIntent, VectorPrimaryRow}; +use crate::types::plan::KvInsertIntent; use crate::types::query::EngineType; use crate::types_array::{ ArrayBinaryOpAst, ArrayCoordLiteral, ArrayInsertRow, ArrayReducerAst, ArraySliceAst, }; use crate::types_expr::{SqlExpr, SqlValue}; -use nodedb_types::PayloadIndexKind; -use nodedb_types::VectorQuantization; /// Executor parity contract: every [`SqlPlan`] variant must be handled. /// Implement this trait and call [`dispatch`](super::dispatch) to route plans. @@ -121,6 +120,7 @@ pub trait PlanVisitor { fn truncate( &mut self, collection: &str, + engine: EngineType, restart_identity: bool, ) -> Result; @@ -324,13 +324,28 @@ pub trait PlanVisitor { /// Handle [`SqlPlan::VectorPrimaryInsert`]. fn vector_primary_insert( + &mut self, + args: VectorPrimaryInsertVisitArgs<'_>, + ) -> Result; + + /// Handle [`SqlPlan::VectorPrimaryDelete`]. + fn vector_primary_delete( + &mut self, + args: VectorPrimaryDeleteVisitArgs<'_>, + ) -> Result; + + /// Handle [`SqlPlan::VectorPrimaryTruncate`]. + fn vector_primary_truncate( &mut self, collection: &str, field: &str, - quantization: &VectorQuantization, - storage_dtype: &nodedb_types::VectorStorageDtype, - payload_indexes: &[(String, PayloadIndexKind)], - rows: &[VectorPrimaryRow], + restart_identity: bool, + ) -> Result; + + /// Handle [`SqlPlan::VectorPrimaryUpdate`]. + fn vector_primary_update( + &mut self, + args: VectorPrimaryUpdateVisitArgs<'_>, ) -> Result; /// Handle [`SqlPlan::CreateIndex`]. diff --git a/nodedb-sql/tests/sql_suite/cases/mod.rs b/nodedb-sql/tests/sql_suite/cases/mod.rs index b0914c67c..f3c300cfa 100644 --- a/nodedb-sql/tests/sql_suite/cases/mod.rs +++ b/nodedb-sql/tests/sql_suite/cases/mod.rs @@ -6,3 +6,4 @@ mod on_conflict_update_range_check; mod point_get_operand_order; mod positional_insert_column_binding; mod schema_qualified_rejection; +mod truncate_engine_routing; diff --git a/nodedb-sql/tests/sql_suite/cases/truncate_engine_routing.rs b/nodedb-sql/tests/sql_suite/cases/truncate_engine_routing.rs new file mode 100644 index 000000000..fd99a4253 --- /dev/null +++ b/nodedb-sql/tests/sql_suite/cases/truncate_engine_routing.rs @@ -0,0 +1,164 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `TRUNCATE ` resolves the collection through the catalog and routes +//! through `EngineRules::plan_truncate`: every table engine lowers to +//! `SqlPlan::Truncate` tagged with its own `EngineType`, a vector-primary +//! collection lowers to `SqlPlan::VectorPrimaryTruncate`, an array refuses +//! with a typed error naming `DROP ARRAY`, and an unknown name is +//! `UnknownTable`. + +use nodedb_sql::types::{CollectionInfo, EngineType}; +use nodedb_sql::{SqlCatalog, SqlCatalogError, SqlError, SqlPlan, plan_sql}; +use nodedb_types::DatabaseId; + +struct Catalog; + +fn info(name: &str, engine: EngineType) -> CollectionInfo { + CollectionInfo { + name: name.into(), + engine, + columns: Vec::new(), + primary_key: Some("id".into()), + has_auto_tier: false, + indexes: Vec::new(), + bitemporal: false, + primary: nodedb_types::PrimaryEngine::Document, + vector_primary: None, + partition_strategy: nodedb_types::PartitionStrategy::CollectionHomed, + open_schema: CollectionInfo::open_schema_for(engine), + } +} + +impl SqlCatalog for Catalog { + fn get_collection( + &self, + _: DatabaseId, + name: &str, + ) -> std::result::Result, SqlCatalogError> { + let info = match name { + "docs" => Some(info(name, EngineType::DocumentSchemaless)), + "strict" => Some(info(name, EngineType::DocumentStrict)), + "kv" => Some(info(name, EngineType::KeyValue)), + "cols" => Some(info(name, EngineType::Columnar)), + "ts" => Some(info(name, EngineType::Timeseries)), + "geo" => Some(info(name, EngineType::Spatial)), + "vecs" => { + let mut i = info(name, EngineType::DocumentSchemaless); + i.primary = nodedb_types::PrimaryEngine::Vector; + i.vector_primary = Some(nodedb_types::VectorPrimaryConfig { + vector_field: "emb".into(), + dim: 3, + ..nodedb_types::VectorPrimaryConfig::default() + }); + Some(i) + } + _ => None, + }; + Ok(info) + } + + fn lookup_array(&self, _name: &str) -> Option { + None + } + + fn array_exists(&self, name: &str) -> bool { + name == "arr" + } +} + +fn plan_one(sql: &str) -> SqlPlan { + let mut plans = plan_sql(sql, &Catalog).expect("planning must succeed"); + assert_eq!(plans.len(), 1, "expected exactly one plan for: {sql}"); + plans.pop().expect("one plan") +} + +#[test] +fn truncate_tags_each_table_engine() { + for (name, engine) in [ + ("docs", EngineType::DocumentSchemaless), + ("strict", EngineType::DocumentStrict), + ("kv", EngineType::KeyValue), + ("cols", EngineType::Columnar), + ("ts", EngineType::Timeseries), + ("geo", EngineType::Spatial), + ] { + match plan_one(&format!("TRUNCATE {name}")) { + SqlPlan::Truncate { + collection, + engine: got, + restart_identity, + } => { + assert_eq!(collection, name); + assert_eq!(got, engine); + assert!(!restart_identity); + } + other => panic!("expected SqlPlan::Truncate for {name}, got {other:?}"), + } + } +} + +#[test] +fn truncate_restart_identity_is_carried() { + match plan_one("TRUNCATE kv RESTART IDENTITY") { + SqlPlan::Truncate { + restart_identity, .. + } => assert!(restart_identity), + other => panic!("expected SqlPlan::Truncate, got {other:?}"), + } +} + +#[test] +fn truncate_vector_primary_lowers_to_its_own_plan() { + match plan_one("TRUNCATE vecs RESTART IDENTITY") { + SqlPlan::VectorPrimaryTruncate { + collection, + field, + restart_identity, + } => { + assert_eq!(collection, "vecs"); + assert_eq!(field, "emb"); + assert!(restart_identity); + } + other => panic!("expected SqlPlan::VectorPrimaryTruncate, got {other:?}"), + } +} + +#[test] +fn truncate_array_is_refused_naming_drop_array() { + let err = plan_sql("TRUNCATE arr", &Catalog).expect_err("array truncate must refuse"); + match err { + SqlError::Unsupported { detail } => { + assert!(detail.contains("DROP ARRAY "), "{detail}"); + assert!( + detail.contains("DELETE FROM ARRAY WHERE COORDS IN (...)"), + "{detail}" + ); + } + other => panic!("expected SqlError::Unsupported, got {other:?}"), + } +} + +#[test] +fn truncate_unknown_collection_is_unknown_table() { + let err = plan_sql("TRUNCATE nope", &Catalog).expect_err("unknown table must refuse"); + assert!( + matches!(err, SqlError::UnknownTable { ref name } if name == "nope"), + "got {err:?}" + ); +} + +#[test] +fn truncate_many_plans_one_per_table() { + let plans = plan_sql("TRUNCATE docs, kv", &Catalog).expect("planning must succeed"); + let engines: Vec = plans + .iter() + .map(|p| match p { + SqlPlan::Truncate { engine, .. } => *engine, + other => panic!("expected SqlPlan::Truncate, got {other:?}"), + }) + .collect(); + assert_eq!( + engines, + vec![EngineType::DocumentSchemaless, EngineType::KeyValue] + ); +} diff --git a/nodedb-test-support/src/cluster_harness/node/inspect/catalog.rs b/nodedb-test-support/src/cluster_harness/node/inspect/catalog.rs index 3cbf68a94..2b1bd3276 100644 --- a/nodedb-test-support/src/cluster_harness/node/inspect/catalog.rs +++ b/nodedb-test-support/src/cluster_harness/node/inspect/catalog.rs @@ -51,6 +51,24 @@ impl TestClusterNode { .map(|(_, current, _)| current) } + /// The value the next `nextval` call on this sequence returns: + /// `current_value + increment`. Holds whether or not the sequence has + /// been called yet — a restart stores `value - increment` with + /// `called = false` precisely so this formula gives `value` right + /// after `RESTART WITH value`, matching Postgres semantics. + pub fn sequence_next_value(&self, tenant_id: u64, name: &str) -> Option { + let db = nodedb_types::DatabaseId::DEFAULT.as_u64(); + let current = self + .shared + .sequence_registry + .list(db, tenant_id) + .into_iter() + .find(|(n, _, _)| n == name) + .map(|(_, current, _)| current)?; + let def = self.shared.sequence_registry.get_def(db, tenant_id, name)?; + Some(current + def.increment) + } + /// Check whether a trigger with the given name exists in this /// node's in-memory trigger registry. pub fn has_trigger(&self, tenant_id: u64, name: &str) -> bool { diff --git a/nodedb-test-support/src/cluster_harness/node/mod.rs b/nodedb-test-support/src/cluster_harness/node/mod.rs index 3ec6057a1..f2bbfa68b 100644 --- a/nodedb-test-support/src/cluster_harness/node/mod.rs +++ b/nodedb-test-support/src/cluster_harness/node/mod.rs @@ -7,5 +7,6 @@ pub mod graph; pub mod inspect; pub mod lifecycle; pub mod native_client; +pub mod raw_pgwire; pub use lifecycle::TestClusterNode; diff --git a/nodedb-test-support/src/cluster_harness/node/raw_pgwire.rs b/nodedb-test-support/src/cluster_harness/node/raw_pgwire.rs new file mode 100644 index 000000000..3148f4eda --- /dev/null +++ b/nodedb-test-support/src/cluster_harness/node/raw_pgwire.rs @@ -0,0 +1,22 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Raw pgwire connection construction for [`TestClusterNode`]. +//! +//! `tokio_postgres` decodes `CommandComplete` down to a row count, so a test +//! that asserts the tag verb (`INSERT 0 1` vs `OK`) reads the wire directly. + +use crate::pgwire_harness::raw_pgwire::RawPgConn; + +use super::lifecycle::{HARNESS_SUPERUSER, TestClusterNode}; + +/// Database the pre-wired `client` field connects to. +const HARNESS_DATABASE: &str = "default"; + +impl TestClusterNode { + /// A raw simple-query pgwire connection to this node, authenticated as + /// the harness's bootstrapped trust superuser on the same database the + /// `client` field uses. + pub async fn raw_pgwire(&self) -> RawPgConn { + RawPgConn::connect(self.pg_addr.port(), HARNESS_SUPERUSER, HARNESS_DATABASE).await + } +} diff --git a/nodedb-test-support/src/native_harness/frames.rs b/nodedb-test-support/src/native_harness/frames.rs index 5b88cf251..0ccbd8588 100644 --- a/nodedb-test-support/src/native_harness/frames.rs +++ b/nodedb-test-support/src/native_harness/frames.rs @@ -189,6 +189,7 @@ pub async fn send_request( Some(aggregate.rows_affected.unwrap_or(0) + rows_affected); } aggregate.watermark_lsn = aggregate.watermark_lsn.max(response.watermark_lsn); + aggregate.command = response.command; aggregate.status = response.status; aggregate.error = response.error; aggregate.auth = response.auth; diff --git a/nodedb-test-support/src/pgwire_harness/mod.rs b/nodedb-test-support/src/pgwire_harness/mod.rs index 9eda43576..f38c81cd9 100644 --- a/nodedb-test-support/src/pgwire_harness/mod.rs +++ b/nodedb-test-support/src/pgwire_harness/mod.rs @@ -7,6 +7,7 @@ mod multicore; mod query; +pub mod raw_pgwire; mod restart; mod start; mod support; diff --git a/nodedb-test-support/src/pgwire_harness/raw_pgwire.rs b/nodedb-test-support/src/pgwire_harness/raw_pgwire.rs new file mode 100644 index 000000000..0d7d70a4b --- /dev/null +++ b/nodedb-test-support/src/pgwire_harness/raw_pgwire.rs @@ -0,0 +1,116 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Raw PostgreSQL v3 wire protocol connection, for tests that need bytes +//! the `tokio_postgres` client does not expose (command tag words, column +//! type OIDs) rather than its decoded `SimpleQueryMessage`s. + +use tokio::io::{AsyncReadExt, AsyncWriteExt}; +use tokio::net::TcpStream; + +/// PostgreSQL protocol version 3.0, sent in every `StartupMessage`. +const PROTOCOL_V3: i32 = 196_608; // 0x0003_0000 + +/// A raw simple-query pgwire connection: startup handshake done, +/// `ReadyForQuery` consumed. +pub struct RawPgConn { + stream: TcpStream, +} + +impl RawPgConn { + /// Connect, complete a trust-mode `StartupMessage` for `user`/`database`, + /// and drain replies through the first `ReadyForQuery`. + pub async fn connect(port: u16, user: &str, database: &str) -> Self { + let stream = TcpStream::connect(("127.0.0.1", port)) + .await + .expect("connect to pgwire port"); + + let mut params = Vec::new(); + params.extend_from_slice(b"user\0"); + params.extend_from_slice(user.as_bytes()); + params.push(0); + params.extend_from_slice(b"database\0"); + params.extend_from_slice(database.as_bytes()); + params.push(0); + params.push(0); // parameter-list terminator + + let total = 4 + 4 + params.len(); + let mut startup = Vec::new(); + startup.extend_from_slice(&(total as i32).to_be_bytes()); + startup.extend_from_slice(&PROTOCOL_V3.to_be_bytes()); + startup.extend_from_slice(¶ms); + + let mut conn = Self { stream }; + conn.stream.write_all(&startup).await.expect("send startup"); + + // Trust mode sends AuthenticationOk, ParameterStatus*, and + // BackendKeyData before the first ReadyForQuery. + loop { + let (tag, body) = conn.read_message().await; + match tag { + b'Z' => break, + b'E' => panic!("startup error: {}", String::from_utf8_lossy(&body)), + _ => {} + } + } + conn + } + + /// Read exactly one backend message: a 1-byte type tag followed by an + /// i32 length (which counts itself but not the tag). Returns + /// `(tag, body)` where `body` excludes the 4-byte length prefix. + pub async fn read_message(&mut self) -> (u8, Vec) { + let mut tag = [0u8; 1]; + self.stream + .read_exact(&mut tag) + .await + .expect("read message tag"); + let mut len_buf = [0u8; 4]; + self.stream + .read_exact(&mut len_buf) + .await + .expect("read message length"); + let len = i32::from_be_bytes(len_buf) as usize; + let mut body = vec![0u8; len - 4]; + self.stream + .read_exact(&mut body) + .await + .expect("read message body"); + (tag[0], body) + } + + /// Send a Simple Query (`Q`) and read every reply up to (not including) + /// the terminating `ReadyForQuery`. A backend `ErrorResponse` panics — + /// callers that expect the query to fail must not route it through + /// here. + pub async fn simple_query(&mut self, sql: &str) -> Vec<(u8, Vec)> { + let mut qbody = sql.as_bytes().to_vec(); + qbody.push(0); + let mut qmsg = vec![b'Q']; + qmsg.extend_from_slice(&((4 + qbody.len()) as i32).to_be_bytes()); + qmsg.extend_from_slice(&qbody); + self.stream.write_all(&qmsg).await.expect("send query"); + + let mut messages = Vec::new(); + loop { + let (tag, body) = self.read_message().await; + match tag { + b'Z' => break, + b'E' => panic!("query error: {}", String::from_utf8_lossy(&body)), + _ => messages.push((tag, body)), + } + } + messages + } +} + +/// `CommandComplete` (`C`) tag strings in `messages`, in wire order. +pub fn command_tags(messages: &[(u8, Vec)]) -> Vec { + messages + .iter() + .filter(|(tag, _)| *tag == b'C') + .map(|(_, body)| { + let end = body.iter().position(|b| *b == 0).unwrap_or(body.len()); + String::from_utf8_lossy(&body[..end]).into_owned() + }) + .collect() +} diff --git a/nodedb-types/src/columnar/mod.rs b/nodedb-types/src/columnar/mod.rs index b7741b8c4..f6534228b 100644 --- a/nodedb-types/src/columnar/mod.rs +++ b/nodedb-types/src/columnar/mod.rs @@ -10,6 +10,7 @@ pub mod int_width; pub mod profile; pub mod resolved_dml_wal_record; pub mod schema; +pub mod truncate_wal_record; pub mod wal_record; pub use column_def::{ColumnDef, ColumnModifier}; @@ -25,4 +26,5 @@ pub use schema::{ BITEMPORAL_RESERVED_COLUMNS, BITEMPORAL_SYSTEM_FROM, BITEMPORAL_VALID_FROM, BITEMPORAL_VALID_UNTIL, ColumnarSchema, DroppedColumn, SchemaError, SchemaOps, StrictSchema, }; +pub use truncate_wal_record::ColumnarTruncateWalRecord; pub use wal_record::ColumnarWalRecord; diff --git a/nodedb-types/src/columnar/truncate_wal_record.rs b/nodedb-types/src/columnar/truncate_wal_record.rs new file mode 100644 index 000000000..508a62914 --- /dev/null +++ b/nodedb-types/src/columnar/truncate_wal_record.rs @@ -0,0 +1,43 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Columnar-family `TRUNCATE` WAL record payload. +//! +//! Rides `RecordType::ColumnarTruncate` (columnar and spatial collections) +//! and `RecordType::TimeseriesTruncate` (timeseries collections). The record +//! type says which engine clears the collection, so the payload carries the +//! collection name only; `restart_identity` is a Control-Plane sequence +//! concern applied after dispatch and never enters the Data-Plane record. + +use serde::{Deserialize, Serialize}; + +/// Map-encoded columnar-family truncate WAL record. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +#[msgpack(map)] +pub struct ColumnarTruncateWalRecord { + /// Target collection name. + pub collection: String, +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn round_trips() { + let rec = ColumnarTruncateWalRecord { + collection: "metrics".to_string(), + }; + let bytes = zerompk::to_msgpack_vec(&rec).expect("encode"); + let back: ColumnarTruncateWalRecord = zerompk::from_msgpack(&bytes).expect("decode"); + assert_eq!(back, rec); + } +} diff --git a/nodedb-types/src/lib.rs b/nodedb-types/src/lib.rs index 1342db40e..68f14647e 100644 --- a/nodedb-types/src/lib.rs +++ b/nodedb-types/src/lib.rs @@ -140,7 +140,7 @@ pub use temporal::{ pub use text_search::{Bm25Params, QueryMode, TextSearchParams}; pub use trace::{SpanId, TraceId}; pub use typeguard::TypeGuardFieldDef; -pub use value::Value; +pub use value::{NotScalar, Value, scalar_to_raw_bytes}; pub use vector_ann::{VectorAnnOptions, VectorQuantization}; pub use vector_dtype::VectorStorageDtype; pub use vector_index_params::StoredVectorIndexParams; diff --git a/nodedb-types/src/protocol/frames.rs b/nodedb-types/src/protocol/frames.rs index 1a7278226..233cd1cae 100644 --- a/nodedb-types/src/protocol/frames.rs +++ b/nodedb-types/src/protocol/frames.rs @@ -73,6 +73,12 @@ pub struct NativeResponse { /// Number of rows affected (for writes). #[serde(skip_serializing_if = "Option::is_none")] pub rows_affected: Option, + /// The statement's command verb (`INSERT`, `UPDATE`, `DELETE`, `UPSERT`, + /// `MERGE`, `TRUNCATE`, ...). `None` when the statement produced no DML + /// outcome. + #[serde(default, skip_serializing_if = "Option::is_none")] + #[msgpack(default)] + pub command: Option, /// WAL LSN watermark at time of computation. pub watermark_lsn: u64, /// Error details (if status == Error). @@ -136,6 +142,7 @@ impl NativeResponse { columns: None, rows: None, rows_affected: None, + command: None, watermark_lsn: 0, error: None, auth: None, @@ -151,6 +158,7 @@ impl NativeResponse { columns: Some(qr.columns), rows: Some(qr.rows), rows_affected: Some(qr.rows_affected), + command: qr.command, watermark_lsn: lsn, error: None, auth: None, @@ -182,6 +190,7 @@ impl NativeResponse { columns: None, rows: None, rows_affected: None, + command: None, watermark_lsn: 0, error: Some(ErrorPayload { code: code.into(), @@ -201,6 +210,7 @@ impl NativeResponse { columns: None, rows: None, rows_affected: None, + command: None, watermark_lsn: 0, error: None, auth: Some(AuthResponse { @@ -219,6 +229,7 @@ impl NativeResponse { columns: Some(vec!["status".into()]), rows: Some(vec![vec![Value::String(message.into())]]), rows_affected: Some(1), + command: None, watermark_lsn: 0, error: None, auth: None, @@ -259,6 +270,7 @@ mod tests { Value::String("Alice".into()), ]], rows_affected: 0, + command: None, }; let r = NativeResponse::from_query_result(5, qr, 100); assert_eq!(r.seq, 5); @@ -298,6 +310,7 @@ mod tests { columns: vec!["x".into()], rows: vec![vec![Value::Integer(42)]], rows_affected: 0, + command: None, }, 99, ); diff --git a/nodedb-types/src/result.rs b/nodedb-types/src/result.rs index 29e84d290..0185a4ad6 100644 --- a/nodedb-types/src/result.rs +++ b/nodedb-types/src/result.rs @@ -66,6 +66,11 @@ pub struct QueryResult { pub rows: Vec>, /// Number of rows affected (for INSERT/UPDATE/DELETE). pub rows_affected: u64, + /// The statement's command verb (`INSERT`, `UPDATE`, `DELETE`, `UPSERT`, + /// `MERGE`, `TRUNCATE`, ...). `None` when the statement produced no DML + /// outcome. + #[serde(default)] + pub command: Option, } impl QueryResult { @@ -75,6 +80,7 @@ impl QueryResult { columns: Vec::new(), rows: Vec::new(), rows_affected: 0, + command: None, } } @@ -167,6 +173,7 @@ mod tests { columns: vec!["name".into(), "age".into()], rows: vec![vec![Value::String("Alice".into()), Value::Integer(30)]], rows_affected: 0, + command: None, }; let row = qr.row_as_map(0).unwrap(); assert_eq!(row["name"].as_str(), Some("Alice")); diff --git a/nodedb-types/src/value/mod.rs b/nodedb-types/src/value/mod.rs index 6dec3e618..621113b6c 100644 --- a/nodedb-types/src/value/mod.rs +++ b/nodedb-types/src/value/mod.rs @@ -6,6 +6,8 @@ pub mod core; pub mod display; pub mod json; pub mod msgpack; +pub mod raw_bytes; pub mod sql_literal; pub use core::Value; +pub use raw_bytes::{NotScalar, scalar_to_raw_bytes}; diff --git a/nodedb-types/src/value/raw_bytes.rs b/nodedb-types/src/value/raw_bytes.rs new file mode 100644 index 000000000..747af7102 --- /dev/null +++ b/nodedb-types/src/value/raw_bytes.rs @@ -0,0 +1,88 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The one rule for turning a scalar into a raw KV body. +//! +//! A KV row written through the single-`value` SQL column, or through RESP +//! `SET`, stores its scalar as raw bytes rather than a msgpack map. The SQL +//! lowering and every KV read-modify-write encode through this function, so +//! a merged row re-encodes byte-for-byte as a fresh insert of the same +//! scalar. + +use crate::value::core::Value; + +/// The value is not a scalar, so it has no raw KV body form. +#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)] +#[error("{kind} has no raw KV body form; only a scalar does")] +pub struct NotScalar { + /// `Value::type_name()` of the offending value. + pub kind: &'static str, +} + +/// Raw KV body bytes for a scalar. +/// +/// - string, UUID, ULID: the UTF-8 bytes +/// - bytes: verbatim +/// - integer, float, decimal, bool: decimal / `true` / `false` text +/// - timestamps: ISO-8601 text +/// - null: empty +pub fn scalar_to_raw_bytes(value: &Value) -> Result, NotScalar> { + match value { + Value::Null => Ok(Vec::new()), + Value::Bool(b) => Ok(b.to_string().into_bytes()), + Value::Integer(i) => Ok(i.to_string().into_bytes()), + Value::Float(f) => Ok(f.to_string().into_bytes()), + Value::Decimal(d) => Ok(d.to_string().into_bytes()), + Value::String(s) | Value::Uuid(s) | Value::Ulid(s) => Ok(s.as_bytes().to_vec()), + Value::Bytes(b) => Ok(b.clone()), + Value::DateTime(dt) | Value::NaiveDateTime(dt) => Ok(dt.to_iso8601().into_bytes()), + Value::Duration(_) + | Value::Array(_) + | Value::Object(_) + | Value::Set(_) + | Value::Regex(_) + | Value::Geometry(_) + | Value::Range { .. } + | Value::Record { .. } + | Value::ArrayCell(_) + | Value::Vector(_) => Err(NotScalar { + kind: value.type_name(), + }), + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::collections::HashMap; + + #[test] + fn string_is_its_utf8_bytes() { + assert_eq!( + scalar_to_raw_bytes(&Value::String("v1".into())).unwrap(), + b"v1".to_vec() + ); + } + + #[test] + fn integer_is_decimal_text() { + assert_eq!( + scalar_to_raw_bytes(&Value::Integer(-42)).unwrap(), + b"-42".to_vec() + ); + } + + #[test] + fn bool_and_null() { + assert_eq!( + scalar_to_raw_bytes(&Value::Bool(true)).unwrap(), + b"true".to_vec() + ); + assert!(scalar_to_raw_bytes(&Value::Null).unwrap().is_empty()); + } + + #[test] + fn object_is_not_scalar() { + let err = scalar_to_raw_bytes(&Value::Object(HashMap::new())).unwrap_err(); + assert_eq!(err.kind, Value::Object(HashMap::new()).type_name()); + } +} diff --git a/nodedb-vector/src/collection/lifecycle_compact.rs b/nodedb-vector/src/collection/lifecycle_compact.rs index 2571fa694..7d7dc13bc 100644 --- a/nodedb-vector/src/collection/lifecycle_compact.rs +++ b/nodedb-vector/src/collection/lifecycle_compact.rs @@ -1,15 +1,54 @@ // SPDX-License-Identifier: Apache-2.0 -//! Compact and snapshot operations for `VectorCollection`. +//! Truncate, compact, and snapshot operations for `VectorCollection`. use nodedb_types::Surrogate; use super::lifecycle::VectorCollection; +use crate::flat::FlatIndex; /// One exported vector: global node id, full-precision data, optional surrogate. pub type ExportedVector = (u32, Vec, Option); impl VectorCollection { + /// Drop every vector, sealed segment, in-flight build, surrogate binding, + /// and payload bitmap row. Returns the number of live vectors dropped. + /// + /// Configuration survives: dimension, HNSW params, index config, + /// quantization, seal threshold, memory budget, and the registered + /// payload index fields. `next_id` and `next_segment_id` keep counting + /// so a build completion for a segment sealed before the truncate finds + /// no matching entry in `building` and is ignored, and an mmap file name + /// is never reused. The mmap file of each dropped sealed segment is + /// removed from disk. + pub fn truncate(&mut self) -> usize { + let dropped = self.live_count(); + self.growing = FlatIndex::new(self.dim, self.params.metric); + self.growing_base_id = self.next_id; + for seg in self.sealed.drain(..) { + let mmap_path = seg.mmap_vectors.as_ref().map(|m| m.path().to_path_buf()); + // Unmap before the file goes. + drop(seg); + if let Some(path) = mmap_path + && let Err(e) = std::fs::remove_file(&path) + { + tracing::warn!( + path = %path.display(), + error = %e, + "vector truncate: mmap segment file not removed" + ); + } + } + self.building.clear(); + self.mmap_segment_count = 0; + self.surrogate_map.clear(); + self.surrogate_to_local.clear(); + self.multi_doc_map.clear(); + self.codec_dispatch = None; + self.payload.clear_rows(); + dropped + } + /// Compact sealed segments by removing tombstoned nodes. /// /// Rewrites `surrogate_map` and `multi_doc_map` for every sealed @@ -116,3 +155,57 @@ impl VectorCollection { Ok(result) } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::hnsw::HnswParams; + use nodedb_types::{PayloadIndexKind, Value}; + use std::collections::HashMap; + + #[test] + fn truncate_drops_every_row_and_keeps_config() { + let mut coll = VectorCollection::with_seal_threshold(2, HnswParams::default(), 2); + coll.payload.add_index("owner", PayloadIndexKind::Equality); + let mut fields = HashMap::new(); + fields.insert("owner".to_string(), Value::String("a".into())); + for (i, v) in [[1.0, 0.0], [0.0, 1.0], [1.0, 1.0]].into_iter().enumerate() { + let s = Surrogate::new(i as u32 + 1); + let id = coll.insert_with_surrogate(v.to_vec(), s); + coll.payload.insert_row(id, &fields); + } + assert!( + coll.seal("k").is_some(), + "threshold reached, growing sealed" + ); + assert_eq!(coll.live_count(), 3); + let next_before = coll.next_id; + + assert_eq!(coll.truncate(), 3); + assert_eq!(coll.live_count(), 0); + assert!(coll.surrogate_to_local.is_empty()); + assert!(coll.surrogate_map.is_empty()); + assert!(coll.building.is_empty()); + assert!(coll.sealed.is_empty()); + assert_eq!(coll.dim(), 2); + assert_eq!(coll.next_id, next_before, "ids stay monotonic"); + assert_eq!(coll.growing_base_id, next_before); + assert!( + coll.payload.field_names().any(|f| f == "owner"), + "registered payload index survives" + ); + let hits = coll + .payload + .pre_filter(&super::super::payload_index::FilterPredicate::Eq { + field: "owner".to_string(), + value: Value::String("a".into()), + }) + .expect("owner is indexed"); + assert!(hits.is_empty(), "payload rows cleared"); + + let s = Surrogate::new(42); + let id = coll.insert_with_surrogate(vec![0.5, 0.5], s); + assert_eq!(coll.local_for_surrogate(s), Some(id)); + assert_eq!(coll.live_count(), 1); + } +} diff --git a/nodedb-vector/src/collection/lifecycle_insert_ops.rs b/nodedb-vector/src/collection/lifecycle_insert_ops.rs index 8029d51ba..f962f1508 100644 --- a/nodedb-vector/src/collection/lifecycle_insert_ops.rs +++ b/nodedb-vector/src/collection/lifecycle_insert_ops.rs @@ -18,7 +18,19 @@ impl VectorCollection { /// Insert a vector with an associated surrogate. The surrogate is /// allocated by the Control Plane before the call; the engine only /// stores the binding. + /// + /// One surrogate names one live node. A node already bound to + /// `surrogate` is soft-deleted before the new one is bound, so a + /// re-insert never leaves an unreachable node scoring in searches. + /// The caller owns the payload bitmap entries of the old node and + /// removes them with [`Self::local_for_surrogate`] before this call. pub fn insert_with_surrogate(&mut self, vector: Vec, surrogate: Surrogate) -> u32 { + if surrogate != Surrogate::ZERO + && let Some(old) = self.surrogate_to_local.get(&surrogate).copied() + { + self.delete_inner(old); + self.surrogate_map.remove(&old); + } let id = self.insert(vector); if surrogate != Surrogate::ZERO { self.surrogate_map.insert(id, surrogate); @@ -109,6 +121,43 @@ impl VectorCollection { false } + /// The live FP32 vector stored under global `id`, whichever segment + /// holds it. `None` for an unknown or soft-deleted id. + pub fn vector_for_id(&self, id: u32) -> Option> { + if id >= self.growing_base_id { + let local = id - self.growing_base_id; + if (local as usize) < self.growing.len() { + return self.growing.get_vector(local).map(<[f32]>::to_vec); + } + } + for seg in &self.sealed { + if id >= seg.base_id { + let local = id - seg.base_id; + if (local as usize) < seg.index.len() { + if seg.index.is_deleted(local) { + return None; + } + return sealed_vector(seg, local); + } + } + } + for seg in &self.building { + if id >= seg.base_id { + let local = id - seg.base_id; + if (local as usize) < seg.flat.len() { + return seg.flat.get_vector(local).map(<[f32]>::to_vec); + } + } + } + None + } + + /// The live FP32 vector bound to `surrogate`, if any. + pub fn vector_for_surrogate(&self, surrogate: Surrogate) -> Option> { + self.local_for_surrogate(surrogate) + .and_then(|id| self.vector_for_id(id)) + } + /// Soft-delete a vector by surrogate. pub fn delete_by_surrogate(&mut self, surrogate: Surrogate) -> bool { let Some(global_id) = self.surrogate_to_local.get(&surrogate).copied() else { @@ -151,3 +200,80 @@ impl VectorCollection { false } } + +/// The FP32 vector at `local` in a sealed segment: the mmap tier when the +/// segment lives there, else the HNSW node (decoded from a narrow dtype or +/// fetched from the segment backing when the node holds no local copy). +fn sealed_vector(seg: &super::segment::SealedSegment, local: u32) -> Option> { + if let Some(mmap) = &seg.mmap_vectors { + return mmap.get_vector(local).map(<[f32]>::to_vec); + } + #[cfg(not(target_arch = "wasm32"))] + { + seg.index + .get_vector_or_backing(local) + .map(std::borrow::Cow::into_owned) + } + #[cfg(target_arch = "wasm32")] + { + seg.index.get_vector(local).map(<[f32]>::to_vec) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::hnsw::HnswParams; + + fn collection() -> VectorCollection { + VectorCollection::new(2, HnswParams::default()) + } + + #[test] + fn re_insert_under_the_same_surrogate_leaves_one_live_node() { + let mut coll = collection(); + let s = Surrogate::new(7); + let first = coll.insert_with_surrogate(vec![1.0, 0.0], s); + let second = coll.insert_with_surrogate(vec![0.0, 1.0], s); + assert_ne!(first, second); + assert_eq!(coll.live_count(), 1, "the old node must be tombstoned"); + assert_eq!(coll.local_for_surrogate(s), Some(second)); + assert_eq!(coll.get_surrogate(first), None); + assert_eq!(coll.get_surrogate(second), Some(s)); + } + + #[test] + fn delete_then_insert_under_the_same_surrogate_leaves_one_live_node() { + let mut coll = collection(); + let s = Surrogate::new(9); + let first = coll.insert_with_surrogate(vec![1.0, 0.0], s); + assert!(coll.delete_by_surrogate(s)); + assert_eq!(coll.local_for_surrogate(s), None); + let second = coll.insert_with_surrogate(vec![0.0, 1.0], s); + assert_ne!(first, second); + assert_eq!(coll.live_count(), 1); + assert_eq!(coll.local_for_surrogate(s), Some(second)); + assert!(!coll.delete(first), "the first node is already gone"); + } + + #[test] + fn delete_by_surrogate_is_idempotent() { + let mut coll = collection(); + let s = Surrogate::new(3); + coll.insert_with_surrogate(vec![1.0, 0.0], s); + assert!(coll.delete_by_surrogate(s)); + assert!(!coll.delete_by_surrogate(s)); + assert_eq!(coll.live_count(), 0); + } + + #[test] + fn vector_for_surrogate_reads_the_growing_segment_and_hides_deletes() { + let mut coll = collection(); + let s = Surrogate::new(11); + coll.insert_with_surrogate(vec![0.5, 0.25], s); + assert_eq!(coll.vector_for_surrogate(s), Some(vec![0.5, 0.25])); + assert!(coll.delete_by_surrogate(s)); + assert_eq!(coll.vector_for_surrogate(s), None); + assert_eq!(coll.vector_for_id(999), None); + } +} diff --git a/nodedb-vector/src/collection/payload_index.rs b/nodedb-vector/src/collection/payload_index.rs index b461211d6..c7daeaba1 100644 --- a/nodedb-vector/src/collection/payload_index.rs +++ b/nodedb-vector/src/collection/payload_index.rs @@ -148,6 +148,13 @@ impl PayloadIndexBitmaps { } } + fn clear(&mut self) { + match self { + Self::Equality(m) => m.clear(), + Self::Range(m) => m.clear(), + } + } + fn iter(&self) -> Box + '_> { match self { Self::Equality(m) => Box::new(m.iter()), @@ -288,6 +295,14 @@ impl PayloadIndexSet { } } + /// Drop every node from every index. The registered fields and their + /// kinds stay, so the next `insert_row` indexes the same columns. + pub fn clear_rows(&mut self) { + for idx in self.indexes.values_mut() { + idx.bitmaps.clear(); + } + } + /// Evaluate a filter predicate against the bitmap indexes. /// /// Returns `Some(bitmap)` when the predicate is **fully covered** by diff --git a/nodedb-wal/src/record/types.rs b/nodedb-wal/src/record/types.rs index 2dfaa4b28..89a8af891 100644 --- a/nodedb-wal/src/record/types.rs +++ b/nodedb-wal/src/record/types.rs @@ -49,6 +49,46 @@ pub enum RecordType { /// Required: skipping on replay loses an acknowledged vector-primary write. VectorDirectUpsert = 13 | 0x8000, + /// Vector engine: delete rows of a vector-primary collection by surrogate + /// or by sidecar predicate. + /// + /// Required: skipping on replay resurrects deleted vector-primary rows. + VectorDirectDelete = 19 | 0x8000, + + /// Vector engine: update rows of a vector-primary collection — a new + /// vector, a payload patch, or both — by surrogate or by sidecar predicate. + /// + /// Required: skipping on replay loses an acknowledged vector-primary write. + VectorDirectUpdate = 23 | 0x8000, + + /// Vector engine: apply the row mutations a governed vector-primary + /// write resolved to — deletes, rewrites, and whole-row upserts, each + /// naming its surrogate and carrying its full stored image. + /// + /// Required: skipping on replay loses an acknowledged vector-primary write. + VectorResolvedDirectWrite = 24 | 0x8000, + + /// Vector engine: remove every row of a vector-primary collection + /// (`TRUNCATE`). + /// + /// Required: skipping on replay resurrects every truncated row, because + /// the `VectorDirectUpsert` records that created them are still in the log. + VectorDirectTruncate = 25 | 0x8000, + + /// Columnar engine: remove every row of a columnar or spatial collection + /// (`TRUNCATE`). + /// + /// Required: skipping on replay resurrects every truncated row, because + /// the `TimeseriesBatch` records that created them are still in the log. + ColumnarTruncate = 26 | 0x8000, + + /// Timeseries engine: remove every row and partition of a timeseries + /// collection (`TRUNCATE`). + /// + /// Required: skipping on replay resurrects every truncated row, because + /// the `TimeseriesBatch` records that created them are still in the log. + TimeseriesTruncate = 27 | 0x8000, + /// Vector engine: insert (upsert) a sparse vector into the inverted index. /// /// Targets the `SparseInvertedIndex` (keyed by document id), a separate @@ -323,6 +363,12 @@ impl RecordType { x if x == 11 | 0x8000 => Some(Self::VectorDelete), x if x == 12 | 0x8000 => Some(Self::VectorParams), x if x == 13 | 0x8000 => Some(Self::VectorDirectUpsert), + x if x == 19 | 0x8000 => Some(Self::VectorDirectDelete), + x if x == 23 | 0x8000 => Some(Self::VectorDirectUpdate), + x if x == 24 | 0x8000 => Some(Self::VectorResolvedDirectWrite), + x if x == 25 | 0x8000 => Some(Self::VectorDirectTruncate), + x if x == 26 | 0x8000 => Some(Self::ColumnarTruncate), + x if x == 27 | 0x8000 => Some(Self::TimeseriesTruncate), x if x == 14 | 0x8000 => Some(Self::SparseVectorPut), x if x == 15 | 0x8000 => Some(Self::SparseVectorDelete), x if x == 16 | 0x8000 => Some(Self::MultiVectorPut), @@ -401,6 +447,12 @@ mod tests { RecordType::VectorDelete, RecordType::VectorParams, RecordType::VectorDirectUpsert, + RecordType::VectorDirectDelete, + RecordType::VectorDirectUpdate, + RecordType::VectorResolvedDirectWrite, + RecordType::VectorDirectTruncate, + RecordType::ColumnarTruncate, + RecordType::TimeseriesTruncate, RecordType::SparseVectorPut, RecordType::SparseVectorDelete, RecordType::MultiVectorPut, diff --git a/nodedb/src/control/cluster/array_cluster_exec/executor.rs b/nodedb/src/control/cluster/array_cluster_exec/executor.rs index 9308de4d1..157bed5f2 100644 --- a/nodedb/src/control/cluster/array_cluster_exec/executor.rs +++ b/nodedb/src/control/cluster/array_cluster_exec/executor.rs @@ -20,6 +20,8 @@ use crate::control::cluster::array_cluster_helpers::{ cluster_err, encode_err, finalize_agg_partials, }; use crate::control::state::SharedState; +use crate::data::executor::response_codec; +use crate::types::TxnId; use nodedb_physical::physical_plan::ClusterArrayOp; use zerompk; @@ -48,6 +50,7 @@ struct SliceArgs<'a> { prefix_bits: u8, system_time: nodedb_types::SystemTimeScope, valid_at_ms: Option, + txn_id: Option, } struct AggArgs<'a> { @@ -59,6 +62,7 @@ struct AggArgs<'a> { prefix_bits: u8, system_as_of: Option, valid_at_ms: Option, + txn_id: Option, } impl ClusterArrayExecutor { @@ -91,7 +95,17 @@ impl ClusterArrayExecutor { /// Execute a `ClusterArrayOp` and return raw response bytes ready to be /// returned to the client. The response format mirrors local `ArrayOp` /// responses so downstream decode logic is unchanged. - pub(crate) async fn execute(&self, op: &ClusterArrayOp) -> crate::Result> { + /// + /// `txn_id` is the reading transaction's id. A `Slice`/`Agg` inside a + /// transaction block carries it to every shard so each shard folds its + /// own staged cells into the result. A `Put`/`Delete` never runs here + /// inside a transaction block: the staging gate fans it out per shard + /// (`session::array_fanout_stage`) before it reaches this executor. + pub(crate) async fn execute( + &self, + op: &ClusterArrayOp, + txn_id: Option, + ) -> crate::Result> { match op { ClusterArrayOp::Slice { array_id, @@ -112,6 +126,7 @@ impl ClusterArrayExecutor { prefix_bits: *prefix_bits, system_time: *system_time, valid_at_ms: *valid_at_ms, + txn_id, }) .await } @@ -135,6 +150,7 @@ impl ClusterArrayExecutor { prefix_bits: *prefix_bits, system_as_of: *system_as_of, valid_at_ms: *valid_at_ms, + txn_id, }) .await } @@ -173,6 +189,7 @@ impl ClusterArrayExecutor { prefix_bits, system_time, valid_at_ms, + txn_id, } = args; let coordinator = ArrayCoordinator::for_slice( self.source_node, @@ -201,6 +218,7 @@ impl ClusterArrayExecutor { shard_hilbert_range: None, system_time, valid_at_ms, + txn_id: txn_id.map(TxnId::as_u64), }; let result = coordinator .coord_slice(req, limit, system_time) @@ -248,6 +266,7 @@ impl ClusterArrayExecutor { prefix_bits, system_as_of, valid_at_ms, + txn_id, } = args; let coordinator = ArrayCoordinator::for_slice( self.source_node, @@ -270,6 +289,7 @@ impl ClusterArrayExecutor { shard_hilbert_range: None, system_as_of, valid_at_ms, + txn_id: txn_id.map(TxnId::as_u64), }; let agg = coordinator.coord_agg(req).await.map_err(cluster_err)?; @@ -308,7 +328,7 @@ impl ClusterArrayExecutor { source_node: self.source_node, timeout_ms: ARRAY_RPC_TIMEOUT_MS, }; - coord_put( + let resps = coord_put( ¶ms, array_id_msgpack.to_vec(), prefix_bits, @@ -320,10 +340,12 @@ impl ClusterArrayExecutor { .await .map_err(cluster_err)?; - // Return a simple `{"affected": N}` JSON payload — same shape as the - // local ArrayOp::Put response so downstream decode is unchanged. - let affected = cells.len() as u64; - zerompk::to_msgpack_vec(&affected).map_err(encode_err) + // Sum each shard's real `affected` count — never `cells.len()`, which + // counts cells named in the request, not cells the Data Plane + // actually wrote. `{"inserted": n}` matches what the local + // `ArrayOp::Put` handler emits, so `extract_affected_count` reads both. + let affected: u64 = resps.iter().map(|r| r.affected).sum(); + response_codec::encode_count("inserted", affected as usize) } async fn execute_delete( @@ -337,7 +359,7 @@ impl ClusterArrayExecutor { source_node: self.source_node, timeout_ms: ARRAY_RPC_TIMEOUT_MS, }; - coord_delete( + let resps = coord_delete( ¶ms, array_id_msgpack.to_vec(), prefix_bits, @@ -349,7 +371,11 @@ impl ClusterArrayExecutor { .await .map_err(cluster_err)?; - let deleted = coords.len() as u64; - zerompk::to_msgpack_vec(&deleted).map_err(encode_err) + // Sum each shard's real `affected` count — never `coords.len()`, + // which counts coordinates named, not coordinates that existed and + // were removed. A DELETE of an absent coordinate must answer 0, and + // `{"deleted": n}` matches the local `ArrayOp::Delete` handler. + let affected: u64 = resps.iter().map(|r| r.affected).sum(); + response_codec::encode_count("deleted", affected as usize) } } diff --git a/nodedb/src/control/cluster/array_executor/executor.rs b/nodedb/src/control/cluster/array_executor/executor.rs index 3689a1c3c..46d0a381b 100644 --- a/nodedb/src/control/cluster/array_executor/executor.rs +++ b/nodedb/src/control/cluster/array_executor/executor.rs @@ -11,7 +11,7 @@ use nodedb_cluster::error::{ClusterError, Result}; use crate::bridge::envelope::{Priority, Request}; use crate::control::state::SharedState; use crate::event::types::EventSource; -use crate::types::{ReadConsistency, RequestId, TraceId, VShardId}; +use crate::types::{ReadConsistency, RequestId, TraceId, TxnId, VShardId}; use nodedb_physical::physical_plan::PhysicalPlan; /// Timeout for a single shard-side array operation dispatched through the @@ -36,14 +36,20 @@ impl DataPlaneArrayExecutor { /// Dispatch a `PhysicalPlan` through the local SPSC bridge and await the /// single (non-streaming) response. + /// + /// `txn_id` is the reading transaction's id for read-your-own-writes + /// against this shard's staging overlay. `None` for an autocommit read + /// and for every write: a cluster array write is never inside a + /// transaction block, where it is staged per shard instead. pub(super) async fn dispatch_and_await( &self, array_id: &ArrayId, local_vshard_id: VShardId, plan: PhysicalPlan, + txn_id: Option, ) -> Result { let request_id = self.state.next_request_id(); - let request = local_request(request_id, array_id, local_vshard_id, plan); + let request = local_request(request_id, array_id, local_vshard_id, plan, txn_id); let mut rx = self.state.tracker.register(request_id); @@ -81,6 +87,7 @@ fn local_request( array_id: &ArrayId, local_vshard_id: VShardId, plan: PhysicalPlan, + txn_id: Option, ) -> Request { Request { request_id, @@ -97,7 +104,7 @@ fn local_request( user_roles: Vec::new(), user_id: None, statement_digest: None, - txn_id: None, + txn_id, wal_lsn: None, resolved_now_ms: None, admission: crate::bridge::envelope::Admission::Exempt( @@ -139,7 +146,7 @@ mod tests { assert_eq!(entry.tenant_id, tenant_id.as_u64()); assert_eq!(entry.database_id, database_id.as_u64()); - let request = local_request(RequestId::new(7), &array_id, vshard_id, plan); + let request = local_request(RequestId::new(7), &array_id, vshard_id, plan, None); assert_eq!(request.tenant_id, tenant_id); assert_eq!(request.database_id, database_id); assert_eq!(request.vshard_id, vshard_id); @@ -160,8 +167,8 @@ mod tests { provenance: None, }); - let read_request = local_request(RequestId::new(8), &array_id, vshard_id, read); - let write_request = local_request(RequestId::new(9), &array_id, vshard_id, write); + let read_request = local_request(RequestId::new(8), &array_id, vshard_id, read, None); + let write_request = local_request(RequestId::new(9), &array_id, vshard_id, write, None); assert_eq!(read_request.vshard_id, vshard_id); assert_eq!(write_request.vshard_id, vshard_id); diff --git a/nodedb/src/control/cluster/array_executor/read.rs b/nodedb/src/control/cluster/array_executor/read.rs index 5a70d053d..3eb9c6472 100644 --- a/nodedb/src/control/cluster/array_executor/read.rs +++ b/nodedb/src/control/cluster/array_executor/read.rs @@ -11,7 +11,7 @@ use nodedb_cluster::error::{ClusterError, Result}; use nodedb_query::msgpack_scan; use nodedb_types::Surrogate; -use crate::types::VShardId; +use crate::types::{TxnId, VShardId}; use nodedb_types::SurrogateBitmap; use super::executor::DataPlaneArrayExecutor; @@ -53,7 +53,12 @@ impl DataPlaneArrayExecutor { }); let resp = self - .dispatch_and_await(&array_id, VShardId::new(local_vshard_id), plan) + .dispatch_and_await( + &array_id, + VShardId::new(local_vshard_id), + plan, + req.txn_id.map(TxnId::new), + ) .await?; if resp.status == crate::bridge::envelope::Status::Error { @@ -123,7 +128,12 @@ impl DataPlaneArrayExecutor { }); let resp = self - .dispatch_and_await(&array_id, VShardId::new(local_vshard_id), plan) + .dispatch_and_await( + &array_id, + VShardId::new(local_vshard_id), + plan, + req.txn_id.map(TxnId::new), + ) .await?; if resp.status == crate::bridge::envelope::Status::Error { @@ -175,7 +185,7 @@ impl DataPlaneArrayExecutor { }); let resp = self - .dispatch_and_await(&array_id, VShardId::new(local_vshard_id), plan) + .dispatch_and_await(&array_id, VShardId::new(local_vshard_id), plan, None) .await?; if resp.status == crate::bridge::envelope::Status::Error { diff --git a/nodedb/src/control/cluster/array_executor/trait_impl.rs b/nodedb/src/control/cluster/array_executor/trait_impl.rs index 87c0f7773..20d67c428 100644 --- a/nodedb/src/control/cluster/array_executor/trait_impl.rs +++ b/nodedb/src/control/cluster/array_executor/trait_impl.rs @@ -11,7 +11,9 @@ use async_trait::async_trait; use nodedb_cluster::distributed_array::wire::{ ArrayShardAggReq, ArrayShardDeleteReq, ArrayShardPutReq, }; -use nodedb_cluster::distributed_array::{ArrayAggExec, ArrayLocalExecutor, ArraySliceExec}; +use nodedb_cluster::distributed_array::{ + ArrayAggExec, ArrayLocalExecutor, ArrayShardWriteOutcome, ArraySliceExec, +}; use nodedb_cluster::error::Result; use super::executor::DataPlaneArrayExecutor; @@ -30,11 +32,19 @@ impl ArrayLocalExecutor for DataPlaneArrayExecutor { self.agg(local_vshard_id, req).await } - async fn exec_put(&self, local_vshard_id: u32, req: &ArrayShardPutReq) -> Result { + async fn exec_put( + &self, + local_vshard_id: u32, + req: &ArrayShardPutReq, + ) -> Result { self.put(local_vshard_id, req).await } - async fn exec_delete(&self, local_vshard_id: u32, req: &ArrayShardDeleteReq) -> Result { + async fn exec_delete( + &self, + local_vshard_id: u32, + req: &ArrayShardDeleteReq, + ) -> Result { self.delete(local_vshard_id, req).await } diff --git a/nodedb/src/control/cluster/array_executor/write.rs b/nodedb/src/control/cluster/array_executor/write.rs index dc19de8ba..71c9269d9 100644 --- a/nodedb/src/control/cluster/array_executor/write.rs +++ b/nodedb/src/control/cluster/array_executor/write.rs @@ -8,6 +8,7 @@ //! whose redo record is the array engine's only durability (it's a memtable). use nodedb_array::types::ArrayId; +use nodedb_cluster::distributed_array::ArrayShardWriteOutcome; use nodedb_cluster::distributed_array::wire::{ArrayShardDeleteReq, ArrayShardPutReq}; use nodedb_cluster::error::{ClusterError, Result}; @@ -16,11 +17,16 @@ use super::executor::DataPlaneArrayExecutor; use crate::control::server::dispatch_utils::{ ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, submit_write, }; +use crate::control::server::shared::sql::staging_predicates::require_affected_count; use crate::types::{TraceId, VShardId}; use nodedb_physical::physical_plan::{ArrayOp, PhysicalPlan}; impl DataPlaneArrayExecutor { - pub(super) async fn put(&self, local_vshard_id: u32, req: &ArrayShardPutReq) -> Result { + pub(super) async fn put( + &self, + local_vshard_id: u32, + req: &ArrayShardPutReq, + ) -> Result { let array_id: ArrayId = zerompk::from_msgpack(&req.array_id_msgpack).map_err(|e| ClusterError::Codec { detail: format!("array_id decode in exec_put: {e}"), @@ -51,7 +57,7 @@ impl DataPlaneArrayExecutor { &self, local_vshard_id: u32, req: &ArrayShardDeleteReq, - ) -> Result { + ) -> Result { let array_id: ArrayId = zerompk::from_msgpack(&req.array_id_msgpack).map_err(|e| ClusterError::Codec { detail: format!("array_id decode in exec_delete: {e}"), @@ -86,9 +92,10 @@ impl DataPlaneArrayExecutor { /// Replicate `plan` to the owning shard's data Raft group when a proposer /// exists; otherwise apply it locally through the write funnel. Returns - /// the `applied_lsn` the coordinator acks with. Both branches use the - /// vShard from the validated RPC envelope so all paths select the same - /// Data Plane core. + /// the `applied_lsn` the coordinator acks with, plus the real cell count + /// the Data Plane handler's `{"inserted"|"deleted": n}` response reports. + /// Both branches use the vShard from the validated RPC envelope so all + /// paths select the same Data Plane core. async fn propose_or_dispatch( &self, array_id: &ArrayId, @@ -96,7 +103,7 @@ impl DataPlaneArrayExecutor { plan: PhysicalPlan, wal_lsn: u64, op_label: &str, - ) -> Result { + ) -> Result { if let Some(proposer) = self.state.async_raft_proposer() { let replicable = crate::control::wal_replication::ReplicableWrite::decide_for_replication(&plan) @@ -116,15 +123,27 @@ impl DataPlaneArrayExecutor { detail: format!("{op_label}: plan is not encodable as a replicated entry"), })?; - crate::control::wal_replication::propose_replicated_entry(&self.state, proposer, entry) + let (apply_payload, _write_version) = + crate::control::wal_replication::propose_replicated_entry( + &self.state, + proposer, + entry, + ) .await .map_err(|e| ClusterError::Storage { detail: format!("{op_label} raft propose: {e}"), })?; + let affected = + require_affected_count(&apply_payload).map_err(|e| ClusterError::Storage { + detail: format!("{op_label}: {e}"), + })?; // No LSN exists to report here: each replica mints its own redo, // none authoritative. `wal_lsn` is echoed back verbatim — what // the coordinator sent, not a claim about what was recorded. - return Ok(wal_lsn); + return Ok(ArrayShardWriteOutcome { + applied_lsn: wal_lsn, + affected, + }); } // Single-node: this node's WAL is the write's only durability. The @@ -153,14 +172,24 @@ impl DataPlaneArrayExecutor { // Ack with the LSN the funnel actually minted. `None` would mean the // funnel classified this as appending nothing — a wiring bug. - outcome - .wal_lsn - .map(|lsn| lsn.as_u64()) - .ok_or_else(|| ClusterError::Storage { - detail: format!( - "{op_label}: applied with no WAL redo record — write is not durable" - ), - }) + let applied_lsn = + outcome + .wal_lsn + .map(|lsn| lsn.as_u64()) + .ok_or_else(|| ClusterError::Storage { + detail: format!( + "{op_label}: applied with no WAL redo record — write is not durable" + ), + })?; + let affected = require_affected_count(outcome.response.payload.as_ref()).map_err(|e| { + ClusterError::Storage { + detail: format!("{op_label}: {e}"), + } + })?; + Ok(ArrayShardWriteOutcome { + applied_lsn, + affected, + }) } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs new file mode 100644 index 000000000..206d24982 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs @@ -0,0 +1,163 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Active (dependent-read) transaction dispatch: submits a +//! `CalvinExecuteActive` task once all passive results have landed. + +use std::time::Instant; + +use tracing::error; + +use nodedb_cluster::calvin::types::SequencedTxn; +use nodedb_physical::physical_plan::PhysicalPlan; +use nodedb_physical::physical_plan::meta::MetaOp; + +use super::super::scheduler::Scheduler; +use super::primary_write::{ + participant_change_sets, plans_have_primary_write, plans_have_returning, + txn_has_non_derived_write, +}; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; + +impl Scheduler { + /// Dispatch an active dependent-read txn once all passive results are in. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn dispatch_active_txn( + &mut self, + txn: SequencedTxn, + txn_id: TxnId, + lock_owner: TxnId, + injected_reads: std::collections::BTreeMap< + nodedb_physical::physical_plan::meta::PassiveReadKeyId, + nodedb_types::Value, + >, + ) { + let request_id = self.next_request_id(); + let tenant_id = txn.tx_class.tenant_id; + let epoch = txn.epoch; + let position = txn.position; + + let plans = match super::super::super::helpers::decode_plans(&txn.tx_class.plans) { + Ok(p) => p, + Err(e) => { + error!( + vshard_id = self.vshard_id, + epoch, + position, + error = %e, + "calvin scheduler: active plan decode failed; releasing locks" + ); + self.on_txn_complete(txn_id); + return; + } + }; + let has_non_derived_write = txn_has_non_derived_write(&plans); + let mut plans = + match self.local_calvin_plans(plans, txn.tx_class.database_id, epoch, position) { + Ok(p) if !p.is_empty() => p, + Ok(_) => { + // A dependent-read active txn dispatched here always carries a + // local write slice (the OLLP orchestrator only routes the write + // participant through this path). An empty local slice is a + // routing bug, not a read-only participant — surface it as a + // terminal routing failure rather than dispatching an + // active task with nothing to apply. + let e = crate::Error::Internal { + detail: format!( + "calvin active txn {epoch}/{position} homes no local write plans \ + for vshard {}", + self.vshard_id + ), + }; + error!( + vshard_id = self.vshard_id, + epoch, + position, + error = %e, + "calvin scheduler: active txn homes no local writes; releasing locks" + ); + self.propose_routing_failure(epoch, position, txn_id, &e); + self.on_txn_complete(txn_id); + return; + } + Err(e) => { + error!( + vshard_id = self.vshard_id, + epoch, + position, + error = %e, + "calvin scheduler: active txn routing failed; releasing locks" + ); + self.propose_routing_failure(epoch, position, txn_id, &e); + self.on_txn_complete(txn_id); + return; + } + }; + if !self.bind_local_identities(&mut plans, txn.tx_class.database_id, tenant_id, txn_id) { + return; + } + let has_primary_write = plans_have_primary_write(&plans, has_non_derived_write); + let has_returning = plans_have_returning(&plans); + let change_sets = participant_change_sets(&plans, tenant_id, self.vshard_id); + let plan = PhysicalPlan::Meta(MetaOp::CalvinExecuteActive { + epoch, + position, + tenant_id, + plans, + injected_reads, + epoch_system_ms: txn.epoch_system_ms, + is_group_leader: self.is_group_leader(), + }); + + // Calvin allocates the CalvinApplied WAL LSN post-apply (in the + // scheduler's response handler), so no committed LSN is known at + // dispatch time to stamp here. + let request = + self.build_exempt_request(request_id, tenant_id, txn.tx_class.database_id, plan, None); + + let resp_rx = self.shared.tracker.register(request_id); + + let dispatch_result = match self.shared.dispatcher.lock() { + Ok(mut d) => d.dispatch(request), + Err(poisoned) => poisoned.into_inner().dispatch(request), + }; + + if let Err(e) = dispatch_result { + error!( + vshard_id = self.vshard_id, + epoch, + position, + error = %e, + "calvin scheduler: active dispatch failed; releasing locks" + ); + self.on_txn_complete(txn_id); + return; + } + + self.metrics.record_dispatch(); + + // no-determinism: executor latency observability, off-WAL path + let dispatch_instant = Instant::now(); + + self.spawn_response_bridge(txn_id, request_id, resp_rx); + + self.pending.insert( + txn_id, + super::super::super::types::PendingTxn { + txn, + lock_owner, + // no-determinism: dispatch_time is scheduler observability, not Calvin WAL data + dispatch_time: dispatch_instant, + has_primary_write, + has_returning, + change_sets, + // The dependent-read active path STAGES (leader-verify OLLP + + // buffer, no base apply); its response drives the same + // resolve → redo → flush as the static path, for + // WAL-only-restart durability. `resolve_staged_commit` reads the + // `read_set_valid: None` the active handler returns as "commit". + commit_state: Some(super::super::super::types::CommitState::Staged), + // Set only once the txn parks in `AwaitingVerdict`. + verdict_deadline: None, + }, + ); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/bind_identities.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/bind_identities.rs new file mode 100644 index 000000000..e8ffd9451 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/bind_identities.rs @@ -0,0 +1,54 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Install the pk → surrogate identities a local Calvin write slice carries. +//! +//! The coordinator assigned each surrogate at plan time in its own catalog. +//! Every participant — group leader and follower alike — applies the slice +//! to its Data Plane, so every participant installs the same binding here, +//! or a later point read by primary key resolves nothing on this node. + +use nodedb_physical::physical_plan::PhysicalPlan; +use tracing::error; + +use super::super::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; +use crate::control::surrogate::bind_plan_identities; +use crate::types::{DatabaseId, TenantId}; + +impl Scheduler { + /// Bind every identity in `plans` first-wins and rewrite each surrogate + /// slot with the authoritative value, the same walk the replicated-write + /// decoder runs. On a catalog error the txn is terminated as a routing + /// failure: applying rows nobody can resolve by key is worse than aborting. + /// + /// Returns `false` after terminating the txn; the caller returns at once. + pub(super) fn bind_local_identities( + &mut self, + plans: &mut [PhysicalPlan], + database_id: DatabaseId, + tenant_id: TenantId, + txn_id: TxnId, + ) -> bool { + let assigner = &self.shared.surrogate_assigner; + let bound = plans + .iter_mut() + .try_for_each(|plan| bind_plan_identities(assigner, database_id, tenant_id, plan)); + match bound { + Ok(()) => true, + Err(e) => { + let epoch = txn_id.epoch; + let position = txn_id.position; + error!( + vshard_id = self.vshard_id, + epoch, + position, + error = %e, + "calvin scheduler: surrogate binding failed; releasing locks" + ); + self.propose_routing_failure(epoch, position, txn_id, &e); + self.on_txn_complete(txn_id); + false + } + } + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/mod.rs new file mode 100644 index 000000000..5dc967ae2 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/mod.rs @@ -0,0 +1,8 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Static and active (dependent-read) txn dispatch to the Data Plane. + +mod active_dispatch; +mod bind_identities; +mod primary_write; +mod static_dispatch; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/primary_write.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/primary_write.rs new file mode 100644 index 000000000..054c12dc2 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/primary_write.rs @@ -0,0 +1,98 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-slice write classification: primary-write / RETURNING / change-set +//! predicates shared by the static and active dispatch paths. + +use nodedb_physical::physical_plan::PhysicalPlan; + +use crate::types::VShardId; + +/// Whether this vShard's slice carries a PRIMARY user data write — the write +/// whose applied `Response` (affected-count + any RETURNING rows) the +/// coordinator surfaces. +/// +/// A primary write is a Document / KV / Vector / Timeseries / Columnar / Array +/// write — NOT the implicit graph-edge cleanup (`EdgePut` / `EdgeDelete`) that +/// dual-homes alongside a document delete/update. For a single-collection user +/// DML (plus its implicit edges) exactly ONE participant carries the primary +/// write, so only it deposits the applied `Response` into the coordinator's +/// sidecar and the edge participants never clobber the entry. +/// +/// This gate subsumes the RETURNING case (a RETURNING write IS a primary write, +/// so its rows are still deposited) while ALSO carrying the affected-count of a +/// plain (non-RETURNING) write — which a RETURNING-only gate dropped, making a +/// routed plain write report zero rows affected. +pub(super) fn participant_change_sets( + plans: &[PhysicalPlan], + tenant_id: crate::types::TenantId, + vshard_id: u32, +) -> Vec { + plans + .iter() + .filter(|plan| match plan { + // Edge plans are dual-homed; only the source participant publishes + // the one logical Control-Plane event. + PhysicalPlan::Graph( + nodedb_physical::physical_plan::GraphOp::EdgePut { src_id, .. } + | nodedb_physical::physical_plan::GraphOp::EdgeDelete { src_id, .. }, + ) => VShardId::from_key(src_id.as_bytes()).as_u32() == vshard_id, + _ => true, + }) + .map(|plan| { + crate::control::server::dispatch_utils::extract_write_change_set(plan, tenant_id) + }) + .collect() +} + +/// Whether the transaction carries any write that is NOT a derived side +/// effect (an implicit graph edge, a cross-shard balance delta). Decided over +/// the FULL plan set before it is sliced per vShard, because a slice alone +/// cannot tell a lone derived participant from a derived-only statement. +pub(super) fn txn_has_non_derived_write(plans: &[PhysicalPlan]) -> bool { + crate::control::planner::calvin::write_class::plans_have_user_write(plans) +} + +/// Whether this vShard's slice carries the USER'S own write, as opposed to a +/// derived side effect the Control Plane appended alongside it. +/// +/// It gates the applied-response deposit, and that is the whole reason the +/// distinction has to be made: a statement's `CommandComplete` is shaped from +/// ONE deposited response, primary-write participants coalesce first-wins, and +/// a derived participant's response describes a row the user's statement never +/// named. A balance write that won that race handed an `INSERT` tag a count — +/// or, when its flush found the commit already resolved and answered with an +/// empty payload, no count at all — belonging to a different write entirely. +/// +/// `is_derived_side_effect` is the named predicate rather than an inline +/// `!matches!(plan, PhysicalPlan::Graph(_))`: the implicit graph edge and the +/// cross-shard balance are the same concept, and spelling it inline here is why +/// the second one never inherited the exclusion. +/// +/// `txn_has_non_derived_write` is the transaction-level answer from +/// [`txn_has_non_derived_write`]. When the transaction has one, only the slice +/// holding it is primary. When it has none — the standalone `GRAPH INSERT / +/// DELETE EDGE` DSL's Graph-only tx_class — every write slice is the user's +/// write and deposits; first-wins coalescing then picks one identical count. +/// An empty slice (validate-only read) is never primary. +pub(super) fn plans_have_primary_write( + plans: &[PhysicalPlan], + txn_has_non_derived_write: bool, +) -> bool { + if txn_has_non_derived_write { + return self::txn_has_non_derived_write(plans); + } + plans + .iter() + .any(crate::control::planner::calvin::is_write_plan) +} + +/// Whether this vShard's slice carries a RETURNING-bearing write — a plan whose +/// applied response is DATA-ROWs rather than a bare affected-count. Uses the +/// SAME `describe_plan` classification the coordinator's response-shaping uses, +/// so the two never disagree about which participant owns the returned rows. +pub(super) fn plans_have_returning(plans: &[PhysicalPlan]) -> bool { + use crate::control::server::response_shape::types::{PlanKind, describe_plan}; + plans + .iter() + .any(|plan| matches!(describe_plan(plan), PlanKind::ReturningRows)) +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs similarity index 51% rename from nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch.rs rename to nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs index 724cc03dc..e8443cf30 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs @@ -1,89 +1,24 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Static and active (dependent-read) txn dispatch to the Data Plane. +//! Static-set ready-transaction dispatch: local-plan routing, group-leader +//! resolution, and the `CalvinExecuteStatic` submit/stage orchestration. use std::time::Instant; use tracing::error; use nodedb_cluster::calvin::types::SequencedTxn; - -use super::routing::PlanRouting; -use super::scheduler::Scheduler; -use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; -use crate::types::{DatabaseId, VShardId}; use nodedb_physical::physical_plan::PhysicalPlan; use nodedb_physical::physical_plan::meta::MetaOp; -/// Whether this vShard's slice carries a PRIMARY user data write — the write -/// whose applied `Response` (affected-count + any RETURNING rows) the -/// coordinator surfaces. -/// -/// A primary write is a Document / KV / Vector / Timeseries / Columnar / Array -/// write — NOT the implicit graph-edge cleanup (`EdgePut` / `EdgeDelete`) that -/// dual-homes alongside a document delete/update. For a single-collection user -/// DML (plus its implicit edges) exactly ONE participant carries the primary -/// write, so only it deposits the applied `Response` into the coordinator's -/// sidecar and the edge participants never clobber the entry. -/// -/// This gate subsumes the RETURNING case (a RETURNING write IS a primary write, -/// so its rows are still deposited) while ALSO carrying the affected-count of a -/// plain (non-RETURNING) write — which a RETURNING-only gate dropped, making a -/// routed plain write report zero rows affected. -fn participant_change_sets( - plans: &[PhysicalPlan], - tenant_id: crate::types::TenantId, - vshard_id: u32, -) -> Vec { - plans - .iter() - .filter(|plan| match plan { - // Edge plans are dual-homed; only the source participant publishes - // the one logical Control-Plane event. - PhysicalPlan::Graph( - nodedb_physical::physical_plan::GraphOp::EdgePut { src_id, .. } - | nodedb_physical::physical_plan::GraphOp::EdgeDelete { src_id, .. }, - ) => VShardId::from_key(src_id.as_bytes()).as_u32() == vshard_id, - _ => true, - }) - .map(|plan| { - crate::control::server::dispatch_utils::extract_write_change_set(plan, tenant_id) - }) - .collect() -} - -/// Whether this vShard's slice carries the USER'S own write, as opposed to a -/// derived side effect the Control Plane appended alongside it. -/// -/// It gates the applied-response deposit, and that is the whole reason the -/// distinction has to be made: a statement's `CommandComplete` is shaped from -/// ONE deposited response, primary-write participants coalesce first-wins, and -/// a derived participant's response describes a row the user's statement never -/// named. A balance write that won that race handed an `INSERT` tag a count — -/// or, when its flush found the commit already resolved and answered with an -/// empty payload, no count at all — belonging to a different write entirely. -/// -/// `is_derived_side_effect` is the named predicate rather than an inline -/// `!matches!(plan, PhysicalPlan::Graph(_))`: the implicit graph edge and the -/// cross-shard balance are the same concept, and spelling it inline here is why -/// the second one never inherited the exclusion. -pub(crate) fn plans_have_primary_write(plans: &[PhysicalPlan]) -> bool { - plans.iter().any(|plan| { - crate::control::planner::calvin::is_write_plan(plan) - && !crate::control::planner::calvin::write_class::is_derived_side_effect(plan) - }) -} - -/// Whether this vShard's slice carries a RETURNING-bearing write — a plan whose -/// applied response is DATA-ROWs rather than a bare affected-count. Uses the -/// SAME `describe_plan` classification the coordinator's response-shaping uses, -/// so the two never disagree about which participant owns the returned rows. -pub(crate) fn plans_have_returning(plans: &[PhysicalPlan]) -> bool { - use crate::control::server::response_shape::types::{PlanKind, describe_plan}; - plans - .iter() - .any(|plan| matches!(describe_plan(plan), PlanKind::ReturningRows)) -} +use super::super::routing::PlanRouting; +use super::super::scheduler::Scheduler; +use super::primary_write::{ + participant_change_sets, plans_have_primary_write, plans_have_returning, + txn_has_non_derived_write, +}; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; +use crate::types::DatabaseId; impl Scheduler { /// Whether THIS node is currently the leader of the data-group owning this @@ -115,7 +50,7 @@ impl Scheduler { /// full deadline and report a generic timeout. Mirrors the OllpMismatch /// broadcast in `handle_executor_response`. Shared by `dispatch_txn` and /// `dispatch_active_txn`. - fn propose_routing_failure( + pub(super) fn propose_routing_failure( &self, epoch: u64, position: u32, @@ -151,7 +86,7 @@ impl Scheduler { ) -> crate::Result> { let mut local = Vec::new(); for plan in plans { - match super::routing::plan_vshard_in_database(&plan, database_id) { + match super::super::routing::plan_vshard_in_database(&plan, database_id) { PlanRouting::Vshards(vshards) => { if vshards.iter().any(|v| v.as_u32() == self.vshard_id) { local.push(plan); @@ -203,7 +138,7 @@ impl Scheduler { let epoch = txn.epoch; let position = txn.position; - let plans = match super::super::helpers::decode_plans(&txn.tx_class.plans) { + let plans = match super::super::super::helpers::decode_plans(&txn.tx_class.plans) { Ok(p) => p, Err(e) => { error!( @@ -217,22 +152,26 @@ impl Scheduler { return; } }; - let local = match self.local_calvin_plans(plans, txn.tx_class.database_id, epoch, position) - { - Ok(p) => p, - Err(e) => { - error!( - vshard_id = self.vshard_id, - epoch, - position, - error = %e, - "calvin scheduler: static txn routing failed; releasing locks" - ); - self.propose_routing_failure(epoch, position, txn_id, &e); - self.on_txn_complete(txn_id); - return; - } - }; + let has_non_derived_write = txn_has_non_derived_write(&plans); + let mut local = + match self.local_calvin_plans(plans, txn.tx_class.database_id, epoch, position) { + Ok(p) => p, + Err(e) => { + error!( + vshard_id = self.vshard_id, + epoch, + position, + error = %e, + "calvin scheduler: static txn routing failed; releasing locks" + ); + self.propose_routing_failure(epoch, position, txn_id, &e); + self.on_txn_complete(txn_id); + return; + } + }; + if !self.bind_local_identities(&mut local, txn.tx_class.database_id, tenant_id, txn_id) { + return; + } // A participant with no local WRITE slice is either a READ-ONLY // participant — writes home elsewhere, but a read homes HERE, so it must @@ -240,7 +179,7 @@ impl Scheduler { // vote — or a routing bug (neither writes nor reads home here). Only the // latter is an error; the former stages a validate-only task below. if local.is_empty() - && !super::routing::homes_versioned_read( + && !super::super::routing::homes_versioned_read( &txn.tx_class.versioned_reads, txn.tx_class.database_id, self.vshard_id, @@ -270,7 +209,14 @@ impl Scheduler { // identical static path so each casts a real commit/abort Vote through // stage -> resolve -> verdict. The validate-only task stages no plans; // its response carries only the read-set vote. - self.dispatch_calvin_static(txn, txn_id, lock_owner, tenant_id, local); + self.dispatch_calvin_static( + txn, + txn_id, + lock_owner, + tenant_id, + local, + has_non_derived_write, + ); } /// Build and dispatch a `CalvinExecuteStatic` task, then park the txn in @@ -289,6 +235,7 @@ impl Scheduler { lock_owner: TxnId, tenant_id: crate::types::TenantId, plans: Vec, + has_non_derived_write: bool, ) { // The apply-slot identity (used in the CalvinExecuteStatic task and // error logs) is exactly `txn_id`; deriving it here keeps the two in @@ -296,7 +243,7 @@ impl Scheduler { let epoch = txn_id.epoch; let position = txn_id.position; let request_id = self.next_request_id(); - let has_primary_write = plans_have_primary_write(&plans); + let has_primary_write = plans_have_primary_write(&plans, has_non_derived_write); let has_returning = plans_have_returning(&plans); let change_sets = participant_change_sets(&plans, tenant_id, self.vshard_id); let plan = PhysicalPlan::Meta(MetaOp::CalvinExecuteStatic { @@ -346,7 +293,7 @@ impl Scheduler { self.pending.insert( txn_id, - super::super::types::PendingTxn { + super::super::super::types::PendingTxn { txn, lock_owner, // no-determinism: dispatch_time is scheduler observability, not Calvin WAL data @@ -357,145 +304,7 @@ impl Scheduler { // This dispatch STAGED the txn (validate + buffer, no apply); // its response carries the local commit vote that drives the // subsequent flush-or-drop. - commit_state: Some(super::super::types::CommitState::Staged), - // Set only once the txn parks in `AwaitingVerdict`. - verdict_deadline: None, - }, - ); - } - - /// Dispatch an active dependent-read txn once all passive results are in. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn dispatch_active_txn( - &mut self, - txn: SequencedTxn, - txn_id: TxnId, - lock_owner: TxnId, - injected_reads: std::collections::BTreeMap< - nodedb_physical::physical_plan::meta::PassiveReadKeyId, - nodedb_types::Value, - >, - ) { - let request_id = self.next_request_id(); - let tenant_id = txn.tx_class.tenant_id; - let epoch = txn.epoch; - let position = txn.position; - - let plans = match super::super::helpers::decode_plans(&txn.tx_class.plans) { - Ok(p) => p, - Err(e) => { - error!( - vshard_id = self.vshard_id, - epoch, - position, - error = %e, - "calvin scheduler: active plan decode failed; releasing locks" - ); - self.on_txn_complete(txn_id); - return; - } - }; - let plans = match self.local_calvin_plans(plans, txn.tx_class.database_id, epoch, position) - { - Ok(p) if !p.is_empty() => p, - Ok(_) => { - // A dependent-read active txn dispatched here always carries a - // local write slice (the OLLP orchestrator only routes the write - // participant through this path). An empty local slice is a - // routing bug, not a read-only participant — surface it as a - // terminal routing failure rather than dispatching an - // active task with nothing to apply. - let e = crate::Error::Internal { - detail: format!( - "calvin active txn {epoch}/{position} homes no local write plans \ - for vshard {}", - self.vshard_id - ), - }; - error!( - vshard_id = self.vshard_id, - epoch, - position, - error = %e, - "calvin scheduler: active txn homes no local writes; releasing locks" - ); - self.propose_routing_failure(epoch, position, txn_id, &e); - self.on_txn_complete(txn_id); - return; - } - Err(e) => { - error!( - vshard_id = self.vshard_id, - epoch, - position, - error = %e, - "calvin scheduler: active txn routing failed; releasing locks" - ); - self.propose_routing_failure(epoch, position, txn_id, &e); - self.on_txn_complete(txn_id); - return; - } - }; - let has_primary_write = plans_have_primary_write(&plans); - let has_returning = plans_have_returning(&plans); - let change_sets = participant_change_sets(&plans, tenant_id, self.vshard_id); - let plan = PhysicalPlan::Meta(MetaOp::CalvinExecuteActive { - epoch, - position, - tenant_id, - plans, - injected_reads, - epoch_system_ms: txn.epoch_system_ms, - is_group_leader: self.is_group_leader(), - }); - - // Calvin allocates the CalvinApplied WAL LSN post-apply (in the - // scheduler's response handler), so no committed LSN is known at - // dispatch time to stamp here. - let request = - self.build_exempt_request(request_id, tenant_id, txn.tx_class.database_id, plan, None); - - let resp_rx = self.shared.tracker.register(request_id); - - let dispatch_result = match self.shared.dispatcher.lock() { - Ok(mut d) => d.dispatch(request), - Err(poisoned) => poisoned.into_inner().dispatch(request), - }; - - if let Err(e) = dispatch_result { - error!( - vshard_id = self.vshard_id, - epoch, - position, - error = %e, - "calvin scheduler: active dispatch failed; releasing locks" - ); - self.on_txn_complete(txn_id); - return; - } - - self.metrics.record_dispatch(); - - // no-determinism: executor latency observability, off-WAL path - let dispatch_instant = Instant::now(); - - self.spawn_response_bridge(txn_id, request_id, resp_rx); - - self.pending.insert( - txn_id, - super::super::types::PendingTxn { - txn, - lock_owner, - // no-determinism: dispatch_time is scheduler observability, not Calvin WAL data - dispatch_time: dispatch_instant, - has_primary_write, - has_returning, - change_sets, - // The dependent-read active path now STAGES (leader-verify OLLP - // + buffer, no base apply); its response drives the same - // resolve → redo → flush as the static path, restoring - // WAL-only-restart durability. `resolve_staged_commit` reads the - // `read_set_valid: None` the active handler returns as "commit". - commit_state: Some(super::super::types::CommitState::Staged), + commit_state: Some(super::super::super::types::CommitState::Staged), // Set only once the txn parks in `AwaitingVerdict`. verdict_deadline: None, }, diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs index 5ad829365..af8e2f59b 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs @@ -210,13 +210,26 @@ fn vector_routing(op: &VectorOp, database_id: DatabaseId) -> PlanRouting { | VectorOp::SparseDelete { collection, .. } | VectorOp::MultiVectorInsert { collection, .. } | VectorOp::MultiVectorDelete { collection, .. } - | VectorOp::DirectUpsert { collection, .. } => { + | VectorOp::DirectUpsert { collection, .. } + | VectorOp::DirectInsert { collection, .. } + | VectorOp::DirectInsertIfAbsent { collection, .. } + | VectorOp::DirectDelete { collection, .. } + | VectorOp::DirectTruncate { collection, .. } + | VectorOp::DirectUpdate { collection, .. } => { PlanRouting::Vshards(vec![collection_vshard_in_database( database_id, collection.as_str(), )]) } - VectorOp::Search { .. } + // Never scheduled: the write-resolve orchestrator proposes it through + // Raft directly, on the vshard of the collection it resolved. + VectorOp::ResolvedDirectWrite { .. } => PlanRouting::Unroutable( + "resolved governed vector write: proposed directly by the write-resolve orchestrator", + ), + // Read-only: it reports what the wrapped write would do and mutates + // nothing. + VectorOp::ResolveDirectWrite(_) + | VectorOp::Search { .. } | VectorOp::MultiSearch { .. } | VectorOp::SetParams { .. } | VectorOp::DropIndex { .. } @@ -290,7 +303,7 @@ fn graph_routing(op: &GraphOp) -> PlanRouting { fn timeseries_routing(op: &TimeseriesOp, database_id: DatabaseId) -> PlanRouting { match op { - TimeseriesOp::Ingest { collection, .. } => { + TimeseriesOp::Ingest { collection, .. } | TimeseriesOp::Truncate { collection, .. } => { PlanRouting::Vshards(vec![collection_vshard_in_database( database_id, collection.as_str(), @@ -308,7 +321,8 @@ fn columnar_routing(op: &ColumnarOp, database_id: DatabaseId) -> PlanRouting { | ColumnarOp::Update { collection, .. } | ColumnarOp::Delete { collection, .. } | ColumnarOp::ResolvedUpdate { collection, .. } - | ColumnarOp::ResolvedDelete { collection, .. } => { + | ColumnarOp::ResolvedDelete { collection, .. } + | ColumnarOp::Truncate { collection, .. } => { PlanRouting::Vshards(vec![collection_vshard_in_database( database_id, collection.as_str(), @@ -605,6 +619,7 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "vecs"), field: "emb".to_owned(), surrogate: Surrogate::new(3), + pk_bytes: Vec::new(), vector: vec![0.5, 0.6], payload: vec![1, 2, 3], quantization: VectorQuantization::None, @@ -612,6 +627,8 @@ mod tests { payload_indexes: vec![("tenant_id".to_owned(), PayloadIndexKind::Equality)], returning: None, rls_filters: Vec::new(), + on_conflict_updates: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::decided_earlier_in_request(), }); assert_eq!(vshards_of(&direct_upsert), vec![want]); @@ -672,6 +689,7 @@ mod tests { clauses: Vec::new(), returning: None, resolved_inserts: None, + resolved_insert_identities: Vec::new(), source_rows: None, rls_filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), diff --git a/nodedb/src/control/crdt_admission.rs b/nodedb/src/control/crdt_admission.rs index 2f881405b..726333486 100644 --- a/nodedb/src/control/crdt_admission.rs +++ b/nodedb/src/control/crdt_admission.rs @@ -1220,6 +1220,7 @@ mod tests { fields_json: "{}".into(), surrogate, partial: false, + verb: nodedb_physical::physical_plan::CrdtWriteVerb::Insert, returning: None, rls_filters: Vec::new(), }); diff --git a/nodedb/src/control/gateway/error_map/remote_code.rs b/nodedb/src/control/gateway/error_map/remote_code.rs index 7cc5da0d8..90e6a5987 100644 --- a/nodedb/src/control/gateway/error_map/remote_code.rs +++ b/nodedb/src/control/gateway/error_map/remote_code.rs @@ -30,6 +30,7 @@ pub(super) fn remote_code_to_resp_prefix(code: nodedb_types::error::ErrorCode) - Ec::COLLECTION_NOT_FOUND => "NOTFOUND", Ec::AUTHORIZATION_DENIED => "NOPERM", Ec::CONSTRAINT_VIOLATION => "CONSTRAINT", + Ec::TYPE_MISMATCH => "WRONGTYPE", _ => "ERR", } } @@ -72,6 +73,10 @@ mod tests { remote_code_to_resp_prefix(ErrorCode::CONSTRAINT_VIOLATION), "CONSTRAINT" ); + assert_eq!( + remote_code_to_resp_prefix(ErrorCode::TYPE_MISMATCH), + "WRONGTYPE" + ); } #[test] diff --git a/nodedb/src/control/gateway/error_map/resp.rs b/nodedb/src/control/gateway/error_map/resp.rs index f72c1a78b..97e479900 100644 --- a/nodedb/src/control/gateway/error_map/resp.rs +++ b/nodedb/src/control/gateway/error_map/resp.rs @@ -25,6 +25,7 @@ impl GatewayErrorMap { format!("ERR {detail}") } Error::RejectedConstraint { detail, .. } => format!("CONSTRAINT {detail}"), + Error::TypeMismatch { detail, .. } => format!("WRONGTYPE {detail}"), Error::RetryableSchemaChanged { descriptor } => { format!("ERR schema changed ({descriptor}); please retry") } @@ -79,6 +80,17 @@ mod tests { assert!(msg.starts_with("ERR")); } + #[test] + fn resp_type_mismatch_is_wrongtype() { + let err = Error::TypeMismatch { + collection: "c".into(), + key: "k".into(), + detail: "key holds a bare value, not a hash".into(), + }; + let msg = GatewayErrorMap::to_resp(&err); + assert!(msg.starts_with("WRONGTYPE "), "{msg}"); + } + #[test] fn to_resp_remote_typed_is_wired_to_helper() { use nodedb_types::error::ErrorCode; diff --git a/nodedb/src/control/gateway/version_set/keys.rs b/nodedb/src/control/gateway/version_set/keys.rs new file mode 100644 index 000000000..7b00c1f0b --- /dev/null +++ b/nodedb/src/control/gateway/version_set/keys.rs @@ -0,0 +1,40 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Synthetic pseudo-collection keys folding tenant permission-tree / RLS +//! policy versions into a [`super::GatewayVersionSet`]. + +/// Prefix for the synthetic entry carrying a tenant's permission-tree +/// version. `\0` makes it unrepresentable as a real collection name, so it +/// can never collide with one. +const PERMISSION_TREE_VERSION_KEY_PREFIX: &str = "\0__permission_tree_version::"; + +/// Prefix for the synthetic entry carrying a tenant's RLS policy version. +const RLS_VERSION_KEY_PREFIX: &str = "\0__rls_version::"; + +/// The synthetic pseudo-collection key a tenant's permission-tree version is +/// folded into the set under, so a revoked grant makes a cached gateway plan +/// unlookupable exactly like a bumped descriptor does. +pub fn permission_tree_version_key(tenant_id: u64) -> String { + format!("{PERMISSION_TREE_VERSION_KEY_PREFIX}{tenant_id}") +} + +/// The synthetic pseudo-collection key a tenant's RLS policy version is +/// folded into the set under. +pub fn rls_version_key(tenant_id: u64) -> String { + format!("{RLS_VERSION_KEY_PREFIX}{tenant_id}") +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn pseudo_keys_are_scoped_per_tenant() { + assert_ne!( + permission_tree_version_key(1), + permission_tree_version_key(2) + ); + assert_ne!(rls_version_key(1), rls_version_key(2)); + assert_ne!(permission_tree_version_key(1), rls_version_key(1)); + } +} diff --git a/nodedb/src/control/gateway/version_set/mod.rs b/nodedb/src/control/gateway/version_set/mod.rs new file mode 100644 index 000000000..d465170d9 --- /dev/null +++ b/nodedb/src/control/gateway/version_set/mod.rs @@ -0,0 +1,15 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `GatewayVersionSet` — deterministic ordered set of (collection, version) +//! pairs used as a plan cache key and as the payload for +//! `DescriptorVersionEntry` in `ExecuteRequest`. Collected from a +//! `PhysicalPlan` by walking every variant and extracting the collection +//! name. + +mod keys; +mod plan_keys; +mod set; + +pub use keys::{permission_tree_version_key, rls_version_key}; +pub use plan_keys::touched_collections; +pub use set::GatewayVersionSet; diff --git a/nodedb/src/control/gateway/version_set.rs b/nodedb/src/control/gateway/version_set/plan_keys.rs similarity index 67% rename from nodedb/src/control/gateway/version_set.rs rename to nodedb/src/control/gateway/version_set/plan_keys.rs index fffb0cdb9..9bd354e41 100644 --- a/nodedb/src/control/gateway/version_set.rs +++ b/nodedb/src/control/gateway/version_set/plan_keys.rs @@ -1,134 +1,10 @@ // SPDX-License-Identifier: BUSL-1.1 -//! `GatewayVersionSet` — deterministic ordered set of (collection, version) -//! pairs used as a plan cache key and as the payload for -//! `DescriptorVersionEntry` in `ExecuteRequest`. -//! -//! Collected from a `PhysicalPlan` by walking every variant and extracting -//! the collection name. - -use std::hash::{DefaultHasher, Hash, Hasher}; +//! Extraction of every collection name touched by a `PhysicalPlan`, per +//! engine, feeding [`super::GatewayVersionSet::from_plan`]. use nodedb_physical::physical_plan::PhysicalPlan; -/// Deterministic ordered set of `(collection_name, descriptor_version)` pairs. -/// -/// - Sorted by `collection_name` for stable equality comparisons. -/// - Duplicate names are de-duped (last write wins — within a single plan -/// the version is stable, so duplicates carry the same version). -#[derive(Debug, Clone, PartialEq, Eq, Hash)] -pub struct GatewayVersionSet(Vec<(String, u64)>); - -/// Prefix for the synthetic entry carrying a tenant's permission-tree -/// version. `\0` makes it unrepresentable as a real collection name, so it -/// can never collide with one. -const PERMISSION_TREE_VERSION_KEY_PREFIX: &str = "\0__permission_tree_version::"; - -/// Prefix for the synthetic entry carrying a tenant's RLS policy version. -const RLS_VERSION_KEY_PREFIX: &str = "\0__rls_version::"; - -/// The synthetic pseudo-collection key a tenant's permission-tree version is -/// folded into the set under, so a revoked grant makes a cached gateway plan -/// unlookupable exactly like a bumped descriptor does. -pub fn permission_tree_version_key(tenant_id: u64) -> String { - format!("{PERMISSION_TREE_VERSION_KEY_PREFIX}{tenant_id}") -} - -/// The synthetic pseudo-collection key a tenant's RLS policy version is -/// folded into the set under. -pub fn rls_version_key(tenant_id: u64) -> String { - format!("{RLS_VERSION_KEY_PREFIX}{tenant_id}") -} - -impl GatewayVersionSet { - /// Construct from explicit (name, version) pairs. - pub fn from_pairs(mut pairs: Vec<(String, u64)>) -> Self { - pairs.sort_by(|a, b| a.0.cmp(&b.0)); - pairs.dedup_by(|a, b| a.0 == b.0); - Self(pairs) - } - - /// Fold one more `(name, version)` pair into the set, re-sorting and - /// re-deduping. Used to add the permission-tree / RLS pseudo-entries - /// alongside the real collection entries `from_plan` already collected. - pub fn with_extra(mut self, name: String, version: u64) -> Self { - self.0.push((name, version)); - self.0.sort_by(|a, b| a.0.cmp(&b.0)); - self.0.dedup_by(|a, b| a.0 == b.0); - self - } - - /// Re-look-up the current descriptor version for every collection already - /// in this set, returning a new set with refreshed versions. - /// - /// Used to check a cached version set is still current before trusting a - /// plan-cache hit: if the result equals the original, the cached plan is - /// still valid. `version_fn` receives a collection name and returns the - /// current descriptor version (or 0 if unknown). - pub fn reverify(&self, version_fn: impl Fn(&str) -> u64) -> Self { - let pairs: Vec<(String, u64)> = self - .0 - .iter() - .map(|(name, _)| { - let v = version_fn(name); - (name.clone(), v) - }) - .collect(); - Self::from_pairs(pairs) - } - - /// Collect all collection names touched by a plan with the provided - /// version lookup function. - /// - /// `version_fn` receives a collection name and returns the current - /// descriptor version (or 0 if unknown). - pub fn from_plan(plan: &PhysicalPlan, version_fn: impl Fn(&str) -> u64) -> Self { - let names = touched_collections(plan); - let mut pairs: Vec<(String, u64)> = names - .into_iter() - .map(|name| { - let v = version_fn(&name); - (name, v) - }) - .collect(); - pairs.sort_by(|a, b| a.0.cmp(&b.0)); - pairs.dedup_by(|a, b| a.0 == b.0); - Self(pairs) - } - - /// Iterate over `(collection, version)` pairs. - pub fn iter(&self) -> impl Iterator { - self.0.iter() - } - - /// Returns `true` if the set mentions `name` at any version. - pub fn contains_collection(&self, name: &str) -> bool { - self.0.iter().any(|(n, _)| n == name) - } - - /// Returns `true` if the set mentions `name` at exactly `version`. - pub fn matches(&self, name: &str, version: u64) -> bool { - self.0 - .iter() - .any(|(n, v)| n.as_str() == name && *v == version) - } - - /// Stable u64 hash of this set, used as part of `PlanCacheKey`. - pub fn stable_hash(&self) -> u64 { - let mut h = DefaultHasher::new(); - self.hash(&mut h); - h.finish() - } - - pub fn is_empty(&self) -> bool { - self.0.is_empty() - } - - pub fn len(&self) -> usize { - self.0.len() - } -} - /// Append every collection a KV op names into `out`. /// /// Split out of [`touched_collections`] so `ResolveWrite` can recurse into the @@ -152,7 +28,7 @@ fn kv_touched_collections(op: &nodedb_physical::physical_plan::KvOp, out: &mut V | DropIndex { collection, .. } | FieldGet { collection, .. } | FieldSet { collection, .. } - | Truncate { collection } + | Truncate { collection, .. } | Incr { collection, .. } | IncrFloat { collection, .. } | Cas { collection, .. } @@ -294,7 +170,17 @@ pub fn touched_collections(plan: &PhysicalPlan) -> Vec { | MultiVectorDelete { collection, .. } | MultiVectorScoreSearch { collection, .. } | DirectUpsert { collection, .. } + | DirectInsert { collection, .. } + | DirectInsertIfAbsent { collection, .. } + | DirectDelete { collection, .. } + | DirectUpdate { collection, .. } + | DirectTruncate { collection, .. } + | ResolvedDirectWrite { collection, .. } | DeleteBySurrogate { collection, .. } => out.push(collection.as_str().to_owned()), + // The wrapped op is the intercepted write verbatim. + ResolveDirectWrite(inner) => { + out.extend(inner.direct_write_collection().map(str::to_owned)); + } } } @@ -363,7 +249,8 @@ pub fn touched_collections(plan: &PhysicalPlan) -> Vec { | ResolvedUpdate { collection, .. } | ResolvedDelete { collection, .. } | ResolveDml { collection, .. } - | MaterializeScan { collection, .. } => out.push(collection.as_str().to_owned()), + | MaterializeScan { collection, .. } + | Truncate { collection, .. } => out.push(collection.as_str().to_owned()), } } @@ -371,9 +258,9 @@ pub fn touched_collections(plan: &PhysicalPlan) -> Vec { PhysicalPlan::Timeseries(op) => { use TimeseriesOp::*; match op { - Scan { collection, .. } | Ingest { collection, .. } => { - out.push(collection.as_str().to_owned()) - } + Scan { collection, .. } + | Ingest { collection, .. } + | Truncate { collection, .. } => out.push(collection.as_str().to_owned()), // The wrapped ingest is the intercepted write verbatim. ResolveIngest(inner) => { @@ -528,91 +415,18 @@ pub fn touched_collections(plan: &PhysicalPlan) -> Vec { #[cfg(test)] mod tests { use super::*; - use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; - use nodedb_types::{DatabaseId, QualifiedCollection}; - - #[test] - fn with_extra_folds_in_a_pseudo_entry_and_stays_deterministic() { - let vs = GatewayVersionSet::from_pairs(vec![("orders".into(), 4)]) - .with_extra(permission_tree_version_key(7), 2) - .with_extra(rls_version_key(7), 9); - assert_eq!(vs.len(), 3); - assert!(vs.matches(&permission_tree_version_key(7), 2)); - assert!(vs.matches(&rls_version_key(7), 9)); - assert!(vs.matches("orders", 4)); - } - - /// The plan-cache key is stale the instant the tenant's stamped - /// permission-tree version diverges from what's live, exactly like a - /// bumped collection descriptor. - #[test] - fn with_extra_pseudo_entry_participates_in_equality() { - let a = GatewayVersionSet::from_pairs(vec![("orders".into(), 4)]) - .with_extra(permission_tree_version_key(7), 1); - let b = GatewayVersionSet::from_pairs(vec![("orders".into(), 4)]) - .with_extra(permission_tree_version_key(7), 2); - assert_ne!(a, b); - } - - #[test] - fn pseudo_keys_are_scoped_per_tenant() { - assert_ne!( - permission_tree_version_key(1), - permission_tree_version_key(2) - ); - assert_ne!(rls_version_key(1), rls_version_key(2)); - assert_ne!(permission_tree_version_key(1), rls_version_key(1)); - } - - #[test] - fn from_plan_kv_get() { - let plan = PhysicalPlan::Kv(KvOp::Get { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "users"), - key: b"key".to_vec(), - rls_filters: vec![], - surrogate_ceiling: None, - }); - let vs = GatewayVersionSet::from_plan(&plan, |_| 5); - assert_eq!(vs.len(), 1); - assert!(vs.matches("users", 5)); - } - - #[test] - fn from_plan_deterministic_order() { - let plan = PhysicalPlan::Kv(KvOp::Get { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "alpha"), - key: vec![], - rls_filters: vec![], - surrogate_ceiling: None, - }); - let vs1 = GatewayVersionSet::from_plan(&plan, |_| 1); - let vs2 = GatewayVersionSet::from_plan(&plan, |_| 1); - assert_eq!(vs1, vs2); - assert_eq!(vs1.stable_hash(), vs2.stable_hash()); - } - - #[test] - fn contains_collection() { - let vs = GatewayVersionSet::from_pairs(vec![("orders".into(), 3), ("users".into(), 7)]); - assert!(vs.contains_collection("orders")); - assert!(vs.contains_collection("users")); - assert!(!vs.contains_collection("products")); - } - - #[test] - fn dedup_on_construction() { - let vs = GatewayVersionSet::from_pairs(vec![ - ("a".into(), 1), - ("a".into(), 1), // duplicate - ]); - assert_eq!(vs.len(), 1); - } #[test] fn kv_transfer_item_extracts_both_collections() { - let plan = PhysicalPlan::Kv(KvOp::TransferItem { - source_collection: QualifiedCollection::new(DatabaseId::DEFAULT, "from_col"), - dest_collection: QualifiedCollection::new(DatabaseId::DEFAULT, "to_col"), + let plan = PhysicalPlan::Kv(nodedb_physical::physical_plan::KvOp::TransferItem { + source_collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "from_col", + ), + dest_collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "to_col", + ), item_key: vec![], dest_key: vec![], surrogate: nodedb_types::Surrogate::ZERO, diff --git a/nodedb/src/control/gateway/version_set/set.rs b/nodedb/src/control/gateway/version_set/set.rs new file mode 100644 index 000000000..8acd0d8d8 --- /dev/null +++ b/nodedb/src/control/gateway/version_set/set.rs @@ -0,0 +1,185 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `GatewayVersionSet` — deterministic ordered set of (collection, version) +//! pairs used as a plan cache key and as the payload for +//! `DescriptorVersionEntry` in `ExecuteRequest`. + +use std::hash::{DefaultHasher, Hash, Hasher}; + +use nodedb_physical::physical_plan::PhysicalPlan; + +use super::plan_keys::touched_collections; + +/// Sort `pairs` by collection name and drop duplicate names (last write +/// wins), the invariant every constructor of [`GatewayVersionSet`] keeps. +fn sorted_deduped(mut pairs: Vec<(String, u64)>) -> Vec<(String, u64)> { + pairs.sort_by(|a, b| a.0.cmp(&b.0)); + pairs.dedup_by(|a, b| a.0 == b.0); + pairs +} + +/// Deterministic ordered set of `(collection_name, descriptor_version)` pairs. +/// +/// - Sorted by `collection_name` for stable equality comparisons. +/// - Duplicate names are de-duped (last write wins — within a single plan +/// the version is stable, so duplicates carry the same version). +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +pub struct GatewayVersionSet(Vec<(String, u64)>); + +impl GatewayVersionSet { + /// Construct from explicit (name, version) pairs. + pub fn from_pairs(pairs: Vec<(String, u64)>) -> Self { + Self(sorted_deduped(pairs)) + } + + /// Fold one more `(name, version)` pair into the set, re-sorting and + /// re-deduping. Used to add the permission-tree / RLS pseudo-entries + /// alongside the real collection entries `from_plan` already collected. + pub fn with_extra(mut self, name: String, version: u64) -> Self { + self.0.push((name, version)); + Self(sorted_deduped(self.0)) + } + + /// Re-look-up the current descriptor version for every collection already + /// in this set, returning a new set with refreshed versions. + /// + /// Used to check a cached version set is still current before trusting a + /// plan-cache hit: if the result equals the original, the cached plan is + /// still valid. `version_fn` receives a collection name and returns the + /// current descriptor version (or 0 if unknown). + pub fn reverify(&self, version_fn: impl Fn(&str) -> u64) -> Self { + let pairs: Vec<(String, u64)> = self + .0 + .iter() + .map(|(name, _)| { + let v = version_fn(name); + (name.clone(), v) + }) + .collect(); + Self::from_pairs(pairs) + } + + /// Collect all collection names touched by a plan with the provided + /// version lookup function. + /// + /// `version_fn` receives a collection name and returns the current + /// descriptor version (or 0 if unknown). + pub fn from_plan(plan: &PhysicalPlan, version_fn: impl Fn(&str) -> u64) -> Self { + let names = touched_collections(plan); + let pairs: Vec<(String, u64)> = names + .into_iter() + .map(|name| { + let v = version_fn(&name); + (name, v) + }) + .collect(); + Self(sorted_deduped(pairs)) + } + + /// Iterate over `(collection, version)` pairs. + pub fn iter(&self) -> impl Iterator { + self.0.iter() + } + + /// Returns `true` if the set mentions `name` at any version. + pub fn contains_collection(&self, name: &str) -> bool { + self.0.iter().any(|(n, _)| n == name) + } + + /// Returns `true` if the set mentions `name` at exactly `version`. + pub fn matches(&self, name: &str, version: u64) -> bool { + self.0 + .iter() + .any(|(n, v)| n.as_str() == name && *v == version) + } + + /// Stable u64 hash of this set, used as part of `PlanCacheKey`. + pub fn stable_hash(&self) -> u64 { + let mut h = DefaultHasher::new(); + self.hash(&mut h); + h.finish() + } + + pub fn is_empty(&self) -> bool { + self.0.is_empty() + } + + pub fn len(&self) -> usize { + self.0.len() + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::gateway::version_set::{permission_tree_version_key, rls_version_key}; + use nodedb_physical::physical_plan::KvOp; + use nodedb_types::{DatabaseId, QualifiedCollection}; + + #[test] + fn with_extra_folds_in_a_pseudo_entry_and_stays_deterministic() { + let vs = GatewayVersionSet::from_pairs(vec![("orders".into(), 4)]) + .with_extra(permission_tree_version_key(7), 2) + .with_extra(rls_version_key(7), 9); + assert_eq!(vs.len(), 3); + assert!(vs.matches(&permission_tree_version_key(7), 2)); + assert!(vs.matches(&rls_version_key(7), 9)); + assert!(vs.matches("orders", 4)); + } + + /// The plan-cache key is stale the instant the tenant's stamped + /// permission-tree version diverges from what's live, exactly like a + /// bumped collection descriptor. + #[test] + fn with_extra_pseudo_entry_participates_in_equality() { + let a = GatewayVersionSet::from_pairs(vec![("orders".into(), 4)]) + .with_extra(permission_tree_version_key(7), 1); + let b = GatewayVersionSet::from_pairs(vec![("orders".into(), 4)]) + .with_extra(permission_tree_version_key(7), 2); + assert_ne!(a, b); + } + + #[test] + fn from_plan_kv_get() { + let plan = PhysicalPlan::Kv(KvOp::Get { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "users"), + key: b"key".to_vec(), + rls_filters: vec![], + surrogate_ceiling: None, + }); + let vs = GatewayVersionSet::from_plan(&plan, |_| 5); + assert_eq!(vs.len(), 1); + assert!(vs.matches("users", 5)); + } + + #[test] + fn from_plan_deterministic_order() { + let plan = PhysicalPlan::Kv(KvOp::Get { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "alpha"), + key: vec![], + rls_filters: vec![], + surrogate_ceiling: None, + }); + let vs1 = GatewayVersionSet::from_plan(&plan, |_| 1); + let vs2 = GatewayVersionSet::from_plan(&plan, |_| 1); + assert_eq!(vs1, vs2); + assert_eq!(vs1.stable_hash(), vs2.stable_hash()); + } + + #[test] + fn contains_collection() { + let vs = GatewayVersionSet::from_pairs(vec![("orders".into(), 3), ("users".into(), 7)]); + assert!(vs.contains_collection("orders")); + assert!(vs.contains_collection("users")); + assert!(!vs.contains_collection("products")); + } + + #[test] + fn dedup_on_construction() { + let vs = GatewayVersionSet::from_pairs(vec![ + ("a".into(), 1), + ("a".into(), 1), // duplicate + ]); + assert_eq!(vs.len(), 1); + } +} diff --git a/nodedb/src/control/insert_select/orchestrator.rs b/nodedb/src/control/insert_select/orchestrator.rs index 0c2d5c442..6eb173b0e 100644 --- a/nodedb/src/control/insert_select/orchestrator.rs +++ b/nodedb/src/control/insert_select/orchestrator.rs @@ -19,7 +19,8 @@ use nodedb_types::{DatabaseId, Lsn, Surrogate, TenantId}; use crate::bridge::envelope::{Payload, PhysicalPlan, Response, Status}; use crate::control::insert_select::copy_rows::{assign_page_rows, resolve_copy_spec}; -use crate::control::maintenance::clone_materializer::{dispatch_local, scan_source_page}; +use crate::control::maintenance::clone_materializer::scan_source_page; +use crate::control::orchestrated_write::apply_orchestrated_write; use crate::control::state::SharedState; use nodedb_physical::physical_plan::DocumentOp; @@ -131,9 +132,9 @@ pub(crate) async fn run_insert_select( documents.push((document_id, value)); surrogates.push(surrogate); } - // Resolve this page's sum targets: `dispatch_local` bypasses the - // statement-level resolution pass, so without this the fold has - // no target to credit. Resolved per page since each is its own + // Resolve this page's sum targets: the orchestrated apply bypasses + // the statement-level resolution pass, so without this the fold + // has no target to credit. Resolved per page since each is its own // atomic write. let page_bodies: Vec<&[u8]> = documents.iter().map(|(_, body)| body.as_slice()).collect(); @@ -175,18 +176,18 @@ pub(crate) async fn run_insert_select( returning: None, rls_filters: Vec::new(), resolved_sum_targets, - // Every page dispatches locally to the target's own core, so a - // binding whose target is co-resident folds here; nothing is - // deferred to a sibling task. + // Every page applies on the target's own vShard, so a binding + // whose target is co-resident folds here; nothing is deferred + // to a sibling task. deferred_sum_targets: Vec::new(), }); - let resp = dispatch_local( + // Each page lands on the target's owner and every replica. + let resp = apply_orchestrated_write( state, tenant_id, database_id, req.target_collection, plan, - None, ) .await?; if resp.status != Status::Ok { @@ -194,17 +195,6 @@ pub(crate) async fn run_insert_select( // rows did not land. Surface the DP error verbatim. return Ok(resp); } - // `dispatch_local` bypasses the funnel's post-apply redo minting, - // so a vector-indexed target's write-set arrives unconsumed. Mint - // it now — without it, a WAL-only restart rebuilds the HNSW from - // nothing for these rows: the vectors are lost, not just stale. - crate::control::server::wal_dispatch::mint_dispatch_local_redo( - &state.wal, - tenant_id, - database_id, - req.target_collection, - &resp, - )?; total_inserted += decode_inserted(&resp.payload).unwrap_or(page_len); if resp.watermark_lsn > max_lsn { max_lsn = resp.watermark_lsn; diff --git a/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs b/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs index 46b0502f1..bee1e3ab0 100644 --- a/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs +++ b/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs @@ -156,6 +156,7 @@ async fn resolve_merge_arms( clauses: clauses.clone(), returning: None, resolved_inserts: None, + resolved_insert_identities: Vec::new(), source_rows: Some(source_rows), // Read-only classification pass: writes nothing, so no gate applies. rls_filters: Vec::new(), diff --git a/nodedb/src/control/merge_orchestrator/orchestrator.rs b/nodedb/src/control/merge_orchestrator/orchestrator.rs index 12f6b3d25..f8e024c6b 100644 --- a/nodedb/src/control/merge_orchestrator/orchestrator.rs +++ b/nodedb/src/control/merge_orchestrator/orchestrator.rs @@ -7,10 +7,13 @@ //! TOCTOU-safe round trip: (0) ship the source rows (scanned on its own //! core, since it may differ from the target's) into `source_rows`; (1) //! resolve — the Data Plane classifies the merge read-only and returns -//! NOT-MATCHED rows; (2) assign a fresh registered surrogate per insert row; -//! (3) apply — the Data Plane re-derives the classification, verifies the -//! insert-key set still matches (`OllpRetryRequired` without writing on -//! drift), and applies every arm in one transaction. +//! NOT-MATCHED rows; (2) assign a fresh registered surrogate per insert row +//! and decide the target's write policy over every resolved arm; (3) apply — +//! the resolved plan lands on the target vShard's owner and every replica +//! through `orchestrated_write`, where the Data Plane re-derives the +//! classification, verifies the insert-key set still matches +//! (`OllpRetryRequired` without writing on drift), and applies every arm in +//! one transaction. //! //! Resolve and apply are separate snapshots; concurrent drift between them //! is caught by apply-time verification and retried (bounded; exhaustion @@ -20,6 +23,9 @@ use nodedb_types::{DatabaseId, TenantId}; use crate::bridge::envelope::{ErrorCode, PhysicalPlan, Response, Status}; use crate::control::maintenance::clone_materializer::{dispatch_local, read_all_source_rows}; +use crate::control::orchestrated_write::{ + apply_orchestrated_write, decide_write_policy_over_images, +}; use crate::control::state::SharedState; use nodedb_physical::physical_plan::document::merge_types::MergeClauseOp; use nodedb_physical::physical_plan::{DocumentOp, ReturningSpec}; @@ -27,7 +33,7 @@ use nodedb_physical::physical_plan::{DocumentOp, ReturningSpec}; use super::resolve_arms::decode_resolve; use crate::control::planner::materialized_sum::resolve_sum_targets_for_bodies; use crate::control::target_identity::{ - assign_target_surrogate, bare_collection_name, resolve_target_pk, + assign_target_surrogate, bare_collection_name, derive_document_id, resolve_target_pk, }; /// Upper bound on resolve→apply retries under concurrent source/target drift. @@ -75,6 +81,7 @@ pub async fn run_authorized_merge( source_join_col, clauses, resolved_inserts: None, + resolved_insert_identities: _, source_rows: _, returning, rls_filters, @@ -139,7 +146,12 @@ pub(crate) async fn run_merge(state: &SharedState, args: MergeArgs<'_>) -> crate .await?; // Phase 1: resolve the NOT-MATCHED insert rows (read-only snapshot). - let resolve_plan = merge_plan(&args, true, None, Some(source_rows.clone()), Vec::new()); + let resolve_plan = merge_plan( + &args, + MergePass::Resolve, + Some(source_rows.clone()), + Vec::new(), + ); let resolve_resp = dispatch_local( state, args.tenant_id, @@ -191,10 +203,30 @@ pub(crate) async fn run_merge(state: &SharedState, args: MergeArgs<'_>) -> crate .await?, ); + // Phase 2a: decide the target's write policy over every resolved arm, + // here where the writing identity is live: the post-image of an + // UPDATE or INSERT arm, the pre-image of a DELETE arm. The apply plan + // then carries a decided check every replica applies verbatim. + decide_write_policy_over_images( + args.rls_write_check, + arms.updates + .iter() + .map(|(_, _, body, _)| body.as_slice()) + .chain(arms.deletes.iter().map(|(_, _, body)| body.as_slice())) + .chain(arms.inserts.iter().map(|(_, body)| body.as_slice())), + args.tenant_id, + args.target_collection, + )?; + let insert_rows = arms.inserts; - // Phase 2: assign a fresh, registered surrogate per inserted row. - let mut resolved: Vec<(String, u32)> = Vec::with_capacity(insert_rows.len()); + // Phase 2: assign a fresh, registered surrogate per inserted row, and + // record the document id it stores under so every applying node + // installs the same pk → surrogate binding. + let mut inserts = ResolvedInserts { + by_join_key: Vec::with_capacity(insert_rows.len()), + identities: Vec::with_capacity(insert_rows.len()), + }; for (join_key, body) in &insert_rows { let surrogate = assign_target_surrogate( state, @@ -204,26 +236,29 @@ pub(crate) async fn run_merge(state: &SharedState, args: MergeArgs<'_>) -> crate &target_pk, body, )?; - resolved.push((join_key.clone(), surrogate.as_u32())); + let document_id = derive_document_id(&target_pk, body, surrogate); + inserts + .by_join_key + .push((join_key.clone(), surrogate.as_u32())); + inserts.identities.push((document_id, surrogate.as_u32())); } - // Phase 3: atomic apply with the pre-assigned surrogates + drift verify. - // The apply reuses THIS attempt's source snapshot so the DP re-derives - // the classification from the same source the resolve saw. + // Phase 3: atomic apply with the pre-assigned surrogates + drift verify, + // on the target's owner and every replica. The apply reuses THIS + // attempt's source snapshot so the DP re-derives the classification + // from the same source the resolve saw. let apply_plan = merge_plan( &args, - false, - Some(resolved), + MergePass::Apply(inserts), Some(source_rows), resolved_sum_targets, ); - let apply_resp = dispatch_local( + let apply_resp = apply_orchestrated_write( state, args.tenant_id, args.database_id, args.target_collection, apply_plan, - None, ) .await?; @@ -240,22 +275,26 @@ pub(crate) async fn run_merge(state: &SharedState, args: MergeArgs<'_>) -> crate // the counter is monotonic and gap-tolerant). continue; } - - // `dispatch_local` bypasses the funnel's post-apply redo minting, so - // a vector-indexed target's write-set arrives unconsumed. Mint it - // now — without it a WAL-only restart rebuilds the HNSW from - // pre-merge records. No-op on non-vector targets. - crate::control::server::wal_dispatch::mint_dispatch_local_redo( - &state.wal, - args.tenant_id, - args.database_id, - args.target_collection, - &apply_resp, - )?; return Ok(apply_resp); } } +/// The NOT-MATCHED surrogates an apply pass carries: keyed by source join +/// value for the Data Plane's drift check, and by target document id for the +/// pk → surrogate binding every applying node installs. +struct ResolvedInserts { + by_join_key: Vec<(String, u32)>, + identities: Vec<(String, u32)>, +} + +/// Which orchestrator pass a `DocumentOp::Merge` plan is built for. +enum MergePass { + /// Read-only classification: writes nothing, projects nothing. + Resolve, + /// The write, with its pre-assigned surrogates and a decided policy. + Apply(ResolvedInserts), +} + /// Build a `DocumentOp::Merge` physical plan for one orchestrator pass. /// /// `source_rows` carries the RAW stored source rows scanned on the source's own @@ -263,11 +302,24 @@ pub(crate) async fn run_merge(state: &SharedState, args: MergeArgs<'_>) -> crate /// rather than reading the source from the target core's local store. fn merge_plan( args: &MergeArgs<'_>, - resolve_only: bool, - resolved_inserts: Option>, + pass: MergePass, source_rows: Option)>>, resolved_sum_targets: Vec, ) -> PhysicalPlan { + // Only APPLY can project rows; RESOLVE's payload is the fixed + // `(updates, deletes, inserts)` tuple `decode_resolve` expects. RESOLVE + // carries the live check unread; APPLY carries the decision the + // orchestrator made over the resolved arms, which is what replicates. + let (returning, resolved_inserts, resolved_insert_identities, rls_write_check) = match pass { + MergePass::Resolve => (None, None, Vec::new(), args.rls_write_check.clone()), + MergePass::Apply(inserts) => ( + args.returning.cloned(), + Some(inserts.by_join_key), + inserts.identities, + nodedb_types::RlsWriteCheck::decided_earlier_in_request(), + ), + }; + let resolve_only = resolved_inserts.is_none(); let merge = DocumentOp::Merge { target_collection: nodedb_types::QualifiedCollection::from_stored( args.target_collection.to_string(), @@ -279,19 +331,12 @@ fn merge_plan( target_join_col: args.target_join_col.to_string(), source_join_col: args.source_join_col.to_string(), clauses: args.clauses.to_vec(), - // Only APPLY can project rows; RESOLVE's payload is the fixed - // `(updates, deletes, inserts)` tuple `decode_resolve` expects. - returning: if resolve_only { - None - } else { - args.returning.cloned() - }, + returning, resolved_inserts, + resolved_insert_identities, source_rows, rls_filters: args.rls_filters.to_vec(), - // Carried on both passes (inert on RESOLVE) so a future writing - // resolve cannot silently lose the gate. - rls_write_check: args.rls_write_check.clone(), + rls_write_check, // Empty on RESOLVE (writes nothing); APPLY carries the resolution. resolved_sum_targets, declared_primary_key: args.declared_primary_key.map(str::to_string), diff --git a/nodedb/src/control/metrics/system/fields.rs b/nodedb/src/control/metrics/system/fields.rs index a1cae9bf5..32f854f0d 100644 --- a/nodedb/src/control/metrics/system/fields.rs +++ b/nodedb/src/control/metrics/system/fields.rs @@ -108,7 +108,8 @@ pub struct SystemMetrics { pub tpc_utilization_ratio: AtomicU64, pub arena_memory_bytes: AtomicU64, /// Current number of live per-transaction staging overlays across all - /// Data-Plane cores (txn_overlays + graph_txn_overlays entries). A gauge: + /// Data-Plane cores (txn_overlays + graph_txn_overlays + array_txn_overlays + /// entries). A gauge: /// rises on first staged write of a transaction, falls when the overlay is /// dropped on COMMIT/ROLLBACK/teardown. A persistently non-zero idle value /// indicates leaked abandoned-transaction overlays. diff --git a/nodedb/src/control/mod.rs b/nodedb/src/control/mod.rs index 149542c77..fb48758ae 100644 --- a/nodedb/src/control/mod.rs +++ b/nodedb/src/control/mod.rs @@ -32,6 +32,7 @@ pub mod metadata_proposer; pub mod metrics; pub mod mirror; pub mod notify_bus; +pub(crate) mod orchestrated_write; pub mod otel; pub mod pending_ddl; pub mod planner; diff --git a/nodedb/src/control/orchestrated_write.rs b/nodedb/src/control/orchestrated_write.rs new file mode 100644 index 000000000..bd3987632 --- /dev/null +++ b/nodedb/src/control/orchestrated_write.rs @@ -0,0 +1,140 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The one apply seam for a Control-Plane-orchestrated write (`MERGE`, +//! `UPDATE ... FROM`, `INSERT ... SELECT`): the resolved plan the orchestrator +//! built lands on its target vShard's owner and on every replica. +//! +//! On a cluster the plan proposes through Raft exactly as a plain replicated +//! write does; the proposer forwards to the group leader, so the coordinator +//! never has to own the vShard. Standalone, the plan dispatches to the local +//! Data Plane and mints the redo record the write funnel would have minted. + +use std::sync::atomic::Ordering; + +use nodedb_types::{DatabaseId, TenantId}; + +use crate::bridge::envelope::{ErrorCode, PhysicalPlan, Response, Status}; +use crate::control::maintenance::clone_materializer::dispatch_local; +use crate::control::server::dispatch_utils::publish_origin_change_events; +use crate::control::state::SharedState; +use crate::control::wal_replication::{ + ReplicableWrite, propose_replicated_entry, to_replicated_entry, +}; +use crate::types::{RequestId, VShardId}; + +/// Apply `plan`, a resolved write on `collection`, and return the Data-Plane +/// response the statement renders. +/// +/// A Data-Plane verdict that reached the proposer as `Error::DataPlane` comes +/// back as an error `Response` carrying that code, the same shape a local +/// dispatch returns, so a caller reads `OllpRetryRequired` the one way on +/// both paths. +pub(crate) async fn apply_orchestrated_write( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection: &str, + plan: PhysicalPlan, +) -> crate::Result { + let Some(proposer) = state.async_raft_proposer() else { + let resp = dispatch_local(state, tenant_id, database_id, collection, plan, None).await?; + // `dispatch_local` bypasses the funnel's post-apply redo minting, so a + // vector-indexed target's write-set arrives unconsumed. Without it a + // WAL-only restart rebuilds the index from pre-write records. No-op + // on a target with no write-set. + crate::control::server::wal_dispatch::mint_dispatch_local_redo( + &state.wal, + tenant_id, + database_id, + collection, + &resp, + )?; + return Ok(resp); + }; + + let vshard_id = VShardId::from_collection_in_database(database_id, collection); + let replicable = ReplicableWrite::decide_for_replication(&plan)?; + let entry = + to_replicated_entry(tenant_id, database_id, vshard_id, &replicable)?.ok_or_else(|| { + crate::Error::Internal { + detail: format!( + "orchestrated write on '{collection}' did not map to a replicated write; the \ + orchestrator must propose its resolved shape" + ), + } + })?; + let request_id = RequestId::new(state.request_id_counter.fetch_add(1, Ordering::Relaxed)); + match propose_replicated_entry(state, proposer, entry).await { + Ok((payload, write_version)) => { + let response = Response { + request_id, + status: Status::Ok, + attempt: 1, + partial: false, + payload: payload.into(), + watermark_lsn: write_version, + error_code: None, + read_set_valid: None, + read_version_lsn: write_version, + write_set: Vec::new(), + }; + // The proposing node handled this write exactly once, so it is + // the one node that publishes the CDC change event. + publish_origin_change_events(state, tenant_id, database_id, &plan, &response); + Ok(response) + } + Err(crate::Error::DataPlane(code)) => Ok(data_plane_verdict(request_id, code)), + Err(e) => Err(e), + } +} + +/// The error `Response` a local dispatch returns for a Data-Plane verdict. +fn data_plane_verdict(request_id: RequestId, code: ErrorCode) -> Response { + Response { + request_id, + status: Status::Error, + attempt: 1, + partial: false, + payload: crate::bridge::envelope::Payload::from_vec(Vec::new()), + watermark_lsn: crate::types::Lsn::ZERO, + error_code: Some(Box::new(code)), + read_set_valid: None, + read_version_lsn: crate::types::Lsn::ZERO, + write_set: Vec::new(), + } +} + +/// Decide the target's write policy over the row images an orchestrator +/// resolved, here on the proposing node, so the proposed plan carries a +/// decided check a follower can apply without the writing identity. +/// +/// `images` are the post-images of every row the write creates or rewrites +/// and the pre-images of every row it removes, MessagePack-encoded. A refused +/// image fails the statement before anything is written, on every replica. +pub(crate) fn decide_write_policy_over_images<'a>( + rls_write_check: &nodedb_types::RlsWriteCheck, + images: impl IntoIterator, + tenant_id: TenantId, + collection: &str, +) -> crate::Result<()> { + match rls_write_check.decision() { + nodedb_types::WriteGateDecision::AdmitAll => Ok(()), + nodedb_types::WriteGateDecision::Evaluate(predicate) => { + for image in images { + crate::control::security::rls::admit_compiled_write_image( + predicate, + image, + tenant_id.as_u64(), + collection, + )?; + } + Ok(()) + } + nodedb_types::WriteGateDecision::DenyNotInjected => Err(crate::Error::PlanError { + detail: format!( + "internal invariant break: the orchestrated write on '{collection}' reached its \ + apply before RLS injection decided its write-policy check" + ), + }), + } +} diff --git a/nodedb/src/control/planner/calvin/tx_class/shared.rs b/nodedb/src/control/planner/calvin/tx_class/shared.rs index 16c222ad3..dcce00baf 100644 --- a/nodedb/src/control/planner/calvin/tx_class/shared.rs +++ b/nodedb/src/control/planner/calvin/tx_class/shared.rs @@ -9,7 +9,7 @@ use nodedb_cluster::calvin::types::{ VersionedReadSet, }; use nodedb_physical::physical_plan::{ - DocumentOp, GraphOp, KvOp, PhysicalPlan, TimeseriesOp, VectorOp, + ColumnarOp, DocumentOp, GraphOp, KvOp, PhysicalPlan, TimeseriesOp, VectorOp, VectorWriteTargets, }; /// Map the neutral session read-set into the replicated, LSN-versioned @@ -167,11 +167,36 @@ pub(super) fn vector_write_surrogates(op: &VectorOp) -> Option<(String, Vec collection, surrogate, .. + } + | VectorOp::DirectInsert { + collection, + surrogate, + .. + } + | VectorOp::DirectInsertIfAbsent { + collection, + surrogate, + .. + } + | VectorOp::DirectUpsert { + collection, + surrogate, + .. } => Some((collection.to_string(), vec![surrogate.as_u32()])), VectorOp::BatchInsert { collection, surrogates, .. + } + | VectorOp::DirectDelete { + collection, + targets: VectorWriteTargets::Surrogates(surrogates), + .. + } + | VectorOp::DirectUpdate { + collection, + targets: VectorWriteTargets::Surrogates(surrogates), + .. } => Some(( collection.to_string(), surrogates.iter().map(|s| s.as_u32()).collect(), @@ -213,12 +238,16 @@ pub(crate) fn collection_name_from_plan(plan: &PhysicalPlan) -> String { VectorOp::Insert { collection, .. } | VectorOp::BatchInsert { collection, .. } | VectorOp::Delete { collection, .. } - | VectorOp::DeleteBySurrogate { collection, .. }, + | VectorOp::DeleteBySurrogate { collection, .. } + | VectorOp::DirectTruncate { collection, .. }, ) => collection.to_string(), PhysicalPlan::Graph( GraphOp::EdgePut { collection, .. } | GraphOp::EdgeDelete { collection, .. }, ) => collection.to_string(), - PhysicalPlan::Timeseries(TimeseriesOp::Ingest { collection, .. }) => { + PhysicalPlan::Timeseries( + TimeseriesOp::Ingest { collection, .. } | TimeseriesOp::Truncate { collection, .. }, + ) + | PhysicalPlan::Columnar(ColumnarOp::Truncate { collection, .. }) => { collection.to_string() } _ => String::new(), diff --git a/nodedb/src/control/planner/calvin/write_class.rs b/nodedb/src/control/planner/calvin/write_class.rs index 031cd1fd9..5aa7c3f65 100644 --- a/nodedb/src/control/planner/calvin/write_class.rs +++ b/nodedb/src/control/planner/calvin/write_class.rs @@ -92,6 +92,26 @@ pub fn is_derived_side_effect(plan: &PhysicalPlan) -> bool { } } +/// Whether a statement's plans carry the user's own write, as opposed to +/// only derived side effects. Decided over the FULL plan set: one slice or +/// one task alone cannot tell a lone derived participant from a +/// derived-only statement (the standalone `GRAPH INSERT EDGE` DSL). +pub fn plans_have_user_write<'a>(plans: impl IntoIterator) -> bool { + plans + .into_iter() + .any(|plan| is_write_plan(plan) && !is_derived_side_effect(plan)) +} + +/// Whether `plan`'s applied count answers the client's statement. +/// +/// The one rule every response fold shares with Calvin's deposit: when the +/// statement carries the user's own write, a derived participant's count +/// describes a row the statement never named and folds as opaque. When it +/// carries none, every write is the user's and counts. +pub fn plan_counts_toward_statement_tag(plan: &PhysicalPlan, has_user_write: bool) -> bool { + !(has_user_write && is_derived_side_effect(plan)) +} + fn kv_is_write(op: &KvOp) -> bool { match op { KvOp::Put { .. } @@ -154,8 +174,17 @@ fn vector_is_write(op: &VectorOp) -> bool { | VectorOp::SparseDelete { .. } | VectorOp::MultiVectorInsert { .. } | VectorOp::MultiVectorDelete { .. } - | VectorOp::DirectUpsert { .. } => true, - VectorOp::Search { .. } + | VectorOp::DirectUpsert { .. } + | VectorOp::DirectInsert { .. } + | VectorOp::DirectInsertIfAbsent { .. } + | VectorOp::DirectDelete { .. } + | VectorOp::DirectTruncate { .. } + | VectorOp::DirectUpdate { .. } + // Mutates the rows its mutation list names, like any other write. + | VectorOp::ResolvedDirectWrite { .. } => true, + // Read-only: reports what the wrapped write would apply, mutates nothing. + VectorOp::ResolveDirectWrite(_) + | VectorOp::Search { .. } | VectorOp::MultiSearch { .. } | VectorOp::QueryStats { .. } | VectorOp::SparseSearch { .. } @@ -202,7 +231,7 @@ fn graph_is_write(op: &GraphOp) -> bool { fn timeseries_is_write(op: &TimeseriesOp) -> bool { match op { - TimeseriesOp::Ingest { .. } => true, + TimeseriesOp::Ingest { .. } | TimeseriesOp::Truncate { .. } => true, // The resolve pass writes nothing; the ingest it reports is proposed // separately by the write-resolve orchestrator. TimeseriesOp::ResolveIngest(_) | TimeseriesOp::Scan { .. } => false, @@ -215,7 +244,8 @@ fn columnar_is_write(op: &ColumnarOp) -> bool { | ColumnarOp::Update { .. } | ColumnarOp::Delete { .. } | ColumnarOp::ResolvedUpdate { .. } - | ColumnarOp::ResolvedDelete { .. } => true, + | ColumnarOp::ResolvedDelete { .. } + | ColumnarOp::Truncate { .. } => true, // Read-only: decides the write policy but mutates nothing, so no // vshard lock to take. ColumnarOp::Scan { .. } @@ -401,6 +431,7 @@ mod tests { fn is_write_plan_true_for_kv_truncate() { let plan = PhysicalPlan::Kv(KvOp::Truncate { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "cache"), + restart_identity: false, }); assert!(is_write_plan(&plan), "KvOp::Truncate must be a write"); } @@ -528,6 +559,7 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "vecs"), field: "emb".to_owned(), surrogate: Surrogate::new(3), + pk_bytes: Vec::new(), vector: vec![0.5, 0.6], payload: vec![1, 2, 3], quantization: VectorQuantization::None, @@ -535,6 +567,8 @@ mod tests { payload_indexes: vec![("tenant_id".to_owned(), PayloadIndexKind::Equality)], returning: None, rls_filters: Vec::new(), + on_conflict_updates: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::decided_earlier_in_request(), }); assert!( is_write_plan(&plan), diff --git a/nodedb/src/control/planner/context/query/context.rs b/nodedb/src/control/planner/context/query/context.rs index c9f96e67f..f9fdc30db 100644 --- a/nodedb/src/control/planner/context/query/context.rs +++ b/nodedb/src/control/planner/context/query/context.rs @@ -134,14 +134,16 @@ impl QueryContext { } /// Create a query context from `SharedState` without lease - /// integration. Used by internal sub-planners (check - /// constraints, type guards, ANALYZE, procedural DML, event - /// trigger dispatch) that run inside a pgwire handler whose - /// outer query already acquired leases. Re-acquiring via a - /// sub-planner would be redundant — the lease store's fast - /// path would return instantly anyway, but going through the - /// sub-planner without a direct `Arc` reference - /// would require threading one through every call site. + /// integration. Used by internal sub-planners (neutral DDL + /// readback queries — check constraints, type guards, ANALYZE, + /// COPY TO, materialized view refresh — plus procedural DML, + /// event trigger dispatch, and graph scatter-gather hops) that + /// run inside a handler whose outer query already acquired + /// leases. Re-acquiring via a sub-planner would be redundant — + /// the lease store's fast path would return instantly anyway, + /// but going through the sub-planner without a direct + /// `Arc` reference would require threading one + /// through every call site. pub fn for_state(state: &crate::control::state::SharedState) -> Self { let mut ctx = Self::with_catalog( Arc::clone(&state.credentials), diff --git a/nodedb/src/control/planner/rls_injection/columnar.rs b/nodedb/src/control/planner/rls_injection/columnar.rs index 91ad40bf4..9bbeb1d83 100644 --- a/nodedb/src/control/planner/rls_injection/columnar.rs +++ b/nodedb/src/control/planner/rls_injection/columnar.rs @@ -62,6 +62,13 @@ pub(super) fn inject_columnar(ctx: &RlsCtx<'_>, op: &mut ColumnarOp) -> crate::R ColumnarOp::ResolvedUpdate { .. } | ColumnarOp::ResolvedDelete { .. } | ColumnarOp::ResolveDml { .. } => Ok(()), + + // Refuse: removes every row without reading one, so no image + // exists to evaluate against. Mirrors `KvOp::Truncate`. + ColumnarOp::Truncate { collection, .. } => ctx.refuse_if_write_policy( + collection, + "a truncate removes every row without reading one, so no row image is available", + ), } } @@ -91,6 +98,13 @@ pub(super) fn inject_timeseries(ctx: &RlsCtx<'_>, op: &mut TimeseriesOp) -> crat // Recurse: the resolve pass carries the ingest it is about to decide, // and that ingest's own slots are the ones the policy fills. TimeseriesOp::ResolveIngest(inner) => inject_timeseries(ctx, inner), + + // Refuse: removes every row without reading one, so no image + // exists to evaluate against. Mirrors `KvOp::Truncate`. + TimeseriesOp::Truncate { collection, .. } => ctx.refuse_if_write_policy( + collection, + "a truncate removes every row without reading one, so no row image is available", + ), } } @@ -170,6 +184,45 @@ mod tests { }) } + fn columnar_truncate(collection: &str) -> PhysicalPlan { + PhysicalPlan::Columnar(ColumnarOp::Truncate { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + collection, + ), + restart_identity: false, + }) + } + + fn timeseries_truncate(collection: &str) -> PhysicalPlan { + PhysicalPlan::Timeseries(TimeseriesOp::Truncate { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + collection, + ), + restart_identity: false, + }) + } + + /// A truncate reads no row, so nothing exists for the write policy to + /// decide against; it is refused like `KvOp::Truncate`. + #[test] + fn columnar_family_truncate_is_refused_under_a_write_policy() { + let store = store_with_write_policy("docs"); + for mut plan in [columnar_truncate("docs"), timeseries_truncate("docs")] { + assert_write_refused(inject(&mut plan, &store), "docs"); + } + } + + #[test] + fn columnar_family_truncate_without_a_policy_is_untouched() { + for mut plan in [columnar_truncate("docs"), timeseries_truncate("docs")] { + let before = plan.clone(); + assert!(inject_without_policy(&mut plan).is_ok()); + assert_eq!(plan, before); + } + } + fn ingest(collection: &str, format: &str) -> PhysicalPlan { PhysicalPlan::Timeseries(TimeseriesOp::Ingest { collection: nodedb_types::QualifiedCollection::new( diff --git a/nodedb/src/control/planner/rls_injection/crdt.rs b/nodedb/src/control/planner/rls_injection/crdt.rs index 8b918a237..7f80490f5 100644 --- a/nodedb/src/control/planner/rls_injection/crdt.rs +++ b/nodedb/src/control/planner/rls_injection/crdt.rs @@ -174,6 +174,7 @@ mod tests { fields_json: "{}".into(), surrogate: nodedb_types::Surrogate::ZERO, partial: false, + verb: nodedb_physical::physical_plan::CrdtWriteVerb::Insert, returning: None, rls_filters: Vec::new(), }); diff --git a/nodedb/src/control/planner/rls_injection/document.rs b/nodedb/src/control/planner/rls_injection/document.rs index dde01115f..e4e03f771 100644 --- a/nodedb/src/control/planner/rls_injection/document.rs +++ b/nodedb/src/control/planner/rls_injection/document.rs @@ -387,6 +387,7 @@ mod tests { clauses: Vec::new(), returning: None, resolved_inserts: None, + resolved_insert_identities: Vec::new(), source_rows: None, rls_filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), @@ -649,6 +650,7 @@ mod tests { clauses: Vec::new(), returning: None, resolved_inserts: None, + resolved_insert_identities: Vec::new(), source_rows: None, rls_filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), diff --git a/nodedb/src/control/planner/rls_injection/kv.rs b/nodedb/src/control/planner/rls_injection/kv.rs index 1bc53d6d8..fdc89f98e 100644 --- a/nodedb/src/control/planner/rls_injection/kv.rs +++ b/nodedb/src/control/planner/rls_injection/kv.rs @@ -618,6 +618,7 @@ mod tests { nodedb_types::DatabaseId::DEFAULT, "sessions", ), + restart_identity: false, }); assert_write_refused(inject(&mut plan, &store), "sessions"); } diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/columnar.rs b/nodedb/src/control/planner/rls_injection/permission_tree/columnar.rs index fa1c4d181..2d978456d 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/columnar.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/columnar.rs @@ -57,7 +57,9 @@ pub(super) fn apply_columnar(ctx: &PermCtx<'_>, op: &mut ColumnarOp) -> crate::R ColumnarOp::ResolvedUpdate { collection, .. } => { ctx.authorize(collection, PermTreeLevel::Write) } - ColumnarOp::ResolvedDelete { collection, .. } => { + // Delete level, blanket: a truncate removes rows it never + // enumerates, so there is no predicate to narrow. + ColumnarOp::ResolvedDelete { collection, .. } | ColumnarOp::Truncate { collection, .. } => { ctx.authorize(collection, PermTreeLevel::Delete) } @@ -87,6 +89,12 @@ pub(super) fn apply_timeseries(ctx: &PermCtx<'_>, op: &mut TimeseriesOp) -> crat // Recurse: the resolve pass stands in for the ingest it wraps, so it // needs the same write level on the same collection. TimeseriesOp::ResolveIngest(inner) => apply_timeseries(ctx, inner), + + // Delete level, blanket: a truncate removes rows it never + // enumerates, so there is no predicate to narrow. + TimeseriesOp::Truncate { collection, .. } => { + ctx.authorize(collection, PermTreeLevel::Delete) + } } } diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/kv.rs b/nodedb/src/control/planner/rls_injection/permission_tree/kv.rs index c72cfdad4..7d60dc9be 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/kv.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/kv.rs @@ -77,7 +77,7 @@ pub(super) fn apply_kv(ctx: &PermCtx<'_>, op: &mut KvOp) -> crate::Result<()> { // does not enumerate — by key, by predicate, or wholesale. KvOp::Delete { collection, .. } | KvOp::PredicateDelete { collection, .. } - | KvOp::Truncate { collection } => ctx.authorize(collection, PermTreeLevel::Delete), + | KvOp::Truncate { collection, .. } => ctx.authorize(collection, PermTreeLevel::Delete), // Blanket both levels: delete on source, write on destination. KvOp::TransferItem { diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/vector.rs b/nodedb/src/control/planner/rls_injection/permission_tree/vector.rs index c3a38c524..3a7d76ae5 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/vector.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/vector.rs @@ -2,7 +2,7 @@ //! Permission-tree resolution for vector-engine operations. -use nodedb_physical::physical_plan::VectorOp; +use nodedb_physical::physical_plan::{VectorOp, VectorResolvedMutation}; use super::context::{PermCtx, PermTreeLevel}; use super::plan::walk; @@ -60,19 +60,45 @@ pub(super) fn apply_vector(ctx: &PermCtx<'_>, op: &mut VectorOp) -> crate::Resul | VectorOp::BatchInsert { collection, .. } | VectorOp::SparseInsert { collection, .. } | VectorOp::MultiVectorInsert { collection, .. } - | VectorOp::DirectUpsert { collection, .. } => { + | VectorOp::DirectUpsert { collection, .. } + | VectorOp::DirectInsert { collection, .. } + | VectorOp::DirectInsertIfAbsent { collection, .. } + | VectorOp::DirectUpdate { collection, .. } => { ctx.authorize(collection, PermTreeLevel::Write) } // Filter (delete level, blanket): index deletions remove the row's // entry from the index. VectorOp::Delete { collection, .. } + | VectorOp::DirectDelete { collection, .. } + | VectorOp::DirectTruncate { collection, .. } | VectorOp::DeleteBySurrogate { collection, .. } | VectorOp::SparseDelete { collection, .. } | VectorOp::MultiVectorDelete { collection, .. } => { ctx.authorize(collection, PermTreeLevel::Delete) } + // Recurse: the wrapped op is the intercepted write verbatim. + VectorOp::ResolveDirectWrite(inner) => apply_vector(ctx, inner), + + // Blanket per mutation: each names the row it writes directly, and a + // `Delete` mutation is a removal, so it takes the delete level. + VectorOp::ResolvedDirectWrite { + collection, + mutations, + .. + } => { + for mutation in mutations { + let level = match mutation { + VectorResolvedMutation::Update { .. } + | VectorResolvedMutation::Upsert { .. } => PermTreeLevel::Write, + VectorResolvedMutation::Delete { .. } => PermTreeLevel::Delete, + }; + ctx.authorize(collection, level)?; + } + Ok(()) + } + // No-op: index configuration and index maintenance. They act on the // index structure rather than on rows, and are authorized as DDL. VectorOp::SetParams { .. } diff --git a/nodedb/src/control/planner/rls_injection/vector.rs b/nodedb/src/control/planner/rls_injection/vector.rs index 563008727..0bd93d9f6 100644 --- a/nodedb/src/control/planner/rls_injection/vector.rs +++ b/nodedb/src/control/planner/rls_injection/vector.rs @@ -69,7 +69,13 @@ pub(super) fn inject_vector(ctx: &RlsCtx<'_>, op: &mut VectorOp) -> crate::Resul // `payload` is a zerompk `HashMap`, whose values are // TAGGED, so it goes through the transcoding admission rather than the // raw-image one — see [`RlsCtx::admit_write_value_map_image`]. - VectorOp::DirectUpsert { + VectorOp::DirectInsert { + collection, + payload, + rls_filters, + .. + } + | VectorOp::DirectInsertIfAbsent { collection, payload, rls_filters, @@ -86,6 +92,46 @@ pub(super) fn inject_vector(ctx: &RlsCtx<'_>, op: &mut VectorOp) -> crate::Resul ctx.set_post_filters(collection, rls_filters) } + // Admit the proposed image here; a conflict patch produces its merged + // image on the Data Plane, so the predicate ships for that path. + VectorOp::DirectUpsert { + collection, + payload, + rls_filters, + on_conflict_updates, + rls_write_check, + .. + } => { + ctx.admit_write_value_map_image(collection, payload)?; + if on_conflict_updates.is_empty() { + *rls_write_check = nodedb_types::RlsWriteCheck::decided_earlier_in_request(); + } else { + ctx.set_write_check(collection, rls_write_check)?; + } + ctx.set_post_filters(collection, rls_filters) + } + + // Ship the predicate and gate `RETURNING` as a read: the removed row + // or the patched post-image exists only where it is persisted, and + // the rows the clause hands back are bounded by the same read policy a + // `SELECT` by this principal is. Mirrors `KvOp::{Delete, + // PredicateUpdate}`. + VectorOp::DirectDelete { + collection, + rls_write_check, + rls_filters, + .. + } + | VectorOp::DirectUpdate { + collection, + rls_write_check, + rls_filters, + .. + } => { + ctx.set_write_check(collection, rls_write_check)?; + ctx.set_post_filters(collection, rls_filters) + } + // Refuse: these carry an embedding, a surrogate, or an opaque document // id — never the row body a policy predicate names — so no image is // available for the write policy to be evaluated against. A vector @@ -109,6 +155,20 @@ pub(super) fn inject_vector(ctx: &RlsCtx<'_>, op: &mut VectorOp) -> crate::Resul policy names, so no row image is available for it to be evaluated against", ), + // Refuse: removes every row without reading one, so no image + // exists to evaluate against. Mirrors `KvOp::Truncate`. + VectorOp::DirectTruncate { collection, .. } => ctx.refuse_if_write_policy( + collection, + "a truncate removes every row without reading one, so no row image is available", + ), + + // No-op: already decided by the resolve pass; re-injecting would + // replace a verdict with a predicate no applying node can decide. + VectorOp::ResolvedDirectWrite { .. } => Ok(()), + + // Recurse: the wrapped op is the intercepted write verbatim. + VectorOp::ResolveDirectWrite(inner) => inject_vector(ctx, inner), + // No-op: index parameters and index maintenance write no user row. VectorOp::SetParams { .. } | VectorOp::DropIndex { .. } @@ -162,6 +222,34 @@ mod tests { assert_eq!(plan, before); } + fn vector_truncate(collection: &str) -> PhysicalPlan { + PhysicalPlan::Vector(VectorOp::DirectTruncate { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + collection, + ), + field: "vec".into(), + restart_identity: false, + }) + } + + /// A truncate reads no row, so nothing exists for the write policy to + /// decide against; it is refused like `KvOp::Truncate`. + #[test] + fn vector_truncate_is_refused_under_a_write_policy() { + let store = store_with_write_policy("docs"); + let mut plan = vector_truncate("docs"); + assert_write_refused(inject(&mut plan, &store), "docs"); + } + + #[test] + fn vector_truncate_without_a_policy_is_untouched() { + let mut plan = vector_truncate("docs"); + let before = plan.clone(); + assert!(inject_without_policy(&mut plan).is_ok()); + assert_eq!(plan, before); + } + fn search_with_prefilter(collection: &str, prefilter: Option) -> PhysicalPlan { PhysicalPlan::Vector(VectorOp::Search { collection: nodedb_types::QualifiedCollection::new( @@ -218,6 +306,7 @@ mod tests { ), field: "emb".into(), surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: Vec::new(), vector: vec![0.1, 0.2], payload, quantization: Default::default(), @@ -225,6 +314,8 @@ mod tests { payload_indexes: Vec::new(), returning: None, rls_filters: Vec::new(), + on_conflict_updates: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), }) } @@ -266,12 +357,19 @@ mod tests { )); } - /// With no write policy the same upsert runs untouched. + /// With no write policy the same upsert runs with its payload, filters, + /// and patch untouched; only the write-check slot is stamped as decided. #[test] fn direct_upsert_without_a_policy_is_untouched() { let mut plan = direct_upsert("docs", &[("region", "eu")]); - let before = plan.clone(); + let mut before = plan.clone(); assert!(inject_without_policy(&mut plan).is_ok()); + if let PhysicalPlan::Vector(VectorOp::DirectUpsert { + rls_write_check, .. + }) = &mut before + { + *rls_write_check = nodedb_types::RlsWriteCheck::decided_earlier_in_request(); + } assert_eq!(plan, before); } diff --git a/nodedb/src/control/planner/sql_plan_convert/body.rs b/nodedb/src/control/planner/sql_plan_convert/body.rs index 38720030c..56ed45ad3 100644 --- a/nodedb/src/control/planner/sql_plan_convert/body.rs +++ b/nodedb/src/control/planner/sql_plan_convert/body.rs @@ -72,6 +72,7 @@ pub(super) fn convert_body_to_single_plan( | SqlPlan::UpdateFrom { .. } | SqlPlan::Delete { .. } | SqlPlan::Truncate { .. } + | SqlPlan::VectorPrimaryTruncate { .. } | SqlPlan::Join { .. } | SqlPlan::Aggregate { .. } | SqlPlan::TimeseriesScan { .. } @@ -102,6 +103,8 @@ pub(super) fn convert_body_to_single_plan( | SqlPlan::LateralTopK { .. } | SqlPlan::LateralLoop { .. } | SqlPlan::VectorPrimaryInsert { .. } + | SqlPlan::VectorPrimaryDelete { .. } + | SqlPlan::VectorPrimaryUpdate { .. } | SqlPlan::CreateIndex { .. } | SqlPlan::DropIndex { .. } => { let mut tasks = convert_one(input, tenant_id, ctx)?; diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs index 970986c4e..332605e35 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs @@ -126,6 +126,7 @@ pub(in super::super::super) fn convert_insert( fields_json: super::super::crdt_gate::row_to_fields_json(row)?, surrogate, partial: false, + verb: CrdtWriteVerb::Insert, returning: None, rls_filters: Vec::new(), }) diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/kv_and_vector.rs b/nodedb/src/control/planner/sql_plan_convert/dml/kv_and_vector.rs deleted file mode 100644 index b317b82f4..000000000 --- a/nodedb/src/control/planner/sql_plan_convert/dml/kv_and_vector.rs +++ /dev/null @@ -1,304 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -use nodedb_sql::types::{KvInsertIntent, SqlExpr, SqlValue, VectorPrimaryRow}; - -use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; -use nodedb_physical::physical_plan::*; - -use super::super::convert::ConvertContext; -use super::super::value::{ - assignments_to_update_values, sql_value_to_bytes, sql_value_to_nodedb_value, - write_msgpack_map_header, write_msgpack_str, write_msgpack_value, -}; -use super::insert::assign_for_pk; -use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; - -pub(in super::super) fn convert_kv_insert( - collection: &str, - entries: &[(SqlValue, Vec<(String, SqlValue)>)], - ttl_secs: u64, - intent: KvInsertIntent, - on_conflict_updates: &[(String, SqlExpr)], - tenant_id: TenantId, - ctx: &ConvertContext, -) -> crate::Result> { - let coll_qualified = super::super::convert::db_qualified(ctx.database_id, collection); - let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); - let collection = coll_qualified.as_str(); - let update_values = if on_conflict_updates.is_empty() { - Vec::new() - } else { - assignments_to_update_values(on_conflict_updates)? - }; - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); - let ttl_ms = ttl_secs * 1000; - let mut tasks = Vec::with_capacity(entries.len()); - for (key_val, value_cols) in entries { - // A declared PRIMARY KEY implies NOT NULL. The planner substitutes - // `SqlValue::Null` for a column the statement omitted, so this also - // catches an omitted key, not only an explicit `NULL` literal. - if matches!(key_val, SqlValue::Null) { - return Err(crate::Error::RejectedConstraint { - collection: collection.to_string(), - constraint: "not_null".to_string(), - detail: "primary key cannot be NULL or omitted".to_string(), - }); - } - let key = sql_value_to_bytes(key_val); - let value = if value_cols.len() == 1 && value_cols[0].0 == "value" { - sql_value_to_bytes(&value_cols[0].1) - } else { - let mut buf = Vec::with_capacity(value_cols.len() * 32); - write_msgpack_map_header(&mut buf, value_cols.len()); - for (col, val) in value_cols { - write_msgpack_str(&mut buf, col); - write_msgpack_value(&mut buf, val); - } - buf - }; - let surrogate = assign_for_pk(ctx, collection, &key)?; - let op = match intent { - KvInsertIntent::Insert => KvOp::Insert { - collection: qualified_collection.clone(), - key, - value, - ttl_ms, - surrogate, - // Both filled in after conversion: the RETURNING spec by - // the protocol layer's injection pass, the read filter by the - // RLS injection pass. - returning: None, - rls_filters: Vec::new(), - }, - KvInsertIntent::InsertIfAbsent => KvOp::InsertIfAbsent { - collection: qualified_collection.clone(), - key, - value, - ttl_ms, - surrogate, - // Both filled in after conversion: the RETURNING spec by - // the protocol layer's injection pass, the read filter by the - // RLS injection pass. - returning: None, - rls_filters: Vec::new(), - }, - KvInsertIntent::Put if !update_values.is_empty() => KvOp::InsertOnConflictUpdate { - collection: qualified_collection.clone(), - key, - value, - ttl_ms, - updates: update_values.clone(), - surrogate, - // Filled by the RLS injection pass, which runs after plan - // conversion. - rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), - // Both filled in after conversion: the RETURNING spec by - // the protocol layer's injection pass, the read filter by the - // RLS injection pass. - returning: None, - rls_filters: Vec::new(), - }, - KvInsertIntent::Put => KvOp::Put { - collection: qualified_collection.clone(), - key, - value, - ttl_ms, - surrogate, - // Both filled in after conversion: the RETURNING spec by - // the protocol layer's injection pass, the read filter by the - // RLS injection pass. - returning: None, - rls_filters: Vec::new(), - }, - }; - tasks.push(PhysicalTask { - tenant_id, - vshard_id: vshard, - database_id: ctx.database_id, - plan: PhysicalPlan::Kv(op), - post_set_op: PostSetOp::None, - txn_id: None, - }); - } - Ok(tasks) -} - -pub(in super::super) struct VectorPrimaryInsertCfg<'a> { - pub field: &'a str, - pub quantization: nodedb_types::VectorQuantization, - pub storage_dtype: nodedb_types::VectorStorageDtype, - pub payload_indexes: &'a [(String, nodedb_types::PayloadIndexKind)], -} - -pub(in super::super) fn convert_vector_primary_insert( - collection: &str, - cfg: &VectorPrimaryInsertCfg<'_>, - rows: &[VectorPrimaryRow], - tenant_id: TenantId, - ctx: &ConvertContext, -) -> crate::Result> { - let coll_qualified = super::super::convert::db_qualified(ctx.database_id, collection); - let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); - let mut tasks = Vec::with_capacity(rows.len()); - for row in rows { - // Enforce per-tenant vector dimension quota before building any task. - // 0 means unlimited. - if ctx.max_vector_dim > 0 { - let dim = row.vector.len() as u32; - if dim > ctx.max_vector_dim { - return Err(crate::Error::TenantVectorDimExceeded { - dim, - limit: ctx.max_vector_dim, - }); - } - } - let pk_bytes: Vec = row - .vector - .iter() - .take(4) - .flat_map(|f| f.to_le_bytes()) - .collect(); - let surrogate = assign_for_pk(ctx, collection, &pk_bytes)?; - - let payload = if row.payload_fields.is_empty() { - Vec::new() - } else { - let value_map: std::collections::HashMap = row - .payload_fields - .iter() - .map(|(k, v)| (k.clone(), sql_value_to_nodedb_value(v))) - .collect(); - zerompk::to_msgpack_vec(&value_map).map_err(|e| crate::Error::Serialization { - format: "msgpack".into(), - detail: format!("vector-primary payload: {e}"), - })? - }; - - tasks.push(PhysicalTask { - tenant_id, - vshard_id: vshard, - database_id: ctx.database_id, - plan: PhysicalPlan::Vector(VectorOp::DirectUpsert { - collection: qualified_collection.clone(), - field: cfg.field.to_string(), - surrogate, - vector: row.vector.clone(), - payload, - quantization: cfg.quantization, - storage_dtype: cfg.storage_dtype, - payload_indexes: cfg.payload_indexes.to_vec(), - // Both filled in after conversion: the RETURNING spec by the - // protocol layer's injection pass, the read filter by the RLS - // injection pass. - returning: None, - rls_filters: Vec::new(), - }), - post_set_op: PostSetOp::None, - txn_id: None, - }); - } - Ok(tasks) -} - -#[cfg(test)] -mod tests { - use super::super::super::convert::ConvertContext; - use nodedb_sql::types::VectorPrimaryRow; - use nodedb_types::VectorQuantization; - - fn make_ctx(max_vector_dim: u32) -> ConvertContext { - ConvertContext { - purpose: super::super::super::convert::PlanningPurpose::Execute, - retention_registry: None, - array_catalog: None, - credentials: None, - wal: None, - surrogate_assigner: None, - cluster_enabled: false, - bitemporal_retention_registry: None, - max_vector_dim, - force_shuffle_join: false, - shuffle_num_parts: 0, - force_shuffle_agg: false, - shuffle_agg_num_parts: 0, - broadcast_threshold_bytes: 8 * 1024 * 1024, - shuffle_agg_threshold: 10_000, - database_id: crate::types::DatabaseId::DEFAULT, - tenant_id: crate::types::TenantId::new(0), - } - } - - fn row(dim: usize) -> VectorPrimaryRow { - VectorPrimaryRow { - surrogate: nodedb_types::Surrogate::ZERO, - vector: vec![0.0f32; dim], - payload_fields: std::collections::HashMap::new(), - } - } - - #[test] - fn tenant_vector_dim_under_bound_succeeds() { - let ctx = make_ctx(128); - let rows = vec![row(64), row(128)]; - let result = super::convert_vector_primary_insert( - "vecs", - &super::VectorPrimaryInsertCfg { - field: "emb", - quantization: VectorQuantization::None, - storage_dtype: nodedb_types::VectorStorageDtype::F32, - payload_indexes: &[], - }, - &rows, - crate::types::TenantId::new(1), - &ctx, - ); - assert!(result.is_ok(), "dimensions under/at cap must succeed"); - } - - #[test] - fn tenant_vector_dim_exceeded_rejected() { - let ctx = make_ctx(64); - let rows = vec![row(65)]; - let result = super::convert_vector_primary_insert( - "vecs", - &super::VectorPrimaryInsertCfg { - field: "emb", - quantization: VectorQuantization::None, - storage_dtype: nodedb_types::VectorStorageDtype::F32, - payload_indexes: &[], - }, - &rows, - crate::types::TenantId::new(1), - &ctx, - ); - match result { - Err(crate::Error::TenantVectorDimExceeded { dim, limit }) => { - assert_eq!(dim, 65); - assert_eq!(limit, 64); - } - other => panic!("expected TenantVectorDimExceeded, got {other:?}"), - } - } - - #[test] - fn tenant_vector_dim_zero_means_unlimited() { - let ctx = make_ctx(0); // 0 = unlimited - let rows = vec![row(99999)]; - let result = super::convert_vector_primary_insert( - "vecs", - &super::VectorPrimaryInsertCfg { - field: "emb", - quantization: VectorQuantization::None, - storage_dtype: nodedb_types::VectorStorageDtype::F32, - payload_indexes: &[], - }, - &rows, - crate::types::TenantId::new(1), - &ctx, - ); - assert!(result.is_ok(), "limit=0 means unlimited, must succeed"); - } -} diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/kv_insert.rs b/nodedb/src/control/planner/sql_plan_convert/dml/kv_insert.rs new file mode 100644 index 000000000..7fb7e237d --- /dev/null +++ b/nodedb/src/control/planner/sql_plan_convert/dml/kv_insert.rs @@ -0,0 +1,127 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `SqlPlan::KvInsert` → `PhysicalTask` lowering. + +use nodedb_sql::types::{KvInsertIntent, SqlExpr, SqlValue}; + +use crate::bridge::envelope::PhysicalPlan; +use crate::types::{TenantId, VShardId}; +use nodedb_physical::physical_plan::*; + +use super::super::convert::ConvertContext; +use super::super::value::{ + assignments_to_update_values, sql_value_to_bytes, write_msgpack_map_header, write_msgpack_str, + write_msgpack_value, +}; +use super::insert::assign_for_pk; +use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; + +pub(in super::super) fn convert_kv_insert( + collection: &str, + entries: &[(SqlValue, Vec<(String, SqlValue)>)], + ttl_secs: u64, + intent: KvInsertIntent, + on_conflict_updates: &[(String, SqlExpr)], + tenant_id: TenantId, + ctx: &ConvertContext, +) -> crate::Result> { + let coll_qualified = super::super::convert::db_qualified(ctx.database_id, collection); + let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); + let collection = coll_qualified.as_str(); + let update_values = if on_conflict_updates.is_empty() { + Vec::new() + } else { + assignments_to_update_values(on_conflict_updates)? + }; + let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let ttl_ms = ttl_secs * 1000; + let mut tasks = Vec::with_capacity(entries.len()); + for (key_val, value_cols) in entries { + // A declared PRIMARY KEY implies NOT NULL. The planner substitutes + // `SqlValue::Null` for a column the statement omitted, so this also + // catches an omitted key, not only an explicit `NULL` literal. + if matches!(key_val, SqlValue::Null) { + return Err(crate::Error::RejectedConstraint { + collection: collection.to_string(), + constraint: "not_null".to_string(), + detail: "primary key cannot be NULL or omitted".to_string(), + }); + } + let key = sql_value_to_bytes(key_val)?; + let value = if value_cols.len() == 1 && value_cols[0].0 == "value" { + sql_value_to_bytes(&value_cols[0].1)? + } else { + let mut buf = Vec::with_capacity(value_cols.len() * 32); + write_msgpack_map_header(&mut buf, value_cols.len()); + for (col, val) in value_cols { + write_msgpack_str(&mut buf, col); + write_msgpack_value(&mut buf, val); + } + buf + }; + let surrogate = assign_for_pk(ctx, collection, &key)?; + let op = match intent { + KvInsertIntent::Insert => KvOp::Insert { + collection: qualified_collection.clone(), + key, + value, + ttl_ms, + surrogate, + // Both filled in after conversion: the RETURNING spec by + // the protocol layer's injection pass, the read filter by the + // RLS injection pass. + returning: None, + rls_filters: Vec::new(), + }, + KvInsertIntent::InsertIfAbsent => KvOp::InsertIfAbsent { + collection: qualified_collection.clone(), + key, + value, + ttl_ms, + surrogate, + // Both filled in after conversion: the RETURNING spec by + // the protocol layer's injection pass, the read filter by the + // RLS injection pass. + returning: None, + rls_filters: Vec::new(), + }, + KvInsertIntent::Put if !update_values.is_empty() => KvOp::InsertOnConflictUpdate { + collection: qualified_collection.clone(), + key, + value, + ttl_ms, + updates: update_values.clone(), + surrogate, + // Filled by the RLS injection pass, which runs after plan + // conversion. + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // Both filled in after conversion: the RETURNING spec by + // the protocol layer's injection pass, the read filter by the + // RLS injection pass. + returning: None, + rls_filters: Vec::new(), + }, + KvInsertIntent::Put => KvOp::Put { + collection: qualified_collection.clone(), + key, + value, + ttl_ms, + surrogate, + // Both filled in after conversion: the RETURNING spec by + // the protocol layer's injection pass, the read filter by the + // RLS injection pass. + returning: None, + rls_filters: Vec::new(), + }, + }; + tasks.push(PhysicalTask { + tenant_id, + vshard_id: vshard, + database_id: ctx.database_id, + plan: PhysicalPlan::Kv(op), + post_set_op: PostSetOp::None, + txn_id: None, + }); + } + Ok(tasks) +} diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/merge.rs b/nodedb/src/control/planner/sql_plan_convert/dml/merge.rs index 73d0b9b51..9d67da961 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/merge.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/merge.rs @@ -93,6 +93,7 @@ pub(in super::super) fn convert_merge( // in `ResolveWrite`. This neutral form never reaches the Data // Plane directly. resolved_inserts: None, + resolved_insert_identities: Vec::new(), // The source rows are shipped in by the Control-Plane orchestrator // (cross-core source-ship); the neutral plan carries none. source_rows: None, diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/mod.rs b/nodedb/src/control/planner/sql_plan_convert/dml/mod.rs index 42d087059..675ef2f79 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/mod.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/mod.rs @@ -3,18 +3,22 @@ mod balanced_gate; mod crdt_gate; mod insert; -mod kv_and_vector; +mod kv_insert; mod merge; mod update_delete; mod upsert; +mod vector_primary; pub(super) use insert::{ConvertInsertArgs, convert_insert, declared_primary_key_name}; pub(crate) use insert::{DEFAULT_IDENTITY_COLUMN, build_columnar_schema}; -pub(super) use kv_and_vector::{ - VectorPrimaryInsertCfg, convert_kv_insert, convert_vector_primary_insert, -}; +pub(super) use kv_insert::convert_kv_insert; pub(super) use merge::{ConvertMergeArgs, convert_merge}; pub(super) use update_delete::{ UpdateFromParams, UpdateParams, convert_delete, convert_update, convert_update_from, }; pub(super) use upsert::{ConvertUpsertArgs, convert_upsert}; +pub(super) use vector_primary::{ + VectorPrimaryCfg, VectorPrimaryInsertArgs, VectorPrimaryUpdateArgs, + convert_vector_primary_delete, convert_vector_primary_insert, convert_vector_primary_truncate, + convert_vector_primary_update, +}; diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs index 26bd958f3..623db1e52 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs @@ -54,7 +54,10 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_delete( txn_id: None, }]); } - let keys: Vec> = target_keys.iter().map(sql_value_to_bytes).collect(); + let keys: Vec> = target_keys + .iter() + .map(sql_value_to_bytes) + .collect::>()?; return Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs index 3f31e4027..96138c013 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs @@ -119,7 +119,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update( } }) .collect(); - let key_bytes = sql_value_to_bytes(key); + let key_bytes = sql_value_to_bytes(key)?; // Content-addressed identity: keeps the surrogate the original insert assigned. // `Surrogate::ZERO` only when no assigner is wired (test / embedded-without-catalog). let surrogate = ctx.surrogate_for_pk(collection, &key_bytes)?; @@ -272,6 +272,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update( fields_json: fields_json.clone(), surrogate, partial: true, + verb: CrdtWriteVerb::Update, returning: None, rls_filters: Vec::new(), }) diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs b/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs index b93b0fed7..f37ecf15a 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs @@ -103,6 +103,7 @@ pub(in super::super) fn convert_upsert( fields_json: super::crdt_gate::row_to_fields_json(row)?, surrogate, partial: false, + verb: CrdtWriteVerb::Upsert, returning: None, rls_filters: Vec::new(), }) diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/vector_primary.rs b/nodedb/src/control/planner/sql_plan_convert/dml/vector_primary.rs new file mode 100644 index 000000000..a79cde8a2 --- /dev/null +++ b/nodedb/src/control/planner/sql_plan_convert/dml/vector_primary.rs @@ -0,0 +1,498 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `SqlPlan::VectorPrimary{Insert,Delete,Truncate,Update}` → `PhysicalTask` lowering. +//! +//! A vector-primary row is keyed by its declared primary key, through the +//! same identity path a document row takes: the key content-addresses a +//! surrogate, the surrogate keys the HNSW node and the payload sidecar. A +//! point `DELETE` / `UPDATE` resolves its keys read-only, so a key this +//! statement never created mints no binding. + +use nodedb_sql::types::{Filter, SqlExpr, SqlValue, VectorPrimaryInsertIntent, VectorPrimaryRow}; +use nodedb_types::{RlsWriteCheck, Surrogate}; + +use crate::bridge::envelope::PhysicalPlan; +use crate::types::{TenantId, VShardId}; +use nodedb_physical::physical_plan::{UpdateValue, VectorOp, VectorWriteTargets}; +use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; + +use super::super::convert::{ConvertContext, db_qualified}; +use super::super::filter::serialize_filters; +use super::super::value::{ + assignments_to_update_values, sql_value_to_nodedb_value, sql_value_to_string, +}; +use super::insert::{ + declared_primary_key_name, is_auto_rowid_pk, resolve_doc_identity_with_declared, +}; + +/// Collection-level settings every vector-primary write carries. +pub(in super::super) struct VectorPrimaryCfg<'a> { + pub field: &'a str, + pub quantization: nodedb_types::VectorQuantization, + pub storage_dtype: nodedb_types::VectorStorageDtype, + pub payload_indexes: &'a [(String, nodedb_types::PayloadIndexKind)], +} + +/// Inputs to [`convert_vector_primary_insert`]. +pub(in super::super) struct VectorPrimaryInsertArgs<'a> { + pub collection: &'a str, + pub cfg: &'a VectorPrimaryCfg<'a>, + pub rows: &'a [VectorPrimaryRow], + pub intent: VectorPrimaryInsertIntent, + pub on_conflict_updates: &'a [(String, SqlExpr)], + /// Resolved primary-key column; `id` by convention when nothing is declared. + pub primary_key: &'a str, + pub tenant_id: TenantId, + pub ctx: &'a ConvertContext, +} + +/// The routing every vector-primary task shares. +struct Routing { + qualified: nodedb_types::QualifiedCollection, + collection: String, + vshard: VShardId, +} + +fn routing(ctx: &ConvertContext, collection: &str) -> Routing { + let qualified = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); + let collection = db_qualified(ctx.database_id, collection); + let vshard = VShardId::from_collection_in_database(ctx.database_id, collection.as_str()); + Routing { + qualified, + collection, + vshard, + } +} + +fn task(tenant_id: TenantId, r: &Routing, ctx: &ConvertContext, op: VectorOp) -> PhysicalTask { + PhysicalTask { + tenant_id, + vshard_id: r.vshard, + database_id: ctx.database_id, + plan: PhysicalPlan::Vector(op), + post_set_op: PostSetOp::None, + txn_id: None, + } +} + +/// The declared `PRIMARY KEY` column, read once per statement. `_rowid` +/// carries no declaration. +fn declared_pk( + ctx: &ConvertContext, + collection: &str, + primary_key: &str, +) -> crate::Result> { + if is_auto_rowid_pk(primary_key) { + Ok(None) + } else { + declared_primary_key_name(ctx, collection) + } +} + +/// Encode a row's non-vector columns as the payload image the Data Plane +/// stores verbatim as the sidecar. +fn encode_payload(fields: &std::collections::HashMap) -> crate::Result> { + if fields.is_empty() { + return Ok(Vec::new()); + } + let value_map: std::collections::HashMap = fields + .iter() + .map(|(k, v)| (k.clone(), sql_value_to_nodedb_value(v))) + .collect(); + zerompk::to_msgpack_vec(&value_map).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("vector-primary payload: {e}"), + }) +} + +pub(in super::super) fn convert_vector_primary_insert( + args: VectorPrimaryInsertArgs<'_>, +) -> crate::Result> { + let VectorPrimaryInsertArgs { + collection, + cfg, + rows, + intent, + on_conflict_updates, + primary_key, + tenant_id, + ctx, + } = args; + let r = routing(ctx, collection); + let collection = r.collection.as_str(); + let declared = declared_pk(ctx, collection, primary_key)?; + let update_values = if on_conflict_updates.is_empty() { + Vec::new() + } else { + assignments_to_update_values(on_conflict_updates)? + }; + let mut tasks = Vec::with_capacity(rows.len()); + for row in rows { + // Enforce the per-tenant vector dimension quota before building any + // task. 0 means unlimited. + if ctx.max_vector_dim > 0 { + let dim = row.vector.len() as u32; + if dim > ctx.max_vector_dim { + return Err(crate::Error::TenantVectorDimExceeded { + dim, + limit: ctx.max_vector_dim, + }); + } + } + // Identity comes from the declared primary key, exactly as it does + // for a document row: the key content-addresses the surrogate, and + // a row with no key mints a fresh one and carries its identity under + // the key column so a later point read finds it. + let row_fields: Vec<(String, SqlValue)> = row + .payload_fields + .iter() + .map(|(k, v)| (k.clone(), v.clone())) + .collect(); + let (doc_id, surrogate) = resolve_doc_identity_with_declared( + ctx, + collection, + primary_key, + declared.as_deref(), + &row_fields, + )?; + let key_column = declared.as_deref().unwrap_or(primary_key); + let mut fields = row.payload_fields.clone(); + if !is_auto_rowid_pk(primary_key) + && !fields + .get(key_column) + .is_some_and(|v| !matches!(v, SqlValue::Null)) + { + fields.insert(key_column.to_string(), SqlValue::String(doc_id.clone())); + } + let pk_bytes = doc_id.into_bytes(); + let payload = encode_payload(&fields)?; + + // `returning` and `rls_filters` are filled in after conversion: the + // RETURNING spec by the protocol layer's injection pass, the read + // filter by the RLS injection pass. + let op = match intent { + VectorPrimaryInsertIntent::Insert => VectorOp::DirectInsert { + collection: r.qualified.clone(), + field: cfg.field.to_string(), + surrogate, + pk_bytes, + vector: row.vector.clone(), + payload, + quantization: cfg.quantization, + storage_dtype: cfg.storage_dtype, + payload_indexes: cfg.payload_indexes.to_vec(), + returning: None, + rls_filters: Vec::new(), + }, + VectorPrimaryInsertIntent::InsertIfAbsent => VectorOp::DirectInsertIfAbsent { + collection: r.qualified.clone(), + field: cfg.field.to_string(), + surrogate, + pk_bytes, + vector: row.vector.clone(), + payload, + quantization: cfg.quantization, + storage_dtype: cfg.storage_dtype, + payload_indexes: cfg.payload_indexes.to_vec(), + returning: None, + rls_filters: Vec::new(), + }, + VectorPrimaryInsertIntent::Upsert => VectorOp::DirectUpsert { + collection: r.qualified.clone(), + field: cfg.field.to_string(), + surrogate, + pk_bytes, + vector: row.vector.clone(), + payload, + quantization: cfg.quantization, + storage_dtype: cfg.storage_dtype, + payload_indexes: cfg.payload_indexes.to_vec(), + returning: None, + rls_filters: Vec::new(), + on_conflict_updates: update_values.clone(), + rls_write_check: RlsWriteCheck::pending_injection(), + }, + }; + tasks.push(task(tenant_id, &r, ctx, op)); + } + Ok(tasks) +} + +/// Resolve `target_keys` read-only to the surrogates they are bound to, or +/// serialize `filters` for the Data Plane to evaluate on the sidecar rows. +fn write_targets( + ctx: &ConvertContext, + collection: &str, + filters: &[Filter], + target_keys: &[SqlValue], +) -> crate::Result { + if target_keys.is_empty() { + return Ok(VectorWriteTargets::Predicate(serialize_filters(filters)?)); + } + let mut surrogates: Vec = Vec::with_capacity(target_keys.len()); + for key in target_keys { + let pk_bytes = sql_value_to_string(key).into_bytes(); + surrogates.push(ctx.surrogate_for_existing_pk(collection, &pk_bytes)?); + } + Ok(VectorWriteTargets::Surrogates(surrogates)) +} + +pub(in super::super) fn convert_vector_primary_delete( + collection: &str, + field: &str, + filters: &[Filter], + target_keys: &[SqlValue], + tenant_id: TenantId, + ctx: &ConvertContext, +) -> crate::Result> { + let r = routing(ctx, collection); + let targets = write_targets(ctx, r.collection.as_str(), filters, target_keys)?; + Ok(vec![task( + tenant_id, + &r, + ctx, + VectorOp::DirectDelete { + collection: r.qualified.clone(), + field: field.to_string(), + targets, + // Attached by `inject_returning_spec` after plan conversion; the + // RLS injection pass fills the two policy slots. + returning: None, + rls_filters: Vec::new(), + rls_write_check: RlsWriteCheck::pending_injection(), + }, + )]) +} + +/// Lower a vector-primary `TRUNCATE`. The Data Plane resolves every live +/// surrogate of the primary index itself, so no key is bound here. +pub(in super::super) fn convert_vector_primary_truncate( + collection: &str, + field: &str, + restart_identity: bool, + tenant_id: TenantId, + ctx: &ConvertContext, +) -> Vec { + let r = routing(ctx, collection); + vec![task( + tenant_id, + &r, + ctx, + VectorOp::DirectTruncate { + collection: r.qualified.clone(), + field: field.to_string(), + restart_identity, + }, + )] +} + +/// Inputs to [`convert_vector_primary_update`]. +pub(in super::super) struct VectorPrimaryUpdateArgs<'a> { + pub collection: &'a str, + pub cfg: &'a VectorPrimaryCfg<'a>, + pub new_vector: Option<&'a [f32]>, + pub assignments: &'a [(String, SqlExpr)], + pub filters: &'a [Filter], + pub target_keys: &'a [SqlValue], + pub tenant_id: TenantId, + pub ctx: &'a ConvertContext, +} + +pub(in super::super) fn convert_vector_primary_update( + args: VectorPrimaryUpdateArgs<'_>, +) -> crate::Result> { + let VectorPrimaryUpdateArgs { + collection, + cfg, + new_vector, + assignments, + filters, + target_keys, + tenant_id, + ctx, + } = args; + if let Some(vector) = new_vector + && ctx.max_vector_dim > 0 + && vector.len() as u32 > ctx.max_vector_dim + { + return Err(crate::Error::TenantVectorDimExceeded { + dim: vector.len() as u32, + limit: ctx.max_vector_dim, + }); + } + let r = routing(ctx, collection); + let targets = write_targets(ctx, r.collection.as_str(), filters, target_keys)?; + let payload_patch: Vec<(String, UpdateValue)> = assignments_to_update_values(assignments)?; + Ok(vec![task( + tenant_id, + &r, + ctx, + VectorOp::DirectUpdate { + collection: r.qualified.clone(), + field: cfg.field.to_string(), + targets, + new_vector: new_vector.map(<[f32]>::to_vec), + payload_patch, + quantization: cfg.quantization, + storage_dtype: cfg.storage_dtype, + payload_indexes: cfg.payload_indexes.to_vec(), + // See `convert_vector_primary_delete`. + returning: None, + rls_filters: Vec::new(), + rls_write_check: RlsWriteCheck::pending_injection(), + }, + )]) +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_sql::types::VectorPrimaryRow; + use nodedb_types::VectorQuantization; + + fn make_ctx(max_vector_dim: u32) -> ConvertContext { + ConvertContext { + purpose: super::super::super::convert::PlanningPurpose::Execute, + retention_registry: None, + array_catalog: None, + credentials: None, + wal: None, + surrogate_assigner: None, + cluster_enabled: false, + bitemporal_retention_registry: None, + max_vector_dim, + force_shuffle_join: false, + shuffle_num_parts: 0, + force_shuffle_agg: false, + shuffle_agg_num_parts: 0, + broadcast_threshold_bytes: 8 * 1024 * 1024, + shuffle_agg_threshold: 10_000, + database_id: crate::types::DatabaseId::DEFAULT, + tenant_id: crate::types::TenantId::new(0), + } + } + + fn row(dim: usize, id: &str) -> VectorPrimaryRow { + let mut payload_fields = std::collections::HashMap::new(); + payload_fields.insert("id".to_string(), SqlValue::String(id.to_string())); + VectorPrimaryRow { + surrogate: nodedb_types::Surrogate::ZERO, + vector: vec![0.0f32; dim], + payload_fields, + } + } + + fn cfg() -> VectorPrimaryCfg<'static> { + VectorPrimaryCfg { + field: "emb", + quantization: VectorQuantization::None, + storage_dtype: nodedb_types::VectorStorageDtype::F32, + payload_indexes: &[], + } + } + + fn convert( + ctx: &ConvertContext, + rows: &[VectorPrimaryRow], + intent: VectorPrimaryInsertIntent, + ) -> crate::Result> { + convert_vector_primary_insert(VectorPrimaryInsertArgs { + collection: "vecs", + cfg: &cfg(), + rows, + intent, + on_conflict_updates: &[], + primary_key: "id", + tenant_id: crate::types::TenantId::new(1), + ctx, + }) + } + + #[test] + fn tenant_vector_dim_under_bound_succeeds() { + let ctx = make_ctx(128); + let rows = vec![row(64, "a"), row(128, "b")]; + let result = convert(&ctx, &rows, VectorPrimaryInsertIntent::Insert); + assert!(result.is_ok(), "dimensions under/at cap must succeed"); + } + + #[test] + fn tenant_vector_dim_exceeded_rejected() { + let ctx = make_ctx(64); + let rows = vec![row(65, "a")]; + let result = convert(&ctx, &rows, VectorPrimaryInsertIntent::Insert); + match result { + Err(crate::Error::TenantVectorDimExceeded { dim, limit }) => { + assert_eq!(dim, 65); + assert_eq!(limit, 64); + } + other => panic!("expected TenantVectorDimExceeded, got {other:?}"), + } + } + + #[test] + fn tenant_vector_dim_zero_means_unlimited() { + let ctx = make_ctx(0); + let rows = vec![row(99999, "a")]; + let result = convert(&ctx, &rows, VectorPrimaryInsertIntent::Insert); + assert!(result.is_ok(), "limit=0 means unlimited, must succeed"); + } + + /// The row's primary key, not its vector, is the identity carried to the + /// Data Plane: two rows with the same vector and different keys are two + /// rows, and the key bytes travel for followers to bind. + #[test] + fn identity_is_the_primary_key_not_the_vector() { + let ctx = make_ctx(0); + let rows = vec![row(3, "r1"), row(3, "r2")]; + let tasks = convert(&ctx, &rows, VectorPrimaryInsertIntent::Insert).expect("convert"); + let keys: Vec> = tasks + .iter() + .map(|t| match &t.plan { + PhysicalPlan::Vector(VectorOp::DirectInsert { pk_bytes, .. }) => pk_bytes.clone(), + other => panic!("expected DirectInsert, got {other:?}"), + }) + .collect(); + assert_eq!(keys, vec![b"r1".to_vec(), b"r2".to_vec()]); + } + + #[test] + fn intent_selects_the_physical_op() { + let ctx = make_ctx(0); + let rows = vec![row(3, "r1")]; + let absent = + convert(&ctx, &rows, VectorPrimaryInsertIntent::InsertIfAbsent).expect("convert"); + assert!(matches!( + absent[0].plan, + PhysicalPlan::Vector(VectorOp::DirectInsertIfAbsent { .. }) + )); + let upsert = convert(&ctx, &rows, VectorPrimaryInsertIntent::Upsert).expect("convert"); + assert!(matches!( + upsert[0].plan, + PhysicalPlan::Vector(VectorOp::DirectUpsert { .. }) + )); + } + + /// With no primary-key equality the WHERE clause travels as a predicate + /// for the Data Plane to resolve against the sidecar rows. + #[test] + fn delete_without_point_keys_carries_the_predicate() { + let ctx = make_ctx(0); + let tasks = convert_vector_primary_delete( + "vecs", + "emb", + &[], + &[], + crate::types::TenantId::new(1), + &ctx, + ) + .expect("convert"); + assert!(matches!( + &tasks[0].plan, + PhysicalPlan::Vector(VectorOp::DirectDelete { + targets: VectorWriteTargets::Predicate(bytes), + .. + }) if bytes.is_empty() + )); + } +} diff --git a/nodedb/src/control/planner/sql_plan_convert/output_schema/build.rs b/nodedb/src/control/planner/sql_plan_convert/output_schema/build.rs index 56256573c..6a729a506 100644 --- a/nodedb/src/control/planner/sql_plan_convert/output_schema/build.rs +++ b/nodedb/src/control/planner/sql_plan_convert/output_schema/build.rs @@ -361,7 +361,9 @@ pub fn build_output_schema( | SqlPlan::UpdateFrom { collection, .. } | SqlPlan::Delete { collection, .. } | SqlPlan::TimeseriesIngest { collection, .. } - | SqlPlan::VectorPrimaryInsert { collection, .. } => { + | SqlPlan::VectorPrimaryInsert { collection, .. } + | SqlPlan::VectorPrimaryDelete { collection, .. } + | SqlPlan::VectorPrimaryUpdate { collection, .. } => { build_returning_schema(returning, collection, catalog, database_id) } // Same rule, for the two writes that name their target `target`. @@ -374,6 +376,7 @@ pub fn build_output_schema( // so announcing columns for one would hold a count payload to a row // shape it does not have. SqlPlan::Truncate { .. } + | SqlPlan::VectorPrimaryTruncate { .. } | SqlPlan::CreateArray { .. } | SqlPlan::DropArray { .. } | SqlPlan::AlterArray { .. } diff --git a/nodedb/src/control/planner/sql_plan_convert/output_schema/join_types.rs b/nodedb/src/control/planner/sql_plan_convert/output_schema/join_types.rs index 3fed56b73..f55c0eaca 100644 --- a/nodedb/src/control/planner/sql_plan_convert/output_schema/join_types.rs +++ b/nodedb/src/control/planner/sql_plan_convert/output_schema/join_types.rs @@ -131,6 +131,7 @@ fn collect_sides(plan: &SqlPlan, out: &mut Vec) { | SqlPlan::UpdateFrom { .. } | SqlPlan::Delete { .. } | SqlPlan::Truncate { .. } + | SqlPlan::VectorPrimaryTruncate { .. } | SqlPlan::Aggregate { .. } | SqlPlan::TimeseriesIngest { .. } | SqlPlan::Union { .. } @@ -154,6 +155,8 @@ fn collect_sides(plan: &SqlPlan, out: &mut Vec) { | SqlPlan::LateralTopK { .. } | SqlPlan::LateralLoop { .. } | SqlPlan::VectorPrimaryInsert { .. } + | SqlPlan::VectorPrimaryDelete { .. } + | SqlPlan::VectorPrimaryUpdate { .. } | SqlPlan::CreateIndex { .. } | SqlPlan::DropIndex { .. } => {} } diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/core.rs b/nodedb/src/control/planner/sql_plan_convert/scan/core.rs index 5699c1359..e2256a3ad 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/core.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/core.rs @@ -257,7 +257,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_point_get( let physical = match engine { EngineType::KeyValue => PhysicalPlan::Kv(KvOp::Get { collection: qualified_collection.clone(), - key: sql_value_to_bytes(key_value), + key: sql_value_to_bytes(key_value)?, rls_filters: Vec::new(), surrogate_ceiling: None, }), diff --git a/nodedb/src/control/planner/sql_plan_convert/set_ops.rs b/nodedb/src/control/planner/sql_plan_convert/set_ops.rs index 1fb9530f3..0141aac85 100644 --- a/nodedb/src/control/planner/sql_plan_convert/set_ops.rs +++ b/nodedb/src/control/planner/sql_plan_convert/set_ops.rs @@ -2,7 +2,7 @@ //! Set operations and miscellaneous plan conversions (UNION, INTERSECT, EXCEPT, CTE, etc.). -use nodedb_sql::types::{Projection, SortKey, SqlExpr, SqlPlan, SqlValue, WindowSpec}; +use nodedb_sql::types::{EngineType, Projection, SortKey, SqlExpr, SqlPlan, SqlValue, WindowSpec}; use crate::bridge::envelope::PhysicalPlan; use crate::types::{TenantId, VShardId}; @@ -74,8 +74,15 @@ pub(super) fn convert_constant_result( }]) } +/// Lower `SqlPlan::Truncate` to the engine that stores the rows. The match +/// is exhaustive over `EngineType`: a document-family collection clears its +/// document store, a KV collection clears its hash index, a columnar or +/// spatial collection clears its mutation engine (spatial shares the +/// columnar DML ops, so it shares the truncate op too), and a timeseries +/// collection clears its memtable and partitions. pub(super) fn convert_truncate( collection: &str, + engine: EngineType, restart_identity: bool, tenant_id: TenantId, ctx: &ConvertContext, @@ -84,19 +91,46 @@ pub(super) fn convert_truncate( let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); let collection = coll_qualified.as_str(); let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let plan = match engine { + EngineType::DocumentSchemaless | EngineType::DocumentStrict => { + PhysicalPlan::Document(DocumentOp::Truncate { + collection: qualified_collection, + restart_identity, + // Filled in by the materialized-sum resolution pass, which recon- + // scans the rows this TRUNCATE will remove. + resolved_sum_targets: Vec::new(), + // Names the column each removed row's identity is read from. + declared_primary_key: super::dml::declared_primary_key_name(ctx, collection)?, + }) + } + EngineType::KeyValue => PhysicalPlan::Kv(KvOp::Truncate { + collection: qualified_collection, + restart_identity, + }), + EngineType::Columnar | EngineType::Spatial => { + PhysicalPlan::Columnar(ColumnarOp::Truncate { + collection: qualified_collection, + restart_identity, + }) + } + EngineType::Timeseries => PhysicalPlan::Timeseries(TimeseriesOp::Truncate { + collection: qualified_collection, + restart_identity, + }), + // `ArrayRules::plan_truncate` refuses before a plan is built. + EngineType::Array => { + return Err(crate::Error::Internal { + detail: format!( + "TRUNCATE reached plan conversion for array collection '{collection}'" + ), + }); + } + }; Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, database_id: ctx.database_id, - plan: PhysicalPlan::Document(DocumentOp::Truncate { - collection: qualified_collection, - restart_identity, - // Filled in by the materialized-sum resolution pass, which recon- - // scans the rows this TRUNCATE will remove. - resolved_sum_targets: Vec::new(), - // Names the column each removed row's identity is read from. - declared_primary_key: super::dml::declared_primary_key_name(ctx, collection)?, - }), + plan, post_set_op: PostSetOp::None, txn_id: None, }]) @@ -436,7 +470,122 @@ fn lower_subquery_sort_keys(keys: &[SortKey], merged_doc_body: bool) -> Vec ConvertContext { + ConvertContext { + purpose: crate::control::planner::sql_plan_convert::PlanningPurpose::Execute, + retention_registry: None, + array_catalog: None, + credentials: None, + wal: None, + surrogate_assigner: None, + cluster_enabled: false, + bitemporal_retention_registry: None, + max_vector_dim: 0, + force_shuffle_join: false, + shuffle_num_parts: 0, + force_shuffle_agg: false, + shuffle_agg_num_parts: 0, + broadcast_threshold_bytes: 8 * 1024 * 1024, + shuffle_agg_threshold: 10_000, + database_id: crate::types::DatabaseId::DEFAULT, + tenant_id: crate::types::TenantId::new(0), + } + } + + #[test] + fn convert_truncate_routes_kv_to_kv_op_with_restart_flag() { + let tasks = convert_truncate( + "kvc", + EngineType::KeyValue, + true, + TenantId::new(1), + &bare_ctx(), + ) + .expect("kv truncate converts"); + assert_eq!(tasks.len(), 1); + match &tasks[0].plan { + PhysicalPlan::Kv(KvOp::Truncate { + collection, + restart_identity, + }) => { + assert_eq!(collection.as_str(), "kvc"); + assert!(*restart_identity); + } + other => panic!("expected KvOp::Truncate, got {other:?}"), + } + } + + #[test] + fn convert_truncate_routes_document_engines_to_document_op() { + for engine in [EngineType::DocumentSchemaless, EngineType::DocumentStrict] { + let tasks = convert_truncate("docs", engine, false, TenantId::new(1), &bare_ctx()) + .expect("document truncate converts"); + assert!( + matches!( + &tasks[0].plan, + PhysicalPlan::Document(DocumentOp::Truncate { .. }) + ), + "{engine:?} must lower to DocumentOp::Truncate, got {:?}", + tasks[0].plan + ); + } + } + + #[test] + fn convert_truncate_routes_columnar_and_spatial_to_columnar_op() { + for engine in [EngineType::Columnar, EngineType::Spatial] { + let tasks = convert_truncate("c", engine, true, TenantId::new(1), &bare_ctx()) + .expect("columnar-family truncate converts"); + assert_eq!(tasks.len(), 1); + match &tasks[0].plan { + PhysicalPlan::Columnar(ColumnarOp::Truncate { + collection, + restart_identity, + }) => { + assert_eq!(collection.as_str(), "c"); + assert!(*restart_identity, "{engine:?} must carry restart_identity"); + } + other => panic!("{engine:?} must lower to ColumnarOp::Truncate, got {other:?}"), + } + } + } + + #[test] + fn convert_truncate_routes_timeseries_to_timeseries_op() { + let tasks = convert_truncate( + "ts", + EngineType::Timeseries, + false, + TenantId::new(1), + &bare_ctx(), + ) + .expect("timeseries truncate converts"); + assert_eq!(tasks.len(), 1); + match &tasks[0].plan { + PhysicalPlan::Timeseries(TimeseriesOp::Truncate { + collection, + restart_identity, + }) => { + assert_eq!(collection.as_str(), "ts"); + assert!(!*restart_identity); + } + other => panic!("expected TimeseriesOp::Truncate, got {other:?}"), + } + } + + #[test] + fn convert_truncate_on_array_is_an_internal_error() { + let err = convert_truncate( + "arr", + EngineType::Array, + false, + TenantId::new(1), + &bare_ctx(), + ) + .expect_err("array never reaches conversion"); + assert!(matches!(err, crate::Error::Internal { .. }), "got {err:?}"); + } #[test] fn convert_insert_select_builds_document_op() { diff --git a/nodedb/src/control/planner/sql_plan_convert/value/convert.rs b/nodedb/src/control/planner/sql_plan_convert/value/convert.rs index 178a8a2eb..186d79286 100644 --- a/nodedb/src/control/planner/sql_plan_convert/value/convert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/value/convert.rs @@ -68,13 +68,25 @@ fn hex_encode(bytes: &[u8]) -> String { s } -pub(crate) fn sql_value_to_bytes(v: &SqlValue) -> Vec { +/// Raw bytes for a KV key or a single-`value` column body. +/// +/// A scalar encodes through `nodedb_types::scalar_to_raw_bytes`, the same +/// rule a KV read-modify-write uses to re-encode a raw body. An array is +/// PostgreSQL array text. +pub(crate) fn sql_value_to_bytes(v: &SqlValue) -> crate::Result> { match v { - SqlValue::String(s) => s.as_bytes().to_vec(), - SqlValue::Bytes(b) => b.clone(), - SqlValue::Int(i) => i.to_string().as_bytes().to_vec(), - SqlValue::Decimal(d) => d.to_string().as_bytes().to_vec(), - _ => sql_value_to_string(v).into_bytes(), + SqlValue::Array(_) => Ok(sql_value_to_string(v).into_bytes()), + SqlValue::Int(_) + | SqlValue::Float(_) + | SqlValue::Decimal(_) + | SqlValue::String(_) + | SqlValue::Bool(_) + | SqlValue::Null + | SqlValue::Bytes(_) + | SqlValue::Timestamp(_) + | SqlValue::Timestamptz(_) => Ok(nodedb_types::scalar_to_raw_bytes( + &sql_value_to_nodedb_value(v), + )?), } } @@ -109,6 +121,20 @@ mod tests { ); } + #[test] + fn raw_bytes_follow_the_shared_scalar_rule() { + let cases = [ + (SqlValue::String("v1".into()), b"v1".to_vec()), + (SqlValue::Int(7), b"7".to_vec()), + (SqlValue::Bool(false), b"false".to_vec()), + (SqlValue::Bytes(vec![0xff]), vec![0xff]), + (SqlValue::Null, Vec::new()), + ]; + for (sql, expected) in cases { + assert_eq!(sql_value_to_bytes(&sql).unwrap(), expected, "{sql:?}"); + } + } + #[test] fn read_and_write_scalar_stringifiers_agree() { let cases = [ diff --git a/nodedb/src/control/planner/sql_plan_convert/visitor/arms_dml.rs b/nodedb/src/control/planner/sql_plan_convert/visitor/arms_dml.rs index 607fb34f6..38ca29f35 100644 --- a/nodedb/src/control/planner/sql_plan_convert/visitor/arms_dml.rs +++ b/nodedb/src/control/planner/sql_plan_convert/visitor/arms_dml.rs @@ -154,25 +154,116 @@ macro_rules! impl_dml_arms_for_convert_visitor { } fn vector_primary_insert( + &mut self, + args: nodedb_sql::VectorPrimaryInsertVisitArgs<'_>, + ) -> crate::Result> { + let nodedb_sql::VectorPrimaryInsertVisitArgs { + collection, + field, + quantization, + storage_dtype, + payload_indexes, + rows, + intent, + on_conflict_updates, + primary_key, + } = args; + let primary_key = primary_key.ok_or_else(|| crate::Error::PlanError { + detail: format!( + "vector-primary insert converter reached collection '{collection}' with no resolved primary key" + ), + })?; + super::super::dml::convert_vector_primary_insert( + super::super::dml::VectorPrimaryInsertArgs { + collection, + cfg: &super::super::dml::VectorPrimaryCfg { + field, + quantization, + storage_dtype, + payload_indexes, + }, + rows, + intent, + on_conflict_updates, + primary_key, + tenant_id: self.tenant_id, + ctx: self.ctx, + }, + ) + } + + fn vector_primary_delete( + &mut self, + args: nodedb_sql::VectorPrimaryDeleteVisitArgs<'_>, + ) -> crate::Result> { + let nodedb_sql::VectorPrimaryDeleteVisitArgs { + collection, + field, + filters, + target_keys, + // Point keys are already extracted against it by the planner. + primary_key: _, + } = args; + super::super::dml::convert_vector_primary_delete( + collection, + field, + filters, + target_keys, + self.tenant_id, + self.ctx, + ) + } + + fn vector_primary_truncate( &mut self, collection: &str, field: &str, - quantization: &nodedb_types::VectorQuantization, - storage_dtype: &nodedb_types::VectorStorageDtype, - payload_indexes: &[(String, nodedb_types::PayloadIndexKind)], - rows: &[nodedb_sql::types::plan::VectorPrimaryRow], + restart_identity: bool, ) -> crate::Result> { - super::super::dml::convert_vector_primary_insert( + Ok(super::super::dml::convert_vector_primary_truncate( collection, - &super::super::dml::VectorPrimaryInsertCfg { - field, - quantization: *quantization, - storage_dtype: *storage_dtype, - payload_indexes, - }, - rows, + field, + restart_identity, self.tenant_id, self.ctx, + )) + } + + fn vector_primary_update( + &mut self, + args: nodedb_sql::VectorPrimaryUpdateVisitArgs<'_>, + ) -> crate::Result> { + let nodedb_sql::VectorPrimaryUpdateVisitArgs { + collection, + field, + quantization, + storage_dtype, + payload_indexes, + new_vector, + assignments, + filters, + target_keys, + // Attached by `inject_returning_spec` after plan conversion. + returning: _, + // Point keys are already extracted against it by the planner. + primary_key: _, + } = args; + super::super::dml::convert_vector_primary_update( + super::super::dml::VectorPrimaryUpdateArgs { + collection, + cfg: &super::super::dml::VectorPrimaryCfg { + field, + quantization, + storage_dtype, + payload_indexes, + }, + new_vector, + assignments, + filters, + target_keys, + tenant_id: self.tenant_id, + ctx: self.ctx, + }, ) } diff --git a/nodedb/src/control/planner/sql_plan_convert/visitor/arms_set_ops.rs b/nodedb/src/control/planner/sql_plan_convert/visitor/arms_set_ops.rs index ac980de8d..deec36ea0 100644 --- a/nodedb/src/control/planner/sql_plan_convert/visitor/arms_set_ops.rs +++ b/nodedb/src/control/planner/sql_plan_convert/visitor/arms_set_ops.rs @@ -20,10 +20,12 @@ macro_rules! impl_set_ops_arms_for_convert_visitor { fn truncate( &mut self, collection: &str, + engine: nodedb_sql::types::EngineType, restart_identity: bool, ) -> crate::Result> { super::super::set_ops::convert_truncate( collection, + engine, restart_identity, self.tenant_id, self.ctx, diff --git a/nodedb/src/control/security/identity/plan_permission.rs b/nodedb/src/control/security/identity/plan_permission.rs index 7461d7132..03f5c4c6d 100644 --- a/nodedb/src/control/security/identity/plan_permission.rs +++ b/nodedb/src/control/security/identity/plan_permission.rs @@ -36,7 +36,9 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm | VectorOp::MultiSearch { .. } | VectorOp::QueryStats { .. } | VectorOp::SparseSearch { .. } - | VectorOp::MultiVectorScoreSearch { .. }, + | VectorOp::MultiVectorScoreSearch { .. } + // Read-only: reports what the wrapped write would apply; that write is authorized separately. + | VectorOp::ResolveDirectWrite(_), ) => Permission::Read, PhysicalPlan::Crdt( @@ -152,7 +154,14 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm | VectorOp::SparseDelete { .. } | VectorOp::MultiVectorInsert { .. } | VectorOp::MultiVectorDelete { .. } - | VectorOp::DirectUpsert { .. }, + | VectorOp::DirectUpsert { .. } + | VectorOp::DirectInsert { .. } + | VectorOp::DirectInsertIfAbsent { .. } + | VectorOp::DirectDelete { .. } + | VectorOp::DirectTruncate { .. } + | VectorOp::DirectUpdate { .. } + // Never client-issued: write-resolve orchestrator builds it post-authorization. + | VectorOp::ResolvedDirectWrite { .. }, ) => Permission::Write, PhysicalPlan::Document( @@ -193,10 +202,13 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm | ColumnarOp::ResolvedDelete { .. } // Requires the same Write the predicate it resolves requires, though it isn't // itself write-class for admission/replication (see `write_class::columnar_is_write`). - | ColumnarOp::ResolveDml { .. }, + | ColumnarOp::ResolveDml { .. } + | ColumnarOp::Truncate { .. }, ) => Permission::Write, - PhysicalPlan::Timeseries(TimeseriesOp::Ingest { .. }) => Permission::Write, + PhysicalPlan::Timeseries(TimeseriesOp::Ingest { .. } | TimeseriesOp::Truncate { .. }) => { + Permission::Write + } // Transaction batch: requires write (contains writes). PhysicalPlan::Meta(MetaOp::TransactionBatch { .. }) => Permission::Write, diff --git a/nodedb/src/control/sequence/registry.rs b/nodedb/src/control/sequence/registry.rs index 3b4b7d4a2..1fdb83ebf 100644 --- a/nodedb/src/control/sequence/registry.rs +++ b/nodedb/src/control/sequence/registry.rs @@ -299,7 +299,7 @@ impl SequenceRegistry { name: name.to_string(), })?; - handle.setval(restart_value)?; + handle.restart_at(restart_value)?; Ok(()) } @@ -394,7 +394,7 @@ impl SequenceRegistry { for (key, handle) in map.iter() { if key.starts_with(&prefix) && handle.def.name.ends_with(suffix) { let start = handle.def.start_value; - if let Err(e) = handle.setval(start) { + if let Err(e) = handle.restart_at(start) { tracing::warn!( sequence = %handle.def.name, error = %e, diff --git a/nodedb/src/control/sequence/types.rs b/nodedb/src/control/sequence/types.rs index 563fc5d90..e4ddf85e4 100644 --- a/nodedb/src/control/sequence/types.rs +++ b/nodedb/src/control/sequence/types.rs @@ -192,6 +192,24 @@ impl SequenceHandle { Ok(value) } + /// Restart the sequence so the next `nextval` returns `value` itself + /// (`ALTER SEQUENCE ... RESTART WITH`, `TRUNCATE ... RESTART IDENTITY`). + /// Unlike `setval`, the restart value counts as not yet called. + pub fn restart_at(&self, value: i64) -> Result<(), SequenceError> { + if value < self.def.min_value || value > self.def.max_value { + return Err(SequenceError::OutOfRange { + name: self.def.name.clone(), + value, + min: self.def.min_value, + max: self.def.max_value, + }); + } + self.counter + .store(value - self.def.increment, Ordering::Relaxed); + self.called.store(false, Ordering::Relaxed); + Ok(()) + } + /// Check if the period has changed and reset the counter if needed. /// /// Must be called before `nextval()` when the sequence has a reset scope. @@ -382,6 +400,19 @@ mod tests { assert_eq!(h.nextval().unwrap(), 51); } + #[test] + fn restart_at_makes_the_restart_value_the_next_value() { + let h = make_handle(1, 1, 1, 100, false); + assert_eq!(h.nextval().unwrap(), 1); + assert_eq!(h.nextval().unwrap(), 2); + h.restart_at(1).unwrap(); + assert!(!h.is_called()); + assert_eq!(h.nextval().unwrap(), 1); + h.restart_at(40).unwrap(); + assert_eq!(h.nextval().unwrap(), 40); + assert!(h.restart_at(101).is_err()); + } + #[test] fn setval_out_of_range() { let h = make_handle(1, 1, 1, 100, false); diff --git a/nodedb/src/control/server/dispatch_utils/change_events/extract.rs b/nodedb/src/control/server/dispatch_utils/change_events/extract.rs index b99493b8a..96f120f1b 100644 --- a/nodedb/src/control/server/dispatch_utils/change_events/extract.rs +++ b/nodedb/src/control/server/dispatch_utils/change_events/extract.rs @@ -8,7 +8,7 @@ use crate::control::change_stream::ChangeOperation; use crate::types::TenantId; use nodedb_physical::physical_plan::{ ArrayOp, ClusterArrayOp, ColumnarOp, CrdtOp, DocumentOp, DocumentResolvedMutation, KvOp, - KvResolvedMutation, MetaOp, TimeseriesOp, VectorOp, + KvResolvedMutation, MetaOp, TimeseriesOp, VectorOp, VectorResolvedMutation, VectorWriteTargets, }; use nodedb_types::{RowIdentity, StorageKey}; @@ -21,6 +21,30 @@ fn every_row() -> RowIdentity { RowIdentity::from_user_key("*") } +/// One event per point-targeted vector-primary surrogate; a predicate write +/// names every row, like `BulkUpdate` / `BulkDelete`. +fn vector_target_events( + collection: &nodedb_types::QualifiedCollection, + targets: &VectorWriteTargets, + operation: ChangeOperation, +) -> Vec { + match targets { + VectorWriteTargets::Surrogates(surrogates) => surrogates + .iter() + .map(|s| { + ( + collection.to_string(), + StorageKey::for_surrogate(*s).to_identity(), + operation, + ) + }) + .collect(), + VectorWriteTargets::Predicate(_) => { + vec![(collection.to_string(), every_row(), operation)] + } + } +} + /// A KV row's identity is its key bytes, rendered as text for the subscriber. fn kv_identity(key: &[u8]) -> RowIdentity { RowIdentity::from_user_key(String::from_utf8_lossy(key)) @@ -141,13 +165,18 @@ pub(super) fn extract_write_metadata( // Remaining DocumentOp variants are reads or catalog/schema DDL — no row changed. PhysicalPlan::Document(_) => Vec::new(), - // Batch write; document_id="*" indicates a batch. High-cardinality metrics - // would flood the bus otherwise — subscribe via collection_filter. + // Batch write and truncate: document_id="*" names every row. Per-row + // events would flood the bus — subscribe via collection_filter. PhysicalPlan::Timeseries(TimeseriesOp::Ingest { collection, .. }) => { vec![(collection.to_string(), every_row(), ChangeOperation::Insert)] } - // TimeseriesOp::Scan is a read — no row changed. - PhysicalPlan::Timeseries(_) => Vec::new(), + PhysicalPlan::Timeseries(TimeseriesOp::Truncate { collection, .. }) => { + vec![(collection.to_string(), every_row(), ChangeOperation::Delete)] + } + // Scan and the resolve pass are reads — no row changed. + PhysicalPlan::Timeseries(TimeseriesOp::Scan { .. } | TimeseriesOp::ResolveIngest(_)) => { + Vec::new() + } // KV engine write operations. PhysicalPlan::Kv(KvOp::Put { @@ -202,7 +231,7 @@ pub(super) fn extract_write_metadata( PhysicalPlan::Kv(KvOp::BatchPut { collection, .. }) => { vec![(collection.to_string(), every_row(), ChangeOperation::Insert)] } - PhysicalPlan::Kv(KvOp::Truncate { collection }) => { + PhysicalPlan::Kv(KvOp::Truncate { collection, .. }) => { vec![(collection.to_string(), every_row(), ChangeOperation::Delete)] } // Debits + credits two keys in the same collection; not individually addressable, @@ -261,17 +290,14 @@ pub(super) fn extract_write_metadata( PhysicalPlan::Columnar(ColumnarOp::Insert { collection, .. }) => { vec![(collection.to_string(), every_row(), ChangeOperation::Insert)] } - PhysicalPlan::Columnar(ColumnarOp::Update { collection, .. }) => { + // The resolved-row-set forms are the same statements, same CDC event. + PhysicalPlan::Columnar(ColumnarOp::Update { collection, .. }) + | PhysicalPlan::Columnar(ColumnarOp::ResolvedUpdate { collection, .. }) => { vec![(collection.to_string(), every_row(), ChangeOperation::Update)] } - PhysicalPlan::Columnar(ColumnarOp::Delete { collection, .. }) => { - vec![(collection.to_string(), every_row(), ChangeOperation::Delete)] - } - // Resolved-row-set form of the same UPDATE/DELETE — same CDC event as above. - PhysicalPlan::Columnar(ColumnarOp::ResolvedUpdate { collection, .. }) => { - vec![(collection.to_string(), every_row(), ChangeOperation::Update)] - } - PhysicalPlan::Columnar(ColumnarOp::ResolvedDelete { collection, .. }) => { + PhysicalPlan::Columnar(ColumnarOp::Delete { collection, .. }) + | PhysicalPlan::Columnar(ColumnarOp::ResolvedDelete { collection, .. }) + | PhysicalPlan::Columnar(ColumnarOp::Truncate { collection, .. }) => { vec![(collection.to_string(), every_row(), ChangeOperation::Delete)] } // Scan / MaterializeScan are reads — no row changed. @@ -293,17 +319,69 @@ pub(super) fn extract_write_metadata( PhysicalPlan::Graph(_) => Vec::new(), // Vector is normally a Document secondary index — publishing here would duplicate. - // `DirectUpsert` is the exception: the sole write for a vector-primary collection. - // The row carries no user key, so its identity is the decimal surrogate. - PhysicalPlan::Vector(VectorOp::DirectUpsert { - collection, - surrogate, - .. - }) => vec![( + // The direct write family is the exception: the sole writes for a vector-primary + // collection. The row's identity is its PK-bound surrogate, rendered as the decimal + // surrogate, which is what a sidecar read reports as the row id. + PhysicalPlan::Vector( + VectorOp::DirectUpsert { + collection, + surrogate, + .. + } + | VectorOp::DirectInsert { + collection, + surrogate, + .. + } + | VectorOp::DirectInsertIfAbsent { + collection, + surrogate, + .. + }, + ) => vec![( collection.to_string(), StorageKey::for_surrogate(*surrogate).to_identity(), ChangeOperation::Insert, )], + PhysicalPlan::Vector(VectorOp::DirectDelete { + collection, + targets, + .. + }) => vector_target_events(collection, targets, ChangeOperation::Delete), + PhysicalPlan::Vector(VectorOp::DirectTruncate { collection, .. }) => { + vec![(collection.to_string(), every_row(), ChangeOperation::Delete)] + } + PhysicalPlan::Vector(VectorOp::DirectUpdate { + collection, + targets, + .. + }) => vector_target_events(collection, targets, ChangeOperation::Update), + // Reports one event per mutation, naming every row touched — never collapses to "*". + // An upsert's pre-image is what the resolve found stored: absent = insert, present = update. + PhysicalPlan::Vector(VectorOp::ResolvedDirectWrite { + collection, + mutations, + .. + }) => mutations + .iter() + .map(|mutation| { + let operation = match mutation { + VectorResolvedMutation::Delete { .. } => ChangeOperation::Delete, + VectorResolvedMutation::Update { .. } => ChangeOperation::Update, + VectorResolvedMutation::Upsert { old_payload, .. } => match old_payload { + Some(_) => ChangeOperation::Update, + None => ChangeOperation::Insert, + }, + }; + ( + collection.to_string(), + StorageKey::for_surrogate(mutation.surrogate()).to_identity(), + operation, + ) + }) + .collect(), + // The resolve pass reads; the rows it decides are published by the + // resolved write that applies them. PhysicalPlan::Vector(_) => Vec::new(), // Spatial R-tree writes are index maintenance for a row already published @@ -624,6 +702,7 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "embeddings"), field: "emb".into(), surrogate: Surrogate::new(42), + pk_bytes: Vec::new(), vector: vec![0.0, 1.0], payload: Vec::new(), quantization: VectorQuantization::default(), @@ -631,6 +710,8 @@ mod tests { payload_indexes: Vec::new(), returning: None, rls_filters: Vec::new(), + on_conflict_updates: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::decided_earlier_in_request(), }); let meta = extract_write_metadata(&plan, TenantId::new(1)); // The row id is the client identity of the row, the decimal diff --git a/nodedb/src/control/server/http/routes/query/materialized/shape.rs b/nodedb/src/control/server/http/routes/query/materialized/shape.rs index bbd8b28a3..ad9148e54 100644 --- a/nodedb/src/control/server/http/routes/query/materialized/shape.rs +++ b/nodedb/src/control/server/http/routes/query/materialized/shape.rs @@ -121,6 +121,7 @@ pub(super) async fn run_task_loop( clauses: _, returning: _, resolved_inserts: None, + resolved_insert_identities: _, source_rows: _, rls_filters: _, rls_write_check: _, diff --git a/nodedb/src/control/server/http/routes/ws_rpc/execute_sql.rs b/nodedb/src/control/server/http/routes/ws_rpc/execute_sql.rs index b493b74d3..12806adc2 100644 --- a/nodedb/src/control/server/http/routes/ws_rpc/execute_sql.rs +++ b/nodedb/src/control/server/http/routes/ws_rpc/execute_sql.rs @@ -132,6 +132,7 @@ pub async fn execute_sql( clauses: _, returning: _, resolved_inserts: None, + resolved_insert_identities: _, source_rows: _, rls_filters: _, rls_write_check: _, @@ -281,6 +282,8 @@ pub async fn execute_sql( } }; + // WebSocket RPC has no session transaction (no BEGIN / COMMIT), so + // every task is autocommit and forwards with no transaction id. let payloads: crate::Result>> = match shared.gateway.get() { Some(gw) => { let gw_ctx = QueryContext { diff --git a/nodedb/src/control/server/http/server.rs b/nodedb/src/control/server/http/server.rs index 7befca9b7..fcabd76cc 100644 --- a/nodedb/src/control/server/http/server.rs +++ b/nodedb/src/control/server/http/server.rs @@ -251,9 +251,10 @@ pub async fn run( ); let mut shutdown_rx = bus.handle().flat_watch().raw_receiver(); - let query_ctx = Arc::new(crate::control::planner::context::QueryContext::for_state( - &shared, - )); + // A top-level listener plans with descriptor leases, the array catalog + // and the WAL handle, same as a pgwire connection. + let query_ctx = + Arc::new(crate::control::planner::context::QueryContext::for_state_with_lease(&shared)); let state = AppState { shared, auth_mode, diff --git a/nodedb/src/control/server/native/codec.rs b/nodedb/src/control/server/native/codec.rs index 3d59346b6..c44667807 100644 --- a/nodedb/src/control/server/native/codec.rs +++ b/nodedb/src/control/server/native/codec.rs @@ -214,6 +214,7 @@ mod tests { columns: vec!["x".into()], rows: vec![vec![Value::Integer(42)]], rows_affected: 0, + command: None, }, 100, ); diff --git a/nodedb/src/control/server/native/dispatch/cluster_array.rs b/nodedb/src/control/server/native/dispatch/cluster_array.rs new file mode 100644 index 000000000..7a66b1bc8 --- /dev/null +++ b/nodedb/src/control/server/native/dispatch/cluster_array.rs @@ -0,0 +1,57 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `ClusterArray` plan dispatch for the native protocol. +//! +//! `sql_loop.rs` and `single_task.rs` intercept a `PhysicalPlan::ClusterArray` +//! task right after their own in-transaction routing gate and delegate to the +//! shared, protocol-neutral core +//! (`shared::cluster_array_dispatch::execute_cluster_array`), then convert +//! the outcome into native wire columns/rows or a count-bearing outcome. Metering +//! is not applied on this path, matching pgwire's `ClusterArray` +//! short-circuit, which also does not meter it. + +use nodedb_types::Value; + +use crate::control::server::response_shape::schema::OutputSchema; +use crate::control::server::response_shape::types::DmlOutcome; +use crate::control::server::shared::authorization::AuthorizedTask; +use crate::control::server::shared::cluster_array_dispatch::{ + ClusterArrayShaped, execute_cluster_array, +}; + +use super::DispatchCtx; +use super::conversion::to_native_columns_rows; + +/// What one `ClusterArrayOp` answers with, in native wire shape. +pub(crate) enum ClusterArrayOutcome { + /// A read's rows (`Slice` / `Agg`), converted to native columns/rows. + Rows { + columns: Vec, + rows: Vec>, + notice: Option, + }, + /// A write's count-bearing outcome (`Put` / `Delete`): the caller folds + /// it into the statement's tag. + Affected(DmlOutcome), +} + +/// Execute a single `ClusterArrayOp` via the shared core and convert its +/// outcome into native wire shape. +pub(crate) async fn dispatch_cluster_array_task( + ctx: &DispatchCtx<'_>, + authorized: AuthorizedTask, + projection: Option<&OutputSchema>, +) -> crate::Result { + match execute_cluster_array(ctx.state, ctx.auth_context(), authorized, projection).await? { + ClusterArrayShaped::Rows(mut shaped) => { + let notice = shaped.notice.take(); + let (columns, rows) = to_native_columns_rows(&shaped); + Ok(ClusterArrayOutcome::Rows { + columns, + rows, + notice, + }) + } + ClusterArrayShaped::Affected(outcome) => Ok(ClusterArrayOutcome::Affected(outcome)), + } +} diff --git a/nodedb/src/control/server/native/dispatch/conversion.rs b/nodedb/src/control/server/native/dispatch/conversion.rs index 3df2f2c86..9e8e4649c 100644 --- a/nodedb/src/control/server/native/dispatch/conversion.rs +++ b/nodedb/src/control/server/native/dispatch/conversion.rs @@ -7,7 +7,9 @@ use nodedb_types::protocol::NativeResponse; use crate::bridge::envelope::Response; use crate::control::server::native::sqlstate_code::sqlstate_error; -use crate::control::server::response_shape::types::ShapedRows; +use crate::control::server::response_shape::types::{ + DmlFoldError, DmlOutcome, FoldedTag, ShapedRows, +}; use crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate; use crate::control::server::shared::ddl::{DdlError, DdlResult}; @@ -61,6 +63,10 @@ pub(crate) fn error_to_native(seq: u64, e: &crate::Error) -> NativeResponse { let (_severity, sqlstate, message) = error_code_to_sqlstate(code); (sqlstate, message) } + crate::Error::Shaping(e) => ( + crate::control::server::pgwire::types::error_map::numeric_code_to_sqlstate(e.code()), + e.message().to_string(), + ), other => ("XX000", format!("{other}")), }; let ndb_code = crate::error_classify::classify(e).code().0; @@ -100,6 +106,13 @@ pub(crate) fn shape_error_to_native(seq: u64, e: &nodedb_types::NodeDbError) -> NativeResponse::error_with_code(seq, "XX000", e.message().to_string(), e.code().0) } +/// Render a statement-tag fold refusal as a native error frame. Two tasks of +/// one statement disagreeing on their verb is a planner bug, so it is an +/// internal error, the same class pgwire's `dml_fold_error_to_pg` renders. +pub(crate) fn dml_fold_error_to_native(seq: u64, e: &DmlFoldError) -> NativeResponse { + sqlstate_error(seq, "XX000", e.to_string()) +} + /// Render an error [`Response`] from the Data Plane as a native error frame. /// /// The Data Plane already classified the failure into a deterministic @@ -140,7 +153,7 @@ pub(crate) fn error_code_to_native( /// Encode a protocol-neutral DDL dispatch result into a single /// `NativeResponse`. /// -/// Reduction mirrors the previous pgwire→native bridge: on error, an error +/// Reduction mirrors the pgwire→native bridge: on error, an error /// frame carrying the neutral SQLSTATE + message; otherwise the first /// row-returning / status / empty result determines the response (a status tag /// becomes a single-column status row, a row result becomes a columns+rows @@ -163,10 +176,27 @@ pub(crate) fn ddl_result_to_native( message, }) => NativeResponse::error_with_code(seq, sqlstate, message, code.0), // Unknown pgwire response variants are dropped during translation, so - // the first element is the first meaningful result — mirroring the - // previous bridge, which returned on the first known variant. + // the first element is the first meaningful result — the bridge + // returns on the first known variant. Ok(results) => match results.into_iter().next() { - Some(DdlResult::Status { command, .. }) => NativeResponse::status_row(seq, command), + Some(DdlResult::Status { + command, + rows_affected, + }) => match rows_affected { + // A count-bearing status is a DML answer (the `{ ... }` + // document INSERT / UPSERT, the graph edge and label DSL): + // its verb and count, rendered by the same rule the dispatch + // loop's folded tag renders with. pgwire renders it as the + // ` ` command tag. + Some(affected) => { + let mut r = NativeResponse::ok(seq); + apply_dml_status(&mut r, &command, affected); + r + } + // A count-less DDL keeps the status row, whose `Some(1)` + // `rows_affected` sentinel means "one command ran". + None => NativeResponse::status_row(seq, command), + }, Some(DdlResult::Rows(shaped)) => { let (columns, rows) = to_native_columns_rows(&shaped); NativeResponse { @@ -175,6 +205,7 @@ pub(crate) fn ddl_result_to_native( columns: Some(columns), rows: Some(rows), rows_affected: None, + command: None, watermark_lsn: 0, error: None, auth: None, @@ -186,14 +217,15 @@ pub(crate) fn ddl_result_to_native( } } -/// Build the native response for a completed Calvin transaction, surfacing -/// RETURNING rows when the write carried them. +/// Build the native response for a completed Calvin transaction: its +/// RETURNING rows, when a task carried them, and the statement's one folded +/// tag as `(rows_affected, command)`. /// -/// `apply_result` is the applied Data-Plane response drained from the sidecar and -/// `plans` is the completed batch's plans, in dispatch order. A RETURNING plan -/// shapes the payload into native columns/rows; otherwise the batch's -/// count-bearing plan (if any) reports `rows_affected` READ FROM the applied -/// response. +/// `apply_result` is the applied Data-Plane response drained from the sidecar +/// and `plans` is the completed batch's plans, in dispatch order. The fold is +/// the protocol-neutral `response_shape::calvin_fold` pgwire renders too, so +/// a three-row implicit-edge insert answers `(3, INSERT)` here and +/// `INSERT 0 3` there. /// /// There is deliberately no per-statement fallback count. The number of /// dispatched tasks is not the number of affected rows — a single-row delete @@ -209,78 +241,63 @@ pub(crate) fn calvin_native_response( tenant_id: nodedb_types::TenantId, auth: &crate::control::security::auth_context::AuthContext, ) -> NativeResponse { - use crate::control::server::response_shape::compose::{ - ShapeOutcome, shape_response_materialized, + use crate::control::server::response_shape::calvin_fold::{ + CalvinFoldCtx, CalvinFoldError, fold_calvin_batch, }; - use crate::control::server::response_shape::redaction::QueryRedaction; - use crate::control::server::response_shape::request::MaterializedShapeRequest; - use crate::control::server::response_shape::types::{PlanKind, describe_plan}; - let returning_plan = plans - .iter() - .find(|p| matches!(describe_plan(p), PlanKind::ReturningRows)); - let dml_plan = plans - .iter() - .find(|p| matches!(describe_plan(p), PlanKind::DmlResult(_))); - - let redaction = returning_plan.map(|plan| QueryRedaction::for_plan(tenant_id, auth, plan)); - if let (Some(resp), Some(plan)) = (apply_result.as_ref(), returning_plan) - && matches!(describe_plan(plan), PlanKind::ReturningRows) - && let Ok(ShapeOutcome::Rows(shaped)) = - shape_response_materialized(MaterializedShapeRequest { - payload: resp.payload.as_bytes(), - plan, - plan_kind: PlanKind::ReturningRows, - projection: None, - state, - database_id, - tenant_id, - redaction: redaction.as_ref().map(|r| r.ctx(&state.redaction)), - // No projection, so no Control-Plane computed column to resolve. - sequences: None, - }) - { - let (cols, rows) = to_native_columns_rows(&shaped); - let mut r = NativeResponse::ok(seq); - r.watermark_lsn = resp.watermark_lsn.as_u64(); - if !cols.is_empty() { - r.columns = Some(cols); - } - r.rows = Some(rows); - return r; - } + let plan_refs: Vec<&crate::bridge::envelope::PhysicalPlan> = plans.iter().collect(); + let fold = match fold_calvin_batch( + &plan_refs, + apply_result.as_ref(), + &CalvinFoldCtx { + // The native protocol announces no output columns ahead of the rows. + projection: None, + state, + tenant_id, + database_id, + auth, + }, + ) { + Ok(fold) => fold, + Err(CalvinFoldError::Verb(e)) => return dml_fold_error_to_native(seq, &e), + Err(CalvinFoldError::Shape(e)) => return error_to_native(seq, &e), + }; - // Plain write: surface the affected count the mutation itself reported. let mut r = NativeResponse::ok(seq); if let Some(resp) = &apply_result { r.watermark_lsn = resp.watermark_lsn.as_u64(); } - // A batch with no count-bearing plan (pure graph / vector / DDL work) has no - // row count to report, and says so by leaving `rows_affected` unset rather - // than inventing one from the task count. - if dml_plan.is_some() { - let count = apply_result.as_ref().map_or_else( - || { - Err(crate::Error::Internal { - detail: "native Calvin write completed with no applied response to read its \ - affected-row count from" - .to_owned(), - }) - }, - |resp| { - crate::control::server::shared::sql::staging_predicates::require_affected_count( - resp.payload.as_bytes(), - ) - }, - ); - match count { - Ok(n) => r.rows_affected = Some(n), - Err(e) => return error_to_native(seq, &e), + if let Some(shaped) = fold.rows { + let (cols, rows) = to_native_columns_rows(&shaped); + if !cols.is_empty() { + r.columns = Some(cols); } + r.rows = Some(rows); + } + // A batch with no count-bearing plan (vector / DDL work) has no row count + // or verb to report, and says so by leaving both unset rather than + // inventing a count from the task count. + match fold.tag { + Some(FoldedTag::Dml(outcome)) => apply_dml_outcome(&mut r, outcome), + Some(FoldedTag::Opaque) | None => {} } r } +/// Write a statement's count-bearing outcome onto a native response: the +/// verb always, the count only when the verb carries one (`TRUNCATE` does +/// not, matching the bare tag pgwire answers with). +pub(crate) fn apply_dml_outcome(r: &mut NativeResponse, outcome: DmlOutcome) { + apply_dml_status(r, outcome.verb, outcome.affected); +} + +/// The one rendering rule behind [`apply_dml_outcome`], for a verb that is +/// not a `'static` tag name (a DDL-router `DdlResult::Status` command). +fn apply_dml_status(r: &mut NativeResponse, verb: &str, affected: u64) { + r.rows_affected = DmlOutcome::verb_carries_count(verb).then_some(affected); + r.command = Some(verb.to_owned()); +} + /// Convert protocol-neutral `ShapedRows` (produced by /// `response_shape::compose::shape_response_materialized`) into native wire /// columns/rows: each typed cell is carried as-is; a column absent from a @@ -441,6 +458,55 @@ mod tests { ); } + /// A count-bearing DDL-router status (the `{ ... }` document INSERT) is + /// a DML answer: `(rows_affected, command)` exactly as the dispatch + /// loop's folded tag reports, never a status row with no verb. + #[test] + fn count_bearing_ddl_status_reports_verb_and_count() { + let response = ddl_result_to_native( + 1, + Ok(vec![DdlResult::Status { + command: "INSERT".to_owned(), + rows_affected: Some(1), + }]), + ); + assert_eq!(response.rows_affected, Some(1)); + assert_eq!(response.command.as_deref(), Some("INSERT")); + assert!(response.rows.is_none()); + assert!(response.columns.is_none()); + } + + /// `TRUNCATE` carries no count on any protocol: the verb alone. + #[test] + fn count_less_verb_status_reports_verb_only() { + let response = ddl_result_to_native( + 1, + Ok(vec![DdlResult::Status { + command: "TRUNCATE".to_owned(), + rows_affected: Some(9), + }]), + ); + assert_eq!(response.rows_affected, None); + assert_eq!(response.command.as_deref(), Some("TRUNCATE")); + } + + /// A count-less DDL keeps the status row. + #[test] + fn count_less_ddl_status_keeps_the_status_row() { + let response = ddl_result_to_native( + 1, + Ok(vec![DdlResult::Status { + command: "CREATE COLLECTION".to_owned(), + rows_affected: None, + }]), + ); + assert_eq!(response.command, None); + assert_eq!( + response.rows, + Some(vec![vec![Value::String("CREATE COLLECTION".to_owned())]]) + ); + } + /// A site that renders a more specific SQLSTATE than the error implies /// must still take the classification from the error rather than from the /// SQLSTATE it just chose: `42601` is shared by several conditions, while diff --git a/nodedb/src/control/server/native/dispatch/direct_ops.rs b/nodedb/src/control/server/native/dispatch/direct_ops.rs index b80bc4a15..73d7d7e86 100644 --- a/nodedb/src/control/server/native/dispatch/direct_ops.rs +++ b/nodedb/src/control/server/native/dispatch/direct_ops.rs @@ -367,6 +367,7 @@ pub(crate) async fn handle_direct_op( DispatchClass::MultiShard { .. } ); if route_to_calvin { + let plans: Vec = tasks.iter().map(|task| task.plan.clone()).collect(); match dispatch_authorized_tasks_to_calvin( ctx.state, authorized_tasks, @@ -378,13 +379,18 @@ pub(crate) async fn handle_direct_op( ) .await { - // No RETURNING possible here, so the Response carries no rows — - // report one row-affected per task. - Ok(_apply) => { - let mut r = NativeResponse::ok(seq); - r.rows_affected = Some(tasks.len() as u64); - r - } + // The count and verb come from the applied response the + // batch's count-bearing plan reported, never from the task + // count: an implicit-edge cleanup task is not a row. + Ok(apply) => super::conversion::calvin_native_response( + seq, + apply, + &plans, + ctx.state, + ctx.database_id(), + tenant_id, + ctx.auth_context(), + ), Err(e) => error_to_native(seq, &e), } } else { diff --git a/nodedb/src/control/server/native/dispatch/mod.rs b/nodedb/src/control/server/native/dispatch/mod.rs index fbdd95116..344088f22 100644 --- a/nodedb/src/control/server/native/dispatch/mod.rs +++ b/nodedb/src/control/server/native/dispatch/mod.rs @@ -4,6 +4,7 @@ mod admission_op; mod auth; +mod cluster_array; mod conversion; mod ctx; mod direct_ops; @@ -27,8 +28,9 @@ mod transaction_savepoint; pub(crate) use admission_op::admission_operation; pub(crate) use auth::{NativeAuthOutcome, handle_auth, handle_ping}; pub(crate) use conversion::{ - ddl_result_to_native, error_code_to_native, error_response_to_native, error_to_native, - error_to_native_with_sqlstate, shape_error_to_native, to_native_columns_rows, + apply_dml_outcome, ddl_result_to_native, dml_fold_error_to_native, error_code_to_native, + error_response_to_native, error_to_native, error_to_native_with_sqlstate, + shape_error_to_native, to_native_columns_rows, }; pub(crate) use ctx::DispatchCtx; pub(crate) use direct_ops::handle_direct_op; diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs b/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs index 34bf24e58..c0c0282fb 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs @@ -255,6 +255,7 @@ pub(crate) fn build_truncate( ) -> crate::Result { Ok(PhysicalPlan::Kv(KvOp::Truncate { collection: QualifiedCollection::new(ctx.database_id(), collection), + restart_identity: false, })) } diff --git a/nodedb/src/control/server/native/dispatch/response.rs b/nodedb/src/control/server/native/dispatch/response.rs index ac5f31cad..25f012da1 100644 --- a/nodedb/src/control/server/native/dispatch/response.rs +++ b/nodedb/src/control/server/native/dispatch/response.rs @@ -8,9 +8,14 @@ use crate::bridge::envelope::{PhysicalPlan, Response, Status}; use crate::control::server::response_shape::compose::{ShapeOutcome, shape_response_materialized}; use crate::control::server::response_shape::redaction::QueryRedaction; use crate::control::server::response_shape::request::MaterializedShapeRequest; -use crate::control::server::response_shape::types::describe_plan; +use crate::control::server::response_shape::types::{ + PlanKind, describe_plan, payload_to_dml_outcome, +}; -use super::{DispatchCtx, error_response_to_native, shape_error_to_native, to_native_columns_rows}; +use super::{ + DispatchCtx, apply_dml_outcome, error_response_to_native, error_to_native, + shape_error_to_native, to_native_columns_rows, +}; pub(crate) fn data_plane_response_to_native( ctx: &DispatchCtx<'_>, @@ -21,16 +26,29 @@ pub(crate) fn data_plane_response_to_native( if response.status == Status::Error { return error_response_to_native(seq, response); } + let plan_kind = describe_plan(plan); if response.payload.is_empty() { let mut native = NativeResponse::ok(seq); native.watermark_lsn = response.watermark_lsn.as_u64(); + // A count-bearing write with no payload is a handler bug: the count + // reader refuses it rather than answering without a count. + if let PlanKind::DmlResult(_) | PlanKind::DmlResultByOp = plan_kind { + return match payload_to_dml_outcome(&[], plan_kind) { + Ok(Some(outcome)) => { + apply_dml_outcome(&mut native, outcome); + native + } + Ok(None) => native, + Err(e) => error_to_native(seq, &e), + }; + } return native; } let redaction = QueryRedaction::for_plan(ctx.tenant_id(), ctx.auth_context(), plan); match shape_response_materialized(MaterializedShapeRequest { payload: &response.payload, plan, - plan_kind: describe_plan(plan), + plan_kind, projection: None, state: ctx.state, database_id: ctx.database_id(), @@ -47,15 +65,23 @@ pub(crate) fn data_plane_response_to_native( columns: Some(columns), rows: Some(rows), rows_affected: None, + command: None, watermark_lsn: response.watermark_lsn.as_u64(), error: None, auth: None, warnings: shaped.notice.into_iter().collect(), } } + // A count-bearing write reports the count and verb the payload + // carries; an opaque execution reports neither. Ok(ShapeOutcome::Passthrough) => { let mut native = NativeResponse::ok(seq); native.watermark_lsn = response.watermark_lsn.as_u64(); + match payload_to_dml_outcome(&response.payload, plan_kind) { + Ok(Some(outcome)) => apply_dml_outcome(&mut native, outcome), + Ok(None) => {} + Err(e) => return error_to_native(seq, &e), + } native } Err(error) => shape_error_to_native(seq, &error), diff --git a/nodedb/src/control/server/native/dispatch/session_ops.rs b/nodedb/src/control/server/native/dispatch/session_ops.rs index 5d9b1f84c..5edfdaad7 100644 --- a/nodedb/src/control/server/native/dispatch/session_ops.rs +++ b/nodedb/src/control/server/native/dispatch/session_ops.rs @@ -89,6 +89,7 @@ pub(crate) fn show_all(ctx: &DispatchCtx<'_>, seq: u64) -> NativeResponse { columns: Some(vec!["name".into(), "setting".into()]), rows: Some(rows), rows_affected: None, + command: None, watermark_lsn: 0, error: None, auth: None, @@ -104,6 +105,7 @@ fn setting_row(seq: u64, value: String) -> NativeResponse { columns: Some(vec!["setting".into()]), rows: Some(vec![vec![Value::String(value)]]), rows_affected: None, + command: None, watermark_lsn: 0, error: None, auth: None, diff --git a/nodedb/src/control/server/native/dispatch/single_task.rs b/nodedb/src/control/server/native/dispatch/single_task.rs index a3f10f5b7..b417f29d3 100644 --- a/nodedb/src/control/server/native/dispatch/single_task.rs +++ b/nodedb/src/control/server/native/dispatch/single_task.rs @@ -7,16 +7,20 @@ use nodedb_types::protocol::NativeResponse; use crate::bridge::envelope::{Payload, Response, Status}; +use crate::control::server::response_shape::types::staged_dml_outcome; use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; use crate::control::server::shared::quota_admission::admit_quota_for_dispatch; use crate::control::server::shared::session::staging_gate::{ - InTxnRoute, StagingGateError, route_in_tx_write, + InTxnRoute, StagedTagKind, StagingGateError, route_in_tx_write, }; use crate::types::{Lsn, RequestId}; use super::raw_dispatch::dispatch_authorized_single_task; use super::response::data_plane_response_to_native; -use super::{DispatchCtx, error_code_to_native, error_to_native, error_to_native_with_sqlstate}; +use super::{ + DispatchCtx, apply_dml_outcome, error_code_to_native, error_to_native, + error_to_native_with_sqlstate, +}; /// Dispatch one plan via the gateway (when wired) or the local SPSC path, /// converting the Data-Plane response into a `NativeResponse`. @@ -49,8 +53,8 @@ pub(super) async fn dispatch_single_task( let task = authorized.into_staging_task(); // Cloned before `route_in_tx_write` consumes `task`, so a staged write - // whose outcome carries a real affected-count/computed-value payload - // (e.g. `KvBatchPut`'s `{"inserted": n}`) can be shaped into the + // whose outcome carries a computed-value payload (KV `Incr` / `Cas` / + // `GetSet`, see `StagedTagKind::RawPayload`) can be shaped into the // response the same way the non-staged branch below shapes it. let plan_for_staged_response = task.plan.clone(); @@ -97,12 +101,17 @@ pub(super) async fn dispatch_single_task( .await { Ok(InTxnRoute::Read(routed_task)) => *routed_task, - Ok(InTxnRoute::Buffered) => { - let mut r = NativeResponse::ok(seq); - r.rows_affected = Some(1); - return r; - } + // A buffered write applies at COMMIT: no count and no verb yet. + Ok(InTxnRoute::Buffered) => return NativeResponse::ok(seq), Ok(InTxnRoute::Staged(outcome)) => { + // The staging gate decided the verb and counted the rows it + // applied to the overlay; only a computed-value payload is + // forwarded as-is. + if !matches!(outcome.kind, StagedTagKind::RawPayload) { + let mut r = NativeResponse::ok(seq); + apply_dml_outcome(&mut r, staged_dml_outcome(outcome.kind, outcome.affected)); + return r; + } let synthetic = Response { request_id: RequestId::new(0), status: Status::Ok, @@ -123,6 +132,49 @@ pub(super) async fn dispatch_single_task( } }; + // `ClusterArray` plans are handled entirely on the Control Plane by the + // `ArrayCoordinator` — they must never reach the SPSC bridge or the + // trigger/DML machinery. A write reshapes into per-shard `ArrayOp` tasks + // inside `route_in_tx_write` above and never reaches here as + // `ClusterArray`; only reads (`Slice`/`Agg`) and autocommit `Put`/`Delete` + // arrive as this plan by the time the route resolves to `Read`. + // Intercepted here, before `dispatch_authorized_single_task` would + // otherwise route it toward the Data Plane. No SQL output schema exists + // on this direct-op path, so no projection narrows the shaped rows. + // Metering is not applied here, matching pgwire's `ClusterArray` + // short-circuit, which also does not meter this path. + if matches!( + task.plan, + crate::bridge::envelope::PhysicalPlan::ClusterArray(_) + ) { + let authorized = match super::sql_gateway::authorize_native_task(ctx, &task) { + Ok(a) => a, + Err(e) => return error_to_native(seq, &e), + }; + return match super::cluster_array::dispatch_cluster_array_task(ctx, authorized, None).await + { + Ok(super::cluster_array::ClusterArrayOutcome::Rows { + columns, + rows, + notice, + }) => { + let mut r = NativeResponse::ok(seq); + if !columns.is_empty() { + r.columns = Some(columns); + } + r.rows = Some(rows); + r.warnings = notice.into_iter().collect(); + r + } + Ok(super::cluster_array::ClusterArrayOutcome::Affected(outcome)) => { + let mut r = NativeResponse::ok(seq); + apply_dml_outcome(&mut r, outcome); + r + } + Err(e) => error_to_native(seq, &e), + }; + } + let plan_for_response = task.plan.clone(); let task_vshard = task.vshard_id; match dispatch_authorized_single_task( diff --git a/nodedb/src/control/server/native/dispatch/sql_admin.rs b/nodedb/src/control/server/native/dispatch/sql_admin.rs index e6d35d208..f290c8c13 100644 --- a/nodedb/src/control/server/native/dispatch/sql_admin.rs +++ b/nodedb/src/control/server/native/dispatch/sql_admin.rs @@ -93,6 +93,7 @@ pub(super) async fn handle_explain(ctx: &DispatchCtx<'_>, seq: u64, sql: &str) - columns: Some(vec!["plan".into()]), rows: Some(vec![vec![Value::String(format!("DDL: {inner_sql}"))]]), rows_affected: None, + command: None, watermark_lsn: 0, error: None, auth: None, @@ -148,6 +149,7 @@ pub(super) async fn handle_explain(ctx: &DispatchCtx<'_>, seq: u64, sql: &str) - columns: Some(vec!["plan".into()]), rows: Some(vec![vec![Value::String(plan_text)]]), rows_affected: None, + command: None, watermark_lsn: 0, error: None, auth: None, diff --git a/nodedb/src/control/server/native/dispatch/sql_dispatch_task.rs b/nodedb/src/control/server/native/dispatch/sql_dispatch_task.rs index 48132f864..9d5a8822c 100644 --- a/nodedb/src/control/server/native/dispatch/sql_dispatch_task.rs +++ b/nodedb/src/control/server/native/dispatch/sql_dispatch_task.rs @@ -48,6 +48,7 @@ pub(super) async fn dispatch_task( clauses: _, returning: _, resolved_inserts: None, + resolved_insert_identities: _, source_rows: _, rls_filters: _, rls_write_check: _, diff --git a/nodedb/src/control/server/native/dispatch/sql_loop.rs b/nodedb/src/control/server/native/dispatch/sql_loop.rs index 9905d199c..45797f8c4 100644 --- a/nodedb/src/control/server/native/dispatch/sql_loop.rs +++ b/nodedb/src/control/server/native/dispatch/sql_loop.rs @@ -12,12 +12,18 @@ use nodedb_types::protocol::NativeResponse; use nodedb_types::value::Value; use crate::bridge::envelope::Status; +use crate::control::planner::calvin::write_class::{ + plan_counts_toward_statement_tag, plans_have_user_write, +}; use crate::control::sequence::SessionSequenceAccess; use crate::control::server::response_shape::compose::{ShapeOutcome, shape_response_materialized}; use crate::control::server::response_shape::redaction::QueryRedaction; use crate::control::server::response_shape::request::MaterializedShapeRequest; use crate::control::server::response_shape::schema::OutputSchema; -use crate::control::server::response_shape::types::{PlanKind, describe_plan}; +use crate::control::server::response_shape::types::{ + DmlOutcome, FoldedTag, PlanKind, StatementTag, describe_plan, payload_to_dml_outcome, + staged_dml_outcome, +}; use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; use crate::control::server::shared::quota_admission::admit_quota_for_dispatch; use crate::control::server::shared::session::expander_stage::{ @@ -32,8 +38,9 @@ use nodedb_physical::physical_task::PhysicalTask; use super::sql_dispatch_task::dispatch_task; use super::streaming::SqlOutcome; use super::{ - DispatchCtx, error_code_to_native, error_response_to_native, error_to_native, - error_to_native_with_sqlstate, shape_error_to_native, to_native_columns_rows, + DispatchCtx, apply_dml_outcome, dml_fold_error_to_native, error_code_to_native, + error_response_to_native, error_to_native, error_to_native_with_sqlstate, + shape_error_to_native, to_native_columns_rows, }; use crate::control::server::native::sqlstate_code::sqlstate_error; @@ -43,6 +50,17 @@ fn resp(r: NativeResponse) -> SqlOutcome { SqlOutcome::Response(Box::new(r)) } +/// Fold one task's count-bearing outcome into the statement's tag, rendering +/// a verb mismatch as the native error frame. +fn fold_dml( + tag: &mut StatementTag, + seq: u64, + outcome: DmlOutcome, +) -> Result<(), Box> { + tag.fold(outcome) + .map_err(|e| Box::new(dml_fold_error_to_native(seq, &e))) +} + /// Run the per-task dispatch loop for a planned, non-streamed task set, /// materializing all rows/columns/affected-count into a single /// [`SqlOutcome::Response`]. @@ -50,6 +68,13 @@ fn resp(r: NativeResponse) -> SqlOutcome { /// Called from `execute_planned` after the streaming fast path has been /// ruled out (or declined). Buffers writes when in an explicit transaction /// block, exactly like the pgwire dispatch loop. +/// +/// The statement's `rows_affected` and `command` come from one +/// [`StatementTag`] folded over every task, the same fold pgwire renders as +/// its command tag: a count-bearing task contributes its reported count and +/// verb, an opaque task (buffered write, index or graph maintenance, +/// computed-value payload) contributes nothing. Neither field is ever +/// synthesised from the task count. pub(super) async fn run_dispatch_loop( ctx: &DispatchCtx<'_>, seq: u64, @@ -62,12 +87,15 @@ pub(super) async fn run_dispatch_loop( let mut all_rows: Vec> = Vec::new(); let mut warnings: Vec = Vec::new(); let mut last_lsn = 0u64; - let mut total_affected = 0u64; + let mut statement_tag = StatementTag::default(); // Checked once rather than per task — metering is disabled by default, so // this keeps the per-task extraction below (which clones the collection // name) a true no-op on the hot path for every deployment that hasn't // turned it on. let metering_enabled = ctx.state.metering_config.enabled; + // A derived implicit-edge write beside the user's own never answers the + // statement, exactly as Calvin's deposit rule has it. + let has_user_write = plans_have_user_write(tasks.iter().map(|t| &t.plan)); // Session-scoped sequence access for the statement's Control-Plane // computed columns: the registry plus this connection's `currval` map. let session_sequences = ctx.sessions.sequence_values(ctx.peer_addr); @@ -186,7 +214,8 @@ pub(super) async fn run_dispatch_loop( in_transaction_returning_unsupported(), )); } - total_affected += 1; + // Applied at COMMIT: no count and no verb yet. + statement_tag.fold_opaque(); continue; } Ok(InTxnRoute::Staged(outcome)) => { @@ -197,8 +226,20 @@ pub(super) async fn run_dispatch_loop( in_transaction_returning_unsupported(), )); } - if matches!(outcome.kind, StagedTagKind::RawPayload) && !outcome.payload.is_empty() - { + // The staging gate decided the verb and counted the rows it + // applied to the overlay; only a computed-value payload + // (`RawPayload`) is shaped instead of folded. + if !matches!(outcome.kind, StagedTagKind::RawPayload) { + if let Err(e) = fold_dml( + &mut statement_tag, + seq, + staged_dml_outcome(outcome.kind, outcome.affected), + ) { + return SqlOutcome::Response(e); + } + } else if outcome.payload.is_empty() { + statement_tag.fold_opaque(); + } else { let plan_kind = describe_plan(&plan_for_staged_response); let redaction = QueryRedaction::for_plan( ctx.tenant_id(), @@ -226,13 +267,10 @@ pub(super) async fn run_dispatch_loop( } all_rows.extend(rows); } - Ok(ShapeOutcome::Passthrough) => { - total_affected += 1; - } + // The value rides in the payload, never as a count. + Ok(ShapeOutcome::Passthrough) => statement_tag.fold_opaque(), Err(e) => return resp(shape_error_to_native(seq, &e)), } - } else { - total_affected += outcome.affected as u64; } continue; } @@ -242,8 +280,53 @@ pub(super) async fn run_dispatch_loop( } }; + // `ClusterArray` plans are handled entirely on the Control Plane by + // the `ArrayCoordinator` — they must never reach the SPSC bridge or + // the trigger/DML machinery. A write reshapes into per-shard + // `ArrayOp` tasks inside the staging gate above and never reaches + // here as `ClusterArray`; only reads (`Slice`/`Agg`) and autocommit + // `Put`/`Delete` arrive as this plan by the time `routed` resolves to + // `Read`. Intercepted here, before `dispatch_task` would otherwise + // route it through the gateway toward the Data Plane. Metering is + // not applied here, matching pgwire's `ClusterArray` short-circuit, + // which also does not meter this path. + if matches!( + task.plan, + crate::bridge::envelope::PhysicalPlan::ClusterArray(_) + ) { + let authorized = match super::sql_gateway::authorize_native_task(ctx, &task) { + Ok(a) => a, + Err(e) => return resp(error_to_native(seq, &e)), + }; + match super::cluster_array::dispatch_cluster_array_task(ctx, authorized, output_schema) + .await + { + Ok(super::cluster_array::ClusterArrayOutcome::Rows { + columns, + rows, + notice, + }) => { + if let Some(n) = notice { + warnings.push(n); + } + if !columns.is_empty() && all_columns.is_none() { + all_columns = Some(columns); + } + all_rows.extend(rows); + } + Ok(super::cluster_array::ClusterArrayOutcome::Affected(outcome)) => { + if let Err(e) = fold_dml(&mut statement_tag, seq, outcome) { + return SqlOutcome::Response(e); + } + } + Err(e) => return resp(error_to_native(seq, &e)), + } + continue; + } + let plan_for_response = task.plan.clone(); let task_vshard = task.vshard_id; + let task_database_id = task.database_id; let (task_resp, shard_watermarks, dist_reads) = match dispatch_task(ctx, task).await { Ok(r) => r, Err(e) => return resp(error_to_native(seq, &e)), @@ -289,31 +372,34 @@ pub(super) async fn run_dispatch_loop( return resp(error_response_to_native(seq, &task_resp)); } + // --- TRUNCATE RESTART IDENTITY --- + // Autocommit only: a buffered truncate restarts its sequences at + // COMMIT. Same rule as the pgwire dispatch loop. + if let Some((collection, true)) = plan_for_response.truncate_target() { + ctx.state + .sequence_registry + .restart_sequences_for_collection( + task_database_id.as_u64(), + ctx.tenant_id().as_u64(), + collection.as_str(), + ); + } + last_lsn = task_resp.watermark_lsn.as_u64(); // This task's own row count, for metering below — distinct from - // `total_affected`/`all_rows`, which accumulate across every task in + // `statement_tag`/`all_rows`, which accumulate across every task in // the loop. let mut task_rows: Option = None; let plan_kind = describe_plan(&plan_for_response); - if let crate::control::server::response_shape::types::PlanKind::DmlResult(_) = plan_kind { - // A count-bearing write must report the rows it actually touched. - // Adding 1 per dispatched task instead, as the empty-payload - // branch below does, would report a row for a delete that removed - // nothing and for an `ON CONFLICT DO NOTHING` insert that skipped. - match crate::control::server::shared::sql::staging_predicates::require_affected_count( - &task_resp.payload, - ) { - Ok(n) => { - total_affected += n; - task_rows = Some(n); - } - Err(e) => return resp(error_to_native(seq, &e)), - } - } else if task_resp.payload.is_empty() { - // Not a count-bearing plan (graph / vector / index write): one unit - // of work per dispatched task, as before. - total_affected += 1; + let counts_toward_tag = + plan_counts_toward_statement_tag(&plan_for_response, has_user_write); + let count_bearing = matches!(plan_kind, PlanKind::DmlResult(_) | PlanKind::DmlResultByOp); + if task_resp.payload.is_empty() && !count_bearing { + // Not a count-bearing plan (graph / vector / index write): no + // count and no verb to report. A count-bearing plan with no + // payload falls through so the count reader refuses it. + statement_tag.fold_opaque(); } else { let redaction = QueryRedaction::for_plan(ctx.tenant_id(), ctx.auth_context(), &plan_for_response); @@ -339,8 +425,25 @@ pub(super) async fn run_dispatch_loop( task_rows = Some(rows.len() as u64); all_rows.extend(rows); } + // A count-bearing write reports the rows it touched and its + // verb; an opaque execution reports neither. Counting one per + // dispatched task instead would report a row for a delete + // that removed nothing and for an `ON CONFLICT DO NOTHING` + // insert that skipped. + Ok(ShapeOutcome::Passthrough) if !counts_toward_tag => { + statement_tag.fold_opaque(); + } Ok(ShapeOutcome::Passthrough) => { - total_affected += 1; + match payload_to_dml_outcome(&task_resp.payload, plan_kind) { + Ok(Some(outcome)) => { + task_rows = Some(outcome.affected); + if let Err(e) = fold_dml(&mut statement_tag, seq, outcome) { + return SqlOutcome::Response(e); + } + } + Ok(None) => statement_tag.fold_opaque(), + Err(e) => return resp(error_to_native(seq, &e)), + } } Err(e) => return resp(shape_error_to_native(seq, &e)), } @@ -355,23 +458,16 @@ pub(super) async fn run_dispatch_loop( } } - if all_rows.is_empty() { - let mut r = NativeResponse::ok(seq); - r.rows_affected = Some(total_affected); - r.watermark_lsn = last_lsn; - r.warnings = warnings; - resp(r) - } else { - resp(NativeResponse { - seq, - status: nodedb_types::protocol::ResponseStatus::Ok, - columns: all_columns, - rows: Some(all_rows), - rows_affected: Some(total_affected), - watermark_lsn: last_lsn, - error: None, - auth: None, - warnings, - }) + let mut r = NativeResponse::ok(seq); + r.watermark_lsn = last_lsn; + r.warnings = warnings; + if !all_rows.is_empty() { + r.columns = all_columns; + r.rows = Some(all_rows); + } + match statement_tag.finish() { + Some(FoldedTag::Dml(outcome)) => apply_dml_outcome(&mut r, outcome), + Some(FoldedTag::Opaque) | None => {} } + resp(r) } diff --git a/nodedb/src/control/server/native/session/session_chunk.rs b/nodedb/src/control/server/native/session/session_chunk.rs index 15e31fa5e..bce71ba9c 100644 --- a/nodedb/src/control/server/native/session/session_chunk.rs +++ b/nodedb/src/control/server/native/session/session_chunk.rs @@ -36,6 +36,7 @@ pub(super) fn chunk_large_response( columns: response.columns.clone(), rows: Some(rows[..total_rows.min(100)].to_vec()), rows_affected: None, + command: None, watermark_lsn: response.watermark_lsn, error: None, auth: None, @@ -74,6 +75,11 @@ pub(super) fn chunk_large_response( } else { None }, + command: if is_last { + response.command.clone() + } else { + None + }, watermark_lsn: response.watermark_lsn, error: if is_last { response.error.clone() @@ -114,6 +120,7 @@ mod tests { columns: Some(columns), rows: Some(rows), rows_affected: None, + command: None, watermark_lsn: 42, error: None, auth: None, @@ -146,6 +153,7 @@ mod tests { columns: None, rows: None, rows_affected: Some(5), + command: None, watermark_lsn: 42, error: None, auth: None, @@ -181,6 +189,7 @@ mod tests { columns: Some(columns.clone()), rows: Some(rows), rows_affected: None, + command: None, watermark_lsn: 99, error: None, auth: None, diff --git a/nodedb/src/control/server/native/session/session_stream.rs b/nodedb/src/control/server/native/session/session_stream.rs index a979d21d9..6f484ec0a 100644 --- a/nodedb/src/control/server/native/session/session_stream.rs +++ b/nodedb/src/control/server/native/session/session_stream.rs @@ -136,6 +136,7 @@ pub(super) async fn emit_sql_stream( columns, rows: Some(batch_rows), rows_affected: None, + command: None, watermark_lsn: last_lsn, error: None, auth: None, @@ -152,6 +153,7 @@ pub(super) async fn emit_sql_stream( columns: None, rows: Some(Vec::new()), rows_affected: Some(emitted as u64), + command: None, watermark_lsn: last_lsn, error: None, auth: None, diff --git a/nodedb/src/control/server/native/session/state.rs b/nodedb/src/control/server/native/session/state.rs index b1d8e3057..c1e857ba7 100644 --- a/nodedb/src/control/server/native/session/state.rs +++ b/nodedb/src/control/server/native/session/state.rs @@ -97,7 +97,9 @@ impl NativeSession { global_permit: OwnedSemaphorePermit, resources: NativeConnectionResources, ) -> Self { - let query_ctx = QueryContext::for_state(&state); + // A top-level connection plans with descriptor leases, the array + // catalog and the WAL handle, same as a pgwire connection. + let query_ctx = QueryContext::for_state_with_lease(&state); let NativeConnectionResources { sessions, cleanup } = resources; let transport = stream.transport_security(); Self { diff --git a/nodedb/src/control/server/pgwire/command_tag.rs b/nodedb/src/control/server/pgwire/command_tag.rs index 0696b28eb..e51257b18 100644 --- a/nodedb/src/control/server/pgwire/command_tag.rs +++ b/nodedb/src/control/server/pgwire/command_tag.rs @@ -14,7 +14,9 @@ //! `RESTORE TENANT`, `CREATE COLLECTION`, ...) — those keep whatever shape //! their caller already used; the protocol has no rule to conform to. -use pgwire::api::results::Tag; +use pgwire::api::results::{Response, Tag}; + +use crate::control::server::response_shape::types::{DmlOutcome, FoldedTag}; /// OID reported in the `INSERT ` tag. Real Postgres has emitted /// `0` here since 8.x (the OID-based tag only mattered for `oid`-typed @@ -26,15 +28,36 @@ const INSERT_TAG_OID: u32 = 0; /// it affected. `command` must already be the exact tag text (e.g. /// `"INSERT"`, `"UPDATE"`, or a NodeDB-specific name like `"UPSERT"`). pub(in crate::control::server::pgwire) fn dml_tag(command: &str, rows: usize) -> Tag { + // A count-less verb (SQL `TRUNCATE`) renders bare, per the one rule + // `DmlOutcome::verb_carries_count` owns for every protocol. + if !DmlOutcome::verb_carries_count(command) { + return Tag::new(command); + } match command { "INSERT" => Tag::new(command).with_oid(INSERT_TAG_OID).with_rows(rows), - // Real SQL TRUNCATE: Postgres's tag (`TRUNCATE TABLE`) never carries - // a count — drop it here too rather than inventing one. - "TRUNCATE" => Tag::new(command), _ => Tag::new(command).with_rows(rows), } } +/// Render a folded statement outcome as its `CommandComplete` tag. +pub(in crate::control::server::pgwire) fn render(outcome: DmlOutcome) -> Tag { + dml_tag(outcome.verb, outcome.affected as usize) +} + +/// Push the statement's one folded tag onto `responses`, after its rows: its +/// DML tag when it folded a count-bearing task, `OK` when only opaque tasks +/// folded, nothing when it folded no task at all. +pub(in crate::control::server::pgwire) fn push_folded_tag( + responses: &mut Vec, + tag: Option, +) { + match tag { + Some(FoldedTag::Dml(outcome)) => responses.push(Response::Execution(render(outcome))), + Some(FoldedTag::Opaque) => responses.push(Response::Execution(Tag::new("OK"))), + None => {} + } +} + #[cfg(test)] mod tests { use super::*; @@ -57,6 +80,16 @@ mod tests { assert_eq!(tag.tag, "TRUNCATE"); } + #[test] + fn render_follows_the_outcome_verb() { + let tag: pgwire::messages::response::CommandComplete = render(DmlOutcome { + verb: "INSERT", + affected: 3, + }) + .into(); + assert_eq!(tag.tag, "INSERT 0 3"); + } + #[test] fn nodedb_specific_command_keeps_its_count() { let tag: pgwire::messages::response::CommandComplete = dml_tag("UPSERT", 1).into(); diff --git a/nodedb/src/control/server/pgwire/handler/plan.rs b/nodedb/src/control/server/pgwire/handler/plan.rs index 84a8f34f4..a0275f0cf 100644 --- a/nodedb/src/control/server/pgwire/handler/plan.rs +++ b/nodedb/src/control/server/pgwire/handler/plan.rs @@ -6,137 +6,14 @@ use std::sync::Arc; use futures::stream; use pgwire::api::results::{DataRowEncoder, QueryResponse, Response, Tag}; -use pgwire::error::{ErrorInfo, PgWireError, PgWireResult}; use sonic_rs; -use crate::bridge::envelope::PhysicalPlan; use crate::data::executor::response_codec::decode_payload_to_json; -use nodedb_physical::physical_plan::DocumentOp; -use crate::control::server::shared::sql::staging_predicates::{ - StagedTagKind, require_affected_count, -}; - -use super::super::command_tag::dml_tag; use super::super::types::text_field; pub(super) use crate::control::server::response_shape::types::{PlanKind, describe_plan}; -/// Returns `true` when a plan can produce a deterministic pgwire tag without -/// a round-trip to the Data Plane. -/// -/// Folding is only sound for a write that CANNOT be a no-op — one that either -/// applies exactly one row or fails the statement. Any write whose row count -/// depends on state the plan has not read must get its count from the -/// mutation's own response (`calvin_execution_response` surfaces it from the -/// deposited applied `Response`), because a synthesised count is a claim about -/// rows nobody looked at. -/// -/// **Foldable** — writes that unconditionally apply one row: -/// - `PointPut` (Document) → INSERT 0 1 (upsert: always writes) -/// - `KvOp::Put` → INSERT 0 1 (upsert: always writes) -/// -/// **Not foldable**: -/// - `PointDelete`, `PointUpdate`, `KvOp::Delete` — no-op when the target row -/// is absent, which a resolved primary key does NOT rule out: a surrogate -/// outlives the row it was assigned to, so a delete of an already-deleted -/// key reaches the Data Plane looking exactly like a delete of a live row -/// - `PointInsert`, `KvOp::Insert`, `KvOp::InsertIfAbsent` — an -/// `ON CONFLICT DO NOTHING` insert onto an existing key applies 0 rows -/// - `KvOp::InsertOnConflictUpdate` — outcome (insert vs update) is decided -/// by the handler, not the plan -/// - Any plan with `RETURNING` (response stream carries rows, not a tag) -/// - `InsertSelect` (row count from source query; unknown at plan time) -/// - `BatchInsert`, `BatchPut` (N rows; count in payload) -/// - `BulkUpdate`, `BulkDelete` (predicate-based; count in payload) -/// - `TimeseriesOp::Ingest` (separate path) -/// - `ColumnarOp::Insert` (batch path; count in payload) -/// - Any `Array`, `Spatial`, `Vector`, `Graph`, or `Text` write -/// - Any `SELECT` / `Query` plan (mixing read responses with a write tag -/// corrupts the response stream) -/// - Any other plan not explicitly listed above -pub(super) fn is_calvin_foldable(plan: &PhysicalPlan) -> bool { - use nodedb_physical::physical_plan::KvOp; - - match plan { - // Upserts: the row is written whether or not it existed before, so the - // count is 1 without consulting state. - PhysicalPlan::Document(DocumentOp::PointPut { .. }) - | PhysicalPlan::Kv(KvOp::Put { .. }) => true, - - // Everything else: not foldable. The foldable arms above take - // precedence; these inner wildcards catch every remaining op of each - // engine. Exhaustive so a new PhysicalPlan variant forces a decision. - PhysicalPlan::Document(_) - | PhysicalPlan::Kv(_) - | PhysicalPlan::Vector(_) - | PhysicalPlan::Graph(_) - | PhysicalPlan::Text(_) - | PhysicalPlan::Columnar(_) - | PhysicalPlan::Timeseries(_) - | PhysicalPlan::Spatial(_) - | PhysicalPlan::Crdt(_) - | PhysicalPlan::Query(_) - | PhysicalPlan::Meta(_) - | PhysicalPlan::Array(_) - | PhysicalPlan::ClusterArray(_) - | PhysicalPlan::ClusterEvent(_) => false, - } -} - -/// Render a neutral [`StagedTagKind`] (decided by the protocol-neutral -/// staging gate) as the pgwire `CommandComplete` tag, preserving the exact -/// tag strings the pre-refactor `point_write_tag` / `kv_write_tag` produced: -/// `INSERT 0 n` / `UPDATE n` / `DELETE n`, and for a KV -/// `InsertOnConflictUpdate` outcome, `UPDATE n` when the stage handler -/// resolved to an update or `INSERT 0 n` when it resolved to an insert. -pub(super) fn tag_from_staged(kind: StagedTagKind, affected: usize) -> Tag { - match kind { - StagedTagKind::Insert => dml_tag("INSERT", affected), - StagedTagKind::Update => dml_tag("UPDATE", affected), - StagedTagKind::Delete => dml_tag("DELETE", affected), - StagedTagKind::KvUpsert { updated: true } => dml_tag("UPDATE", affected), - StagedTagKind::KvUpsert { updated: false } => dml_tag("INSERT", affected), - // Matches the autocommit `DocumentOp::Upsert` tag exactly: always the - // literal `UPSERT` command, regardless of insert-vs-update outcome - // (see `response_shape::types::describe_plan`'s `DmlResult("UPSERT")` - // arm and `payload_to_response`'s `PlanKind::DmlResult` rendering). - StagedTagKind::DocUpsert => dml_tag("UPSERT", affected), - // Statement-time in-transaction MERGE: the Postgres command tag for a - // MERGE is `MERGE ` across all arms. - StagedTagKind::Merge => dml_tag("MERGE", affected), - // Statement-time in-transaction `UPDATE ... FROM`: an UPDATE reports the - // Postgres `UPDATE ` command tag over the matched target rows. - StagedTagKind::UpdateFromJoin => dml_tag("UPDATE", affected), - // KV `Incr` / `IncrFloat` / `Cas` / `GetSet` never reach pgwire's - // generic tag-rendering path today: their sole SQL surface (`SELECT - // KV_INCR(..)` and friends, in `ddl/neutral/kv_atomic/`) reads - // `StagedWriteOutcome::payload` directly and never calls - // `tag_from_staged`. This arm exists only so the match stays - // exhaustive against a new `PhysicalPlan::Kv` caller; it renders the - // same tag pgwire uses for a function-call `SELECT`. - StagedTagKind::RawPayload => dml_tag("SELECT", affected), - } -} - -/// Synthesise the pgwire `CommandComplete` tag for a Calvin-foldable plan. -/// -/// Caller invariant: `plan` must already have passed `is_calvin_foldable`. -/// The match arms here are kept in lockstep with that predicate so a desync -/// between the two is loud rather than silent. -pub(super) fn calvin_tag_for_plan(plan: &PhysicalPlan) -> PgWireResult { - use nodedb_physical::physical_plan::KvOp; - - match plan { - PhysicalPlan::Document(DocumentOp::PointPut { .. }) - | PhysicalPlan::Kv(KvOp::Put { .. }) => Ok(dml_tag("INSERT", 1)), - - other => Err(invalid_plan_shape(format!( - "calvin_tag_for_plan called on non-foldable plan: {other:?}" - ))), - } -} - /// Outcome of shaping a Data Plane payload into a pgwire `Response`. /// /// `notice` is set when the response shaper detected a condition the client @@ -156,29 +33,6 @@ impl From for ShapedResponse { } } -pub(super) fn payload_to_response(payload: &[u8], kind: PlanKind) -> PgWireResult { - match kind { - PlanKind::Execution => Ok(Response::Execution(Tag::new("OK")).into()), - PlanKind::DmlResult(tag) => { - // The count comes from the write, always. There is no "point - // operations affected exactly 1 row" shortcut: a point delete or a - // conflicting `ON CONFLICT DO NOTHING` insert is the same plan - // whether it touched a row or not, so assuming 1 here reported rows - // that were never there. - let count = require_affected_count(payload).map_err(|e| { - invalid_plan_shape(format!("{tag} response is missing its affected count: {e}")) - })? as usize; - Ok(Response::Execution(dml_tag(tag, count)).into()) - } - PlanKind::ArraySlice | PlanKind::ReturningRows | PlanKind::SingleDocument => { - Err(invalid_plan_shape(format!( - "payload_to_response cannot handle plan kind {kind:?}" - ))) - } - PlanKind::MultiRow => Ok(multirow_payload_to_response(payload)), - } -} - pub(super) fn multirow_payload_to_response(payload: &[u8]) -> ShapedResponse { let schema = Arc::new(vec![text_field("result")]); if payload.is_empty() { @@ -212,48 +66,29 @@ pub(super) fn multirow_payload_to_response(payload: &[u8]) -> ShapedResponse { Response::Query(QueryResponse::new(schema, stream::iter(vec![Ok(row)]))).into() } -fn invalid_plan_shape(message: String) -> PgWireError { - PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - "XX000".to_owned(), - message, - ))) -} - #[cfg(test)] mod tests { + use super::super::super::command_tag::render; use super::*; + use crate::bridge::envelope::PhysicalPlan; + use crate::control::server::response_shape::types::{ + payload_to_dml_outcome, staged_dml_outcome, + }; + use crate::control::server::shared::sql::staging_predicates::StagedTagKind; use nodedb_physical::physical_plan::KvOp; use nodedb_types::{DatabaseId, QualifiedCollection}; - #[test] - fn calvin_tag_rejects_non_foldable_plan() { - let plan = PhysicalPlan::Kv(KvOp::Get { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "items"), - key: Vec::new(), - rls_filters: Vec::new(), - surrogate_ceiling: None, - }); - assert!(calvin_tag_for_plan(&plan).is_err()); - } - - #[test] - fn passthrough_rejects_precomposed_shapes() { - assert!(payload_to_response(&[], PlanKind::ArraySlice).is_err()); - assert!(payload_to_response(&[], PlanKind::ReturningRows).is_err()); - assert!(payload_to_response(&[], PlanKind::SingleDocument).is_err()); - } - #[test] fn multirow_helper_remains_infallible() { let shaped = multirow_payload_to_response(&[]); assert!(matches!(shaped.response, Response::Query(_))); } + /// The folded Calvin upsert tag renders as the SQL `UPSERT` command. #[test] fn foldable_tag_still_matches_operation() { - // An upsert applies one row unconditionally, so its tag needs no - // round-trip. + use crate::control::server::response_shape::calvin_fold::calvin_tag_for_plan; + let plan = PhysicalPlan::Kv(KvOp::Put { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "items"), key: Vec::new(), @@ -263,45 +98,32 @@ mod tests { returning: None, rls_filters: Vec::new(), }); - assert!(is_calvin_foldable(&plan)); - assert!(calvin_tag_for_plan(&plan).is_ok()); + let outcome = calvin_tag_for_plan(&plan).expect("an upsert folds without a round-trip"); + let tag: pgwire::messages::response::CommandComplete = render(outcome).into(); + assert_eq!(tag.tag, "UPSERT 1"); } - /// A write that can legitimately touch nothing must NOT be folded: its count - /// is only knowable from the mutation's own response. Folding a delete let a - /// re-delete of an already-deleted key report a removed row. + /// A staged `TRUNCATE` renders the same bare tag autocommit does. #[test] - fn no_op_capable_writes_are_never_folded() { - let delete = PhysicalPlan::Kv(KvOp::Delete { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "items"), - keys: Vec::new(), - rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), - returning: None, - rls_filters: Vec::new(), - }); - assert!(!is_calvin_foldable(&delete)); - assert!(calvin_tag_for_plan(&delete).is_err()); - - let point_delete = PhysicalPlan::Document(DocumentOp::PointDelete { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "items"), - document_id: "a".into(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - returning: None, - rls_filters: Vec::new(), - rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), - resolved_sum_targets: Vec::new(), - }); - assert!(!is_calvin_foldable(&point_delete)); - assert!(calvin_tag_for_plan(&point_delete).is_err()); + fn staged_truncate_renders_a_bare_tag() { + let outcome = staged_dml_outcome(StagedTagKind::Truncate, 0); + let tag: pgwire::messages::response::CommandComplete = + crate::control::server::pgwire::command_tag::render(outcome).into(); + assert_eq!(tag.tag, "TRUNCATE"); } - /// A count-bearing response with no count is a handler bug, not a `1`. + /// `DmlResultByOp` renders the verb the handler reported. #[test] - fn dml_tag_requires_a_reported_count() { - assert!(payload_to_response(&[], PlanKind::DmlResult("DELETE")).is_err()); - let payload = nodedb_types::json_to_msgpack(&serde_json::json!({ "affected": 0 })) - .expect("encode count payload"); - assert!(payload_to_response(&payload, PlanKind::DmlResult("DELETE")).is_ok()); + fn dml_result_by_op_renders_the_reported_verb() { + let update = nodedb_types::json_to_msgpack(&serde_json::json!({ + "affected": 1, + "op": "update" + })) + .expect("encode payload"); + let outcome = payload_to_dml_outcome(&update, PlanKind::DmlResultByOp) + .expect("update tag") + .expect("count-bearing"); + let tag: pgwire::messages::response::CommandComplete = render(outcome).into(); + assert_eq!(tag.tag, "UPDATE 1"); } } diff --git a/nodedb/src/control/server/pgwire/handler/routing/calvin_dispatch.rs b/nodedb/src/control/server/pgwire/handler/routing/calvin_dispatch.rs index a9208b00e..9c694b685 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/calvin_dispatch.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/calvin_dispatch.rs @@ -15,26 +15,26 @@ use crate::control::planner::calvin::{ }; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::security::request_scope::RequestAuthScope; -use crate::control::server::response_shape::types::ShapedRows; +use crate::control::server::response_shape::calvin_fold::{ + CalvinFoldCtx, CalvinFoldError, fold_calvin_batch, +}; use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; use crate::control::server::shared::session::{SessionId, TransactionState}; use crate::types::{DatabaseId, TenantId}; use nodedb_physical::physical_task::PhysicalTask; -use super::super::super::types::error_to_sqlstate; +use super::super::super::command_tag::push_folded_tag; +use super::super::super::types::{dml_fold_error_to_pg, error_to_pg, error_to_sqlstate}; use super::super::core::NodeDbPgHandler; -use super::calvin_response::{CalvinResponseCtx, CalvinTaskOutcome, calvin_execution_response}; -/// Meter one Calvin task's shaped response, once its response has already -/// been synthesised successfully by `calvin_execution_response` — Calvin -/// applies the whole batch atomically, so by the time responses are being -/// shaped every task in `tasks` has already committed. +/// Meter one Calvin task once the batch has committed — Calvin applies the +/// whole batch atomically, so by the time responses are being shaped every +/// task in `tasks` has already committed. /// -/// `rows: None` — `calvin_execution_response` yields either an `Execution` tag -/// or the task's `ShapedRows`, which the caller folds into the statement's -/// single result set; counting rows here would mean reaching into that fold -/// before it is complete. `meter_dispatch` charges one unit for `None`, correct -/// for the write that just committed. +/// `rows: None` — a task yields either a tag contribution or its rows, which +/// the fold takes once for the statement's single result set; counting rows +/// here would mean reaching into that fold. `meter_dispatch` charges one unit +/// for `None`, correct for the write that just committed. fn meter_calvin_task( state: &crate::control::state::SharedState, identity: &AuthenticatedIdentity, @@ -68,9 +68,9 @@ pub(super) struct CalvinDispatchSession<'a> { impl NodeDbPgHandler { /// Drive Calvin strict multi-shard dispatch for the given task set. /// - /// Returns the response vec on success (one tag per task). The caller - /// should return this immediately — Calvin tasks do not go through the - /// per-task dispatch loop. + /// Returns the response vec on success: the statement's RETURNING rows + /// (if any) and its one command tag. The caller returns this immediately + /// — Calvin tasks do not go through the per-task dispatch loop. pub(super) async fn dispatch_calvin_multishard( &self, tasks: Vec, @@ -137,9 +137,8 @@ impl NodeDbPgHandler { // inputs (cross-shard mode, in-block state) it needs as parameters. // The helper classifies, rejects cross-shard writes inside an // explicit transaction block, builds the static TxClass, and routes - // the SINGLE submit-and-await to the sequencer leader. On success we - // synthesise one command tag per task. This is a pure extraction — - // behaviour is identical to the inlined static branch. + // the SINGLE submit-and-await to the sequencer leader. On success + // every task's outcome folds into the statement's one response. // A cross-shard span in a single statement executed mid-block cannot // be buffered atomically, so it rejects; an autocommit statement // proceeds. (The COMMIT flush of a buffered block routes through the @@ -169,50 +168,27 @@ impl NodeDbPgHandler { ))) })?; - let mut calvin_responses: Vec = Vec::with_capacity(tasks.len()); - // A statement is ONE result set. Calvin deposits ONE applied - // response for the whole transaction (a second RETURNING-bearing - // participant is recorded as a conflict and fails the statement - // upstream), and every task below is shaped from that same payload - // — so the rows are taken once rather than accumulated, which would - // repeat the identical payload per task. This is the shape the - // native Calvin path already uses. - let mut returning_rows: Option = None; - for task in &tasks { - match calvin_execution_response( - task, - apply_resp.as_ref(), - CalvinResponseCtx { - projection, - state: &self.state, - tenant_id, - database_id, - auth, - }, - )? { - CalvinTaskOutcome::Rows(shaped) => { - returning_rows.get_or_insert(shaped); - } - CalvinTaskOutcome::Tag(response) => calvin_responses.push(response), - } - meter_calvin_task(&self.state, identity, database_id, task); - } - if let Some(shaped) = returning_rows { - let (response, _notice) = - super::super::shape_encode::shaped_query_response(shaped, result_formats); - calvin_responses.push(response); - } - return Ok(calvin_responses); + return self.shape_calvin_batch( + &tasks, + apply_resp.as_ref(), + CalvinBatchShaping { + identity, + result_formats, + auth, + projection, + tenant_id, + database_id, + }, + ); } // OLLP path: delegate the full reconnaissance + atomic-submit + drift- // retry orchestration to the protocol-neutral // `dispatch_dependent_edge_recon`. The dependent task is guaranteed // present (the static path returned early above); its `database_id` is - // the recon scan's database. On `Ok` we synthesise the SAME response — - // one CommandComplete tag per accumulated task — and on `Err` we map the - // typed `crate::Error` through the existing pgwire error→SQLSTATE path, - // so externally observable behaviour is byte-identical. + // the recon scan's database. On `Ok` the batch shapes the SAME way as + // the static path; on `Err` the typed `crate::Error` maps through the + // pgwire error→SQLSTATE path. let database_id = dependent_task .ok_or_else(|| { // Unreachable: the static (non-dependent) path returns early @@ -254,34 +230,75 @@ impl NodeDbPgHandler { ))) })?; - let mut calvin_responses: Vec = Vec::with_capacity(tasks.len()); - // One result set per statement, taken once from the batch's single - // applied response — see the static path above. - let mut returning_rows: Option = None; - for task in &tasks { - match calvin_execution_response( - task, - outcome.apply_result.as_ref(), - CalvinResponseCtx { - projection, - state: &self.state, - tenant_id, - database_id, - auth, - }, - )? { - CalvinTaskOutcome::Rows(shaped) => { - returning_rows.get_or_insert(shaped); - } - CalvinTaskOutcome::Tag(response) => calvin_responses.push(response), - } + self.shape_calvin_batch( + &tasks, + outcome.apply_result.as_ref(), + CalvinBatchShaping { + identity, + result_formats, + auth, + projection, + tenant_id, + database_id, + }, + ) + } + + /// Shape a completed Calvin batch into the statement's responses: its + /// RETURNING rows, when a task carried them, then its one command tag. + /// The fold itself is protocol-neutral (`response_shape::calvin_fold`), so + /// native answers the same rows and count for the same batch. + fn shape_calvin_batch( + &self, + tasks: &[PhysicalTask], + apply_resp: Option<&crate::bridge::envelope::Response>, + shaping: CalvinBatchShaping<'_>, + ) -> PgWireResult> { + let CalvinBatchShaping { + identity, + result_formats, + auth, + projection, + tenant_id, + database_id, + } = shaping; + let plans: Vec<&crate::bridge::envelope::PhysicalPlan> = + tasks.iter().map(|task| &task.plan).collect(); + let fold = fold_calvin_batch( + &plans, + apply_resp, + &CalvinFoldCtx { + projection, + state: &self.state, + tenant_id, + database_id, + auth, + }, + ) + .map_err(|e| match e { + CalvinFoldError::Verb(e) => dml_fold_error_to_pg(&e), + CalvinFoldError::Shape(e) => error_to_pg(&e), + })?; + for task in tasks { meter_calvin_task(&self.state, identity, database_id, task); } - if let Some(shaped) = returning_rows { + let mut responses: Vec = Vec::with_capacity(2); + if let Some(shaped) = fold.rows { let (response, _notice) = super::super::shape_encode::shaped_query_response(shaped, result_formats); - calvin_responses.push(response); + responses.push(response); } - Ok(calvin_responses) + push_folded_tag(&mut responses, fold.tag); + Ok(responses) } } + +/// How a completed Calvin batch is shaped back to the client. +struct CalvinBatchShaping<'a> { + identity: &'a AuthenticatedIdentity, + result_formats: &'a [pgwire::api::results::FieldFormat], + auth: &'a crate::control::security::auth_context::AuthContext, + projection: Option<&'a crate::control::server::response_shape::schema::OutputSchema>, + tenant_id: TenantId, + database_id: DatabaseId, +} diff --git a/nodedb/src/control/server/pgwire/handler/routing/calvin_response.rs b/nodedb/src/control/server/pgwire/handler/routing/calvin_response.rs deleted file mode 100644 index bbb3fb614..000000000 --- a/nodedb/src/control/server/pgwire/handler/routing/calvin_response.rs +++ /dev/null @@ -1,128 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Response shaping for a completed Calvin batch. -//! -//! Calvin applies a whole batch atomically and deposits ONE applied response -//! for the transaction, so turning that batch into client-visible responses is -//! a different concern from planning a single statement into tasks — this file -//! owns the former, `planning.rs` the latter. - -use pgwire::api::results::Tag; -use pgwire::error::{ErrorInfo, PgWireError}; - -use crate::control::server::response_shape::types::ShapedRows; -use crate::types::TenantId; -use nodedb_physical::physical_task::PhysicalTask; - -/// Shared inputs for shaping one task of a completed Calvin batch. -pub(super) struct CalvinResponseCtx<'a> { - /// The statement's announced output columns, when it announced any. A - /// `RETURNING` write is held to them here exactly as the single-shard - /// dispatch loop holds it, so the same statement renders the same row - /// whichever route it took. - pub(super) projection: Option<&'a crate::control::server::response_shape::schema::OutputSchema>, - pub(super) state: &'a crate::control::state::SharedState, - pub(super) tenant_id: TenantId, - pub(super) database_id: crate::types::DatabaseId, - /// The requester's resolved context; its roles drive column-level - /// redaction of any RETURNING rows this batch surfaces. - pub(super) auth: &'a crate::control::security::auth_context::AuthContext, -} - -/// One Calvin task's contribution to the statement's response. -pub(super) enum CalvinTaskOutcome { - /// RETURNING rows, to be folded into the statement's single result set. - Rows(ShapedRows), - /// A command tag, emitted as its own response. - Tag(pgwire::api::results::Response), -} - -/// Build the pgwire outcome for one task of a completed Calvin batch. -/// -/// A task whose plan carries a RETURNING clause yields its rows as protocol- -/// neutral [`ShapedRows`] rather than an encoded response — the caller folds -/// every such task's rows into ONE result set for the statement, because a -/// multi-row write plans one task per row and an extended-query client reads a -/// RowDescription/DataRow sequence per task as several results for one -/// statement. Every other task (and a RETURNING task with no carried payload) -/// keeps the synthesised `Response::Execution` command tag. -pub(super) fn calvin_execution_response( - task: &PhysicalTask, - apply_resp: Option<&crate::bridge::envelope::Response>, - ctx: CalvinResponseCtx<'_>, -) -> pgwire::error::PgWireResult { - use super::super::plan::{calvin_tag_for_plan, is_calvin_foldable}; - use crate::control::server::response_shape::compose::{ - ShapeOutcome, shape_response_materialized, - }; - use crate::control::server::response_shape::redaction::QueryRedaction; - use crate::control::server::response_shape::request::MaterializedShapeRequest; - use crate::control::server::response_shape::types::{PlanKind, describe_plan}; - - let CalvinResponseCtx { - projection, - state, - tenant_id, - database_id, - auth, - } = ctx; - - // RETURNING path: shape the applied payload into DATA-ROWs, exactly as the - // non-Calvin dispatch loop does for a RETURNING write. - let redaction = QueryRedaction::for_plan(tenant_id, auth, &task.plan); - if let (PlanKind::ReturningRows, Some(resp)) = (describe_plan(&task.plan), apply_resp) - && let Ok(ShapeOutcome::Rows(shaped)) = - shape_response_materialized(MaterializedShapeRequest { - payload: resp.payload.as_bytes(), - plan: &task.plan, - plan_kind: PlanKind::ReturningRows, - projection, - state, - database_id, - tenant_id, - redaction: Some(redaction.ctx(&state.redaction)), - // A RETURNING list names stored columns only, never a - // Control-Plane computed column. - sequences: None, - }) - { - return Ok(CalvinTaskOutcome::Rows(shaped)); - } - - // Plain (non-RETURNING) write: surface its ACTUAL affected count from the - // payload — exactly as the non-Calvin write path does. - // - // Every primary-write participant deposits its applied `Response` before - // proposing the completion ack (cross-node it rides back on the routed - // submit's RPC reply), so a count-bearing plan ALWAYS has one here. If it - // does not, the deposit path regressed: fail loudly rather than synthesise a - // count, which is what made a delete of an absent row report a removed row. - if let PlanKind::DmlResult(tag) = describe_plan(&task.plan) { - let resp = apply_resp.ok_or_else(|| { - PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - "XX000".to_owned(), - format!( - "internal: Calvin {tag} completed with no applied response to read its \ - affected-row count from" - ), - ))) - })?; - return Ok(CalvinTaskOutcome::Tag( - super::super::plan::payload_to_response( - resp.payload.as_bytes(), - describe_plan(&task.plan), - )? - .response, - )); - } - - let tag = if is_calvin_foldable(&task.plan) { - calvin_tag_for_plan(&task.plan)? - } else { - Tag::new("OK") - }; - Ok(CalvinTaskOutcome::Tag( - pgwire::api::results::Response::Execution(tag), - )) -} diff --git a/nodedb/src/control/server/pgwire/handler/routing/cluster_array.rs b/nodedb/src/control/server/pgwire/handler/routing/cluster_array.rs index b77b9a375..63a402351 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/cluster_array.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/cluster_array.rs @@ -2,39 +2,39 @@ //! ClusterArray plan dispatch for the pgwire handler. //! -//! ClusterArray plans are handled entirely on the Control Plane by the -//! `ArrayCoordinator` — they must never reach the SPSC bridge or the -//! trigger/DML machinery. `dispatch_task_loop` intercepts them and delegates -//! to the helper here, which shapes the coordinator's payload into a single -//! pgwire `Response` (surfacing any client-facing notice via the session). +//! `dispatch_task_loop` intercepts a `PhysicalPlan::ClusterArray` task and +//! delegates to the shared, protocol-neutral core +//! (`shared::cluster_array_dispatch::execute_cluster_array`), then encodes +//! the outcome as one pgwire `Response` (surfacing any client-facing notice +//! via the session) or a `DmlOutcome` the caller folds into the statement tag. use pgwire::api::results::{FieldFormat, Response}; use pgwire::error::{ErrorInfo, PgWireError, PgWireResult}; -use nodedb_physical::physical_plan::{ClusterArrayOp, PhysicalPlan}; - -use crate::control::server::dispatch_utils::publish_cluster_array_change_events; -use crate::control::server::response_shape::compose::{self, ShapeOutcome}; -use crate::control::server::response_shape::redaction::QueryRedaction; use crate::control::server::response_shape::schema::OutputSchema; +use crate::control::server::response_shape::types::DmlOutcome; +use crate::control::server::shared::cluster_array_dispatch::{ + ClusterArrayShaped, execute_cluster_array, +}; use crate::control::server::shared::session::SessionId; -use super::super::super::types::{error_to_sqlstate, shape_error_to_pg}; +use super::super::super::types::error_to_sqlstate; use super::super::core::NodeDbPgHandler; -use super::super::plan::{PlanKind, payload_to_response}; use super::super::shape_encode; +/// What one `ClusterArrayOp` answers with. +pub(super) enum ClusterArrayResult { + /// A read's rows (`Slice` / `Agg`), encoded. Caller pushes the response. + Rows(Response), + /// A write's count (`Put` / `Delete`). Caller folds it into the + /// statement tag. + Dml(DmlOutcome), +} + impl NodeDbPgHandler { - /// Execute a single `ClusterArrayOp` via the `ArrayCoordinator` and shape - /// its payload into one pgwire `Response`. Any carried notice is pushed to - /// the supplied session. - /// - /// On a successful `Put`/`Delete` (writes; `Slice`/`Agg` are reads and - /// publish nothing), publishes a CDC change event keyed by the op's own - /// `wal_lsn` — this path never touches the SPSC bridge, so there is no - /// Data-Plane `Response::watermark_lsn` to read the LSN from the way the - /// normal dispatch funnel does (see `publish_cluster_array_change_events`'s - /// own doc comment). + /// Execute a single `ClusterArrayOp` via the shared core and encode its + /// outcome into one pgwire `Response` or one count-bearing outcome. Any + /// carried notice is pushed to the supplied session. pub(super) async fn dispatch_cluster_array_task( &self, authorized: crate::control::server::shared::authorization::AuthorizedTask, @@ -42,112 +42,26 @@ impl NodeDbPgHandler { result_formats: &[FieldFormat], session_id: SessionId, auth: &crate::control::security::auth_context::AuthContext, - ) -> PgWireResult { - use crate::control::cluster::ClusterArrayExecutor; - use std::sync::Arc; - - let task = authorized.into_physical_task(); - let tenant_id = task.tenant_id; - let database_id = task.database_id; - let PhysicalPlan::ClusterArray(cluster_op) = task.plan else { - return Err(PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - "XX000".to_owned(), - "authorized task is not a ClusterArray operation".to_owned(), - )))); - }; - - let transport = self.state.cluster_transport.as_ref().ok_or_else(|| { - PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - "XX000".to_owned(), - "cluster transport not available for ClusterArray dispatch".to_owned(), - ))) - })?; - let routing = self.state.cluster_routing.as_ref().ok_or_else(|| { - PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - "XX000".to_owned(), - "cluster routing not available for ClusterArray dispatch".to_owned(), - ))) - })?; - let executor = ClusterArrayExecutor::new( - Arc::clone(transport), - Arc::clone(routing), - self.state.node_id, - Arc::clone(&self.state), - ); - let payload_bytes = executor.execute(&cluster_op).await.map_err(|e| { - let (severity, code, message) = error_to_sqlstate(&e); - PgWireError::UserError(Box::new(ErrorInfo::new( - severity.to_owned(), - code.to_owned(), - message, - ))) - })?; - // Publish CDC change event(s) for a successful write. `Slice`/`Agg` - // are reads and publish nothing; `Put`/`Delete` carry their own - // Control-Plane-allocated `wal_lsn` since there is no Data-Plane - // `Response::watermark_lsn` on this coordinator-only path. - let write_lsn = match &cluster_op { - ClusterArrayOp::Put { wal_lsn, .. } | ClusterArrayOp::Delete { wal_lsn, .. } => { - Some(*wal_lsn) - } - ClusterArrayOp::Slice { .. } | ClusterArrayOp::Agg { .. } => None, - }; - if let Some(lsn) = write_lsn { - publish_cluster_array_change_events( - &self.state, - tenant_id, - database_id, - &cluster_op, - lsn, - ); - } - - let cluster_plan_kind = match &cluster_op { - ClusterArrayOp::Slice { .. } => PlanKind::ArraySlice, - ClusterArrayOp::Agg { .. } - | ClusterArrayOp::Put { .. } - | ClusterArrayOp::Delete { .. } => PlanKind::MultiRow, - }; - // This coordinator path never builds a `PhysicalPlan`, so the source - // collection comes straight off the op's array name. A single source - // means bare-key matching, which is what an array's cell rows carry. - let array_name = match &cluster_op { - ClusterArrayOp::Slice { array_id, .. } - | ClusterArrayOp::Agg { array_id, .. } - | ClusterArrayOp::Put { array_id, .. } - | ClusterArrayOp::Delete { array_id, .. } => array_id.name.clone(), - }; - let redaction = - QueryRedaction::for_collections(tenant_id, auth, vec![(String::new(), array_name)]); - // A cluster array plan projects attribute names only, never a - // Control-Plane computed column, so no session sequence access. - match compose::shape_payload_no_plan( - &payload_bytes, - cluster_plan_kind, - projection, - Some(redaction.ctx(&self.state.redaction)), - None, - ) - .map_err(|e| shape_error_to_pg(&e))? - { - ShapeOutcome::Rows(shaped) => { + ) -> PgWireResult { + match execute_cluster_array(&self.state, auth, authorized, projection) + .await + .map_err(|e| { + let (severity, code, message) = error_to_sqlstate(&e); + PgWireError::UserError(Box::new(ErrorInfo::new( + severity.to_owned(), + code.to_owned(), + message, + ))) + })? { + ClusterArrayShaped::Rows(shaped) => { let (response, notice) = shape_encode::shaped_query_response(shaped, result_formats); if let Some(n) = notice { self.sessions.push_notice(session_id, n); } - Ok(response) - } - ShapeOutcome::Passthrough => { - let shaped = payload_to_response(&payload_bytes, cluster_plan_kind)?; - if let Some(notice) = shaped.notice { - self.sessions.push_notice(session_id, notice); - } - Ok(shaped.response) + Ok(ClusterArrayResult::Rows(response)) } + ClusterArrayShaped::Affected(outcome) => Ok(ClusterArrayResult::Dml(outcome)), } } } diff --git a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/finish.rs b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/finish.rs index f8353a117..6d39c5a80 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/finish.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/finish.rs @@ -1,7 +1,8 @@ // SPDX-License-Identifier: BUSL-1.1 //! The statement's tail after every task ran: the folded `RETURNING` rows as -//! one result set, then the set-operation merge of the deferred payloads. +//! one result set, then the set-operation merge of the deferred payloads, +//! then the statement's one command tag. use std::sync::Arc; @@ -13,10 +14,11 @@ use nodedb_physical::physical_task::PostSetOp; use crate::control::sequence::{SequenceAccess, SessionSequenceAccess, SessionSequenceValues}; use crate::control::server::response_shape::redaction::QueryRedaction; use crate::control::server::response_shape::schema::OutputSchema; -use crate::control::server::response_shape::types::ShapedRows; +use crate::control::server::response_shape::types::{ShapedRows, StatementTag}; use crate::control::server::shared::session::SessionId; use crate::types::{DatabaseId, TenantId}; +use super::super::super::super::command_tag::push_folded_tag; use super::super::super::core::NodeDbPgHandler; use super::super::super::shape_encode; use super::super::set_ops; @@ -25,6 +27,8 @@ use super::super::set_ops; pub(super) struct StatementTail<'a> { /// The statement's `RETURNING` rows, folded across every task. pub(super) returning_rows: Option, + /// The statement's command tag, folded across every write task. + pub(super) statement_tag: StatementTag, /// Per-branch payloads deferred for a set-operation merge. pub(super) dedup_payloads: Vec>, pub(super) dedup_set_op: PostSetOp, @@ -52,6 +56,7 @@ impl NodeDbPgHandler { ) -> PgWireResult<()> { let StatementTail { returning_rows, + statement_tag, dedup_payloads, dedup_set_op, projection, @@ -98,6 +103,11 @@ impl NodeDbPgHandler { responses.push(response); } + // The statement's one command tag, after its rows. A statement never + // carries both RETURNING rows and a count-bearing tag: `RETURNING` + // classifies as `ReturningRows`, which folds rows and no count. + push_folded_tag(responses, statement_tag.finish()); + Ok(()) } } diff --git a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs index 5f46b7ef5..29246e846 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs @@ -3,8 +3,8 @@ //! The per-task dispatch loop for non-Calvin pgwire queries: tenant check, //! in-transaction routing, streaming fast path, pre-dispatch hooks, dispatch, //! read tracking, AFTER triggers, and metering. Shaping one task's response -//! lives in `task.rs`; the statement's tail (folded RETURNING rows and the -//! set-op merge) lives in `finish.rs`. +//! lives in `task.rs`; the statement's tail (folded RETURNING rows, the +//! set-op merge, and the one folded command tag) lives in `finish.rs`. //! //! Split out of `execute.rs`, which keeps the plan/authorize/admit entry //! points and hands the admitted task list here. @@ -14,10 +14,13 @@ use std::sync::Arc; use pgwire::api::results::Response; use pgwire::error::{ErrorInfo, PgWireError, PgWireResult}; +use crate::control::planner::calvin::write_class::{ + plan_counts_toward_statement_tag, plans_have_user_write, +}; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::security::request_scope::RequestAuthScope; use crate::control::server::response_shape::redaction::QueryRedaction; -use crate::control::server::response_shape::types::ShapedRows; +use crate::control::server::response_shape::types::{ShapedRows, StatementTag}; use crate::control::server::shared::ddl::neutral::maintenance::auto_analyze; use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; use crate::control::server::shared::quota_admission::admit_quota_for_dispatch; @@ -26,11 +29,12 @@ use crate::types::TenantId; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; use super::super::super::super::types::{ - error_to_sqlstate, response_status_to_sqlstate, sqlstate_error, + dml_fold_error_to_pg, error_to_sqlstate, response_status_to_sqlstate, sqlstate_error, }; use super::super::super::core::NodeDbPgHandler; use super::super::super::plan::{PlanKind, describe_plan}; -use super::super::execute_dml_hooks; +use super::super::cluster_array::ClusterArrayResult; +use super::super::execute_dml_hooks::{self, PreDispatchHandled}; use super::super::result_shaping::ResultShaping; use super::super::streaming::StreamSelectContext; use super::finish::StatementTail; @@ -76,6 +80,10 @@ impl NodeDbPgHandler { // loop rather than as a RowDescription/DataRow sequence per task, which // an extended-query client reads as several results for one statement. let mut returning_rows: Option = None; + // The statement's ONE command tag, folded over its write tasks the + // same way. `execute_sql` calls this loop once per `;`-separated + // statement, so the fold never crosses a statement boundary. + let mut statement_tag = StatementTag::default(); let mut responses = Vec::with_capacity(tasks.len()); // Session-scoped sequence access for the statement's Control-Plane // computed columns, resolved once: every task of one statement @@ -87,6 +95,9 @@ impl NodeDbPgHandler { // the collection name) a true no-op on the hot path for every // deployment that hasn't turned it on. let metering_enabled = self.state.metering_config.enabled; + // A derived implicit-edge write beside the user's own never answers + // the statement, exactly as Calvin's deposit rule has it. + let has_user_write = plans_have_user_write(tasks.iter().map(|t| &t.plan)); for mut task in tasks { if task.tenant_id != tenant_id { @@ -120,7 +131,7 @@ impl NodeDbPgHandler { execute_dml_hooks::TxnRouteOutcome::Proceed(routed_task) => { task = *routed_task; } - execute_dml_hooks::TxnRouteOutcome::Handled(resp) => { + execute_dml_hooks::TxnRouteOutcome::Handled(handled) => { if returns_rows { let (severity, code, message) = error_to_sqlstate( &crate::control::server::shared::returning:: @@ -132,7 +143,7 @@ impl NodeDbPgHandler { message, )))); } - responses.push(resp); + handled.fold_into(&mut statement_tag)?; continue; } } @@ -140,8 +151,10 @@ impl NodeDbPgHandler { // ClusterArray plans are handled entirely on the Control Plane by // the ArrayCoordinator — they must never reach the SPSC bridge or // trigger/DML machinery. Intercepted AFTER the routing gate, so an - // in-transaction write is buffered (reshaped into per-shard - // `ArrayOp` plans) instead of applying here and surviving ROLLBACK. + // in-transaction write is staged per shard and buffered (reshaped + // into per-shard `ArrayOp` plans) instead of applying here and + // surviving ROLLBACK, and an in-transaction read carries the + // transaction id the gate stamped for read-your-own-writes. if matches!( task.plan, nodedb_physical::physical_plan::PhysicalPlan::ClusterArray(_) @@ -158,7 +171,7 @@ impl NodeDbPgHandler { "ClusterArray authorization returned no capability".to_owned(), ))) })?; - let response = self + match self .dispatch_cluster_array_task( authorized, projection, @@ -166,12 +179,18 @@ impl NodeDbPgHandler { session_id, auth_ctx, ) - .await?; - responses.push(response); + .await? + { + ClusterArrayResult::Rows(response) => responses.push(response), + ClusterArrayResult::Dml(outcome) => statement_tag + .fold(outcome) + .map_err(|e| dml_fold_error_to_pg(&e))?, + } continue; } let plan_kind = describe_plan(&task.plan); + let counts_toward_tag = plan_counts_toward_statement_tag(&task.plan, has_user_write); let resp_post_set_op = task.post_set_op; let task_database_id = task.database_id; let task_vshard = task.vshard_id; @@ -244,8 +263,16 @@ impl NodeDbPgHandler { ) .await? { - execute_dml_hooks::PreDispatchOutcome::Handled(resp) => { - responses.push(resp); + execute_dml_hooks::PreDispatchOutcome::Handled(PreDispatchHandled::Rows( + response, + )) => { + responses.push(response); + continue; + } + execute_dml_hooks::PreDispatchOutcome::Handled(PreDispatchHandled::Write( + handled, + )) => { + handled.fold_into(&mut statement_tag)?; continue; } execute_dml_hooks::PreDispatchOutcome::Proceed(proceed) => { @@ -406,6 +433,7 @@ impl NodeDbPgHandler { response: &resp, plan: &plan_for_response, plan_kind, + counts_toward_tag, projection, result_formats, session_id, @@ -416,6 +444,7 @@ impl NodeDbPgHandler { }, &mut responses, &mut returning_rows, + &mut statement_tag, )? }; @@ -434,6 +463,7 @@ impl NodeDbPgHandler { &mut responses, StatementTail { returning_rows, + statement_tag, dedup_payloads, dedup_set_op, projection, diff --git a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/task.rs b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/task.rs index f639afa0d..b71585c16 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/task.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/task.rs @@ -13,13 +13,15 @@ use crate::control::server::response_shape::compose::{self, ShapeOutcome}; use crate::control::server::response_shape::redaction::QueryRedaction; use crate::control::server::response_shape::request::MaterializedShapeRequest; use crate::control::server::response_shape::schema::OutputSchema; -use crate::control::server::response_shape::types::ShapedRows; +use crate::control::server::response_shape::types::{ + ShapedRows, StatementTag, payload_to_dml_outcome, +}; use crate::control::server::shared::session::SessionId; use crate::types::{DatabaseId, TenantId}; -use super::super::super::super::types::shape_error_to_pg; +use super::super::super::super::types::{dml_fold_error_to_pg, error_to_pg, shape_error_to_pg}; use super::super::super::core::NodeDbPgHandler; -use super::super::super::plan::{PlanKind, payload_to_response}; +use super::super::super::plan::PlanKind; use super::super::super::shape_encode; /// Everything needed to shape one task's response. @@ -27,6 +29,9 @@ pub(super) struct ShapeTaskParams<'a> { pub(super) response: &'a crate::bridge::envelope::Response, pub(super) plan: &'a PhysicalPlan, pub(super) plan_kind: PlanKind, + /// Whether this task's count answers the statement. False for a derived + /// implicit-edge write beside the user's own, which folds as opaque. + pub(super) counts_toward_tag: bool, pub(super) projection: Option<&'a OutputSchema>, pub(super) result_formats: &'a [FieldFormat], pub(super) session_id: SessionId, @@ -41,7 +46,9 @@ pub(super) struct ShapeTaskParams<'a> { impl NodeDbPgHandler { /// Shape one task's response: rows are encoded and pushed onto /// `responses`, except `RETURNING` rows, which fold into - /// `returning_rows` so the statement answers with one result set. + /// `returning_rows` so the statement answers with one result set. A + /// passthrough response folds into `statement_tag` so the statement + /// answers with one command tag. /// /// Returns the task's own row count for metering, `None` for a /// passthrough response with no row payload to count. @@ -50,11 +57,13 @@ impl NodeDbPgHandler { params: ShapeTaskParams<'_>, responses: &mut Vec, returning_rows: &mut Option, + statement_tag: &mut StatementTag, ) -> PgWireResult> { let ShapeTaskParams { response, plan, plan_kind, + counts_toward_tag, projection, result_formats, session_id, @@ -103,11 +112,18 @@ impl NodeDbPgHandler { Ok(Some(task_rows)) } ShapeOutcome::Passthrough => { - let shaped = payload_to_response(&response.payload, plan_kind)?; - if let Some(notice) = shaped.notice { - self.sessions.push_notice(session_id, notice); + if !counts_toward_tag { + statement_tag.fold_opaque(); + return Ok(None); + } + match payload_to_dml_outcome(&response.payload, plan_kind) + .map_err(|e| error_to_pg(&e))? + { + Some(outcome) => statement_tag + .fold(outcome) + .map_err(|e| dml_fold_error_to_pg(&e))?, + None => statement_tag.fold_opaque(), } - responses.push(shaped.response); Ok(None) } } diff --git a/nodedb/src/control/server/pgwire/handler/routing/execute.rs b/nodedb/src/control/server/pgwire/handler/routing/execute.rs index f829e435a..6070bce06 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/execute.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/execute.rs @@ -13,7 +13,7 @@ use pgwire::error::{ErrorInfo, PgWireError, PgWireResult}; use crate::control::planner::calvin::{DispatchClass, classify_dispatch}; use crate::control::security::identity::AuthenticatedIdentity; -use crate::control::server::shared::session::SessionId; +use crate::control::server::shared::session::{SessionId, TransactionState}; use crate::control::server::shared::write_admission::all_writes_bufferable; use crate::types::TenantId; @@ -179,24 +179,27 @@ impl NodeDbPgHandler { return Ok(responses); } - if let Some(responses) = self - .maybe_dispatch_tasks_via_gateway( - &tasks, - identity, - tenant_id, - session_id, - ResultShaping { - projection: effective_schema, - formats: shaping.formats, - }, - &auth_ctx, - ) - .await? + // Read once, ahead of the gateway gate: an in-block write must never + // forward here, or it applies durably outside the transaction. + let tx_state = self.sessions.transaction_state(session_id); + if tx_state != TransactionState::InBlock + && let Some(responses) = self + .maybe_dispatch_tasks_via_gateway( + &tasks, + identity, + tenant_id, + session_id, + ResultShaping { + projection: effective_schema, + formats: shaping.formats, + }, + &auth_ctx, + ) + .await? { return Ok(responses); } - let tx_state = self.sessions.transaction_state(session_id); // Autocommit statement routing: the only reads to widen with are the // ones the materialized-sum settlement stamped on the source rows its // shipped balances were folded from. @@ -216,7 +219,7 @@ impl NodeDbPgHandler { // block, fall through to the per-task staging gate when the gate // can buffer every write — COMMIT then flushes the whole buffer // through Calvin. Anything else is refused, never applied. - if tx_state == crate::control::server::shared::session::TransactionState::InBlock { + if tx_state == TransactionState::InBlock { if !all_writes_bufferable(&tasks) { let (severity, code, message) = error_to_sqlstate(&crate::Error::CrossShardInExplicitTransaction); diff --git a/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs b/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs index a479362d8..0b57d5855 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs @@ -10,29 +10,60 @@ use std::collections::HashMap; use std::sync::Arc; -use pgwire::api::results::{Response, Tag}; +use pgwire::api::results::Response; use pgwire::error::{ErrorInfo, PgWireError, PgWireResult}; use crate::control::security::auth_context::AuthContext; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::response_shape::schema::OutputSchema; +use crate::control::server::response_shape::types::{ + DmlOutcome, StatementTag, payload_to_dml_outcome, staged_dml_outcome, +}; use crate::control::server::shared::session::SessionId; use crate::control::trigger::dml_hook::DmlWriteInfo; use crate::types::TenantId; use nodedb_physical::physical_task::PhysicalTask; -use super::super::super::types::{error_to_sqlstate, shape_error_to_pg}; +use super::super::super::types::{ + dml_fold_error_to_pg, error_to_pg, error_to_sqlstate, shape_error_to_pg, +}; use super::super::core::NodeDbPgHandler; use super::super::plan::PlanKind; +/// What a write task handled short of normal dispatch contributes to the +/// statement's one command tag. +pub(super) enum HandledWrite { + /// A count-bearing outcome: folds into the statement tag. + Dml(DmlOutcome), + /// No count and no verb (a buffered write, a trigger that consumed an + /// opaque plan): the statement renders `OK` unless a DML outcome is + /// folded too. + Opaque, +} + +impl HandledWrite { + /// Fold this contribution into the statement's tag. + pub(super) fn fold_into(self, statement_tag: &mut StatementTag) -> PgWireResult<()> { + match self { + HandledWrite::Dml(outcome) => statement_tag + .fold(outcome) + .map_err(|e| dml_fold_error_to_pg(&e)), + HandledWrite::Opaque => { + statement_tag.fold_opaque(); + Ok(()) + } + } + } +} + /// Outcome of routing a single task through the in-transaction staging gate. pub(super) enum TxnRouteOutcome { /// Not staged/buffered: caller proceeds to normal dispatch with the /// (possibly `txn_id`-stamped) task. Proceed(Box), - /// Fully handled (buffered "OK", or a staged write's real command tag). - /// Caller pushes this response and continues the loop. - Handled(Response), + /// Fully handled (buffered, or a staged write with its real count). + /// Caller folds this into the statement tag and continues the loop. + Handled(HandledWrite), } impl NodeDbPgHandler { @@ -106,13 +137,10 @@ impl NodeDbPgHandler { match routed { Ok(InTxnRoute::Read(routed_task)) => Ok(TxnRouteOutcome::Proceed(routed_task)), - Ok(InTxnRoute::Buffered) => Ok(TxnRouteOutcome::Handled(Response::Execution( - Tag::new("OK"), + Ok(InTxnRoute::Buffered) => Ok(TxnRouteOutcome::Handled(HandledWrite::Opaque)), + Ok(InTxnRoute::Staged(outcome)) => Ok(TxnRouteOutcome::Handled(HandledWrite::Dml( + staged_dml_outcome(outcome.kind, outcome.affected), ))), - Ok(InTxnRoute::Staged(outcome)) => { - let tag = super::super::plan::tag_from_staged(outcome.kind, outcome.affected); - Ok(TxnRouteOutcome::Handled(Response::Execution(tag))) - } Err(StagingGateError::Dispatch(e)) => { let (severity, code, message) = error_to_sqlstate(&e); Err(PgWireError::UserError(Box::new(ErrorInfo::new( @@ -138,11 +166,19 @@ impl NodeDbPgHandler { } } +/// What a pre-dispatch hook answered the task with. +pub(super) enum PreDispatchHandled { + /// A clone write's `RETURNING` rows, encoded. Caller pushes the response. + Rows(Response), + /// A write's contribution to the statement tag. Caller folds it. + Write(HandledWrite), +} + /// Outcome of running the pre-dispatch hooks for a single task. pub(super) enum PreDispatchOutcome { /// The task was fully handled (trigger short-circuit, or clone write - /// interception). Caller pushes this response and continues the loop. - Handled(Response), + /// interception). Caller emits the answer and continues the loop. + Handled(PreDispatchHandled), /// No interception occurred (or a mutation was applied in place); /// caller proceeds to normal dispatch with the (possibly mutated) task /// and the trigger bookkeeping needed for the AFTER-trigger phase. @@ -269,9 +305,30 @@ impl NodeDbPgHandler { ))) })? { PreDispatchResult::Handled => { - return Ok(PreDispatchOutcome::Handled(Response::Execution(Tag::new( - "OK", - )))); + // The trigger consumed the row: the statement ran and + // affected nothing. A count-bearing plan keeps its verb + // with a zero count so the fold stays on one verb; an + // opaque plan contributes no count. + let handled = match plan_kind { + PlanKind::DmlResult(verb) => { + HandledWrite::Dml(DmlOutcome { verb, affected: 0 }) + } + // The verb is resolved at apply time and no apply + // happened. The statement is an `INSERT ... ON + // CONFLICT DO UPDATE`, so it reports as `INSERT`. + PlanKind::DmlResultByOp => HandledWrite::Dml(DmlOutcome { + verb: "INSERT", + affected: 0, + }), + PlanKind::Execution + | PlanKind::ArraySlice + | PlanKind::ReturningRows + | PlanKind::SingleDocument + | PlanKind::MultiRow => HandledWrite::Opaque, + }; + return Ok(PreDispatchOutcome::Handled(PreDispatchHandled::Write( + handled, + ))); } PreDispatchResult::Proceed { mutated_fields: Some(fields), @@ -287,19 +344,11 @@ impl NodeDbPgHandler { } // Extract truncate restart_identity info before task is moved. - let truncate_restart_collection = - if let nodedb_physical::physical_plan::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::Truncate { - collection, - restart_identity: true, - .. - }, - ) = &task.plan - { - Some(collection.to_string()) - } else { - None - }; + // Engine-neutral: `truncate_target` names every truncate-shaped op. + let truncate_restart_collection = match task.plan.truncate_target() { + Some((collection, true)) => Some(collection.to_string()), + Some((_, false)) | None => None, + }; // --- Clone write-path interception --- // Protocol-neutral hook (`shared::clone_write`); native, RESP, and @@ -348,18 +397,21 @@ impl NodeDbPgHandler { if let Some(n) = notice { self.sessions.push_notice(session_id, n); } - return Ok(PreDispatchOutcome::Handled(response)); + return Ok(PreDispatchOutcome::Handled(PreDispatchHandled::Rows( + response, + ))); } ShapeOutcome::Passthrough => { - let shaped = - crate::control::server::pgwire::handler::plan::payload_to_response( - resp.payload.as_ref(), - plan_kind, - )?; - if let Some(notice) = shaped.notice { - self.sessions.push_notice(session_id, notice); - } - return Ok(PreDispatchOutcome::Handled(shaped.response)); + let handled = + match payload_to_dml_outcome(resp.payload.as_ref(), plan_kind) + .map_err(|e| error_to_pg(&e))? + { + Some(outcome) => HandledWrite::Dml(outcome), + None => HandledWrite::Opaque, + }; + return Ok(PreDispatchOutcome::Handled(PreDispatchHandled::Write( + handled, + ))); } } } diff --git a/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs b/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs index 5fbacb3ff..97b4b21a7 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs @@ -1,98 +1,54 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Gateway-based dispatch: routes tasks through `Gateway::execute` instead of -//! the old SQL-string `ForwardRequest` forwarding path. +//! Gateway-based dispatch: routes tasks through `Gateway::execute`, which +//! ships each pre-planned `PhysicalPlan` to the leader that owns it over QUIC +//! via `ExecuteRequest`, rather than raw SQL text. //! //! Where a task set runs is decided in `placement`; this file carries it out. -//! -//! `dispatch_tasks_via_gateway` replaces `forward_sql`: each task is dispatched -//! via `gateway.execute(ctx, plan)` which ships pre-planned `PhysicalPlan` bytes -//! over QUIC via `ExecuteRequest`, rather than raw SQL text. +//! Shaping the forwarded payloads into the statement's one answer lives in +//! `gateway_fold`. -use pgwire::api::results::{FieldFormat, Response, Tag}; +use pgwire::api::results::{FieldFormat, Response}; use pgwire::error::{ErrorInfo, PgWireError, PgWireResult}; use crate::control::gateway::GatewayErrorMap; +use crate::control::planner::calvin::write_class::{ + plan_counts_toward_statement_tag, plans_have_user_write, +}; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::security::request_scope::RequestAuthScope; -use crate::control::server::response_shape::compose::{self, ShapeOutcome}; use crate::control::server::response_shape::redaction::QueryRedaction; use crate::control::server::response_shape::schema::OutputSchema; use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; +use crate::control::server::shared::session::SessionId; use crate::types::{TenantId, TraceId}; use nodedb_physical::physical_task::PhysicalTask; -use super::super::super::types::shape_error_to_pg; use super::super::core::NodeDbPgHandler; -use super::super::plan::{PlanKind, multirow_payload_to_response}; -use super::super::shape_encode; +use super::super::plan::describe_plan; +use super::gateway_fold::{GatewayFold, GatewayShaping}; /// Meter one gateway-forwarded task, once its response has already shaped /// successfully — mirrors `calvin_dispatch::meter_calvin_task`, the sibling /// remote-dispatch door that bypasses `dispatch_task_loop` the same way. /// -/// `rows` is `Some(shaped.rows.len())` when the response was decoded into -/// rows by the shaping step just above the call site, `None` for a -/// `Passthrough` shape (no decoded row count) or an empty-payload `OK` tag -/// (no row payload at all) — `meter_dispatch` charges one unit for `None`, -/// correct for a write or a zero-row read. +/// `rows` is the row count the shaping step decoded, `None` for a +/// passthrough payload (no decoded row count) — `meter_dispatch` charges one +/// unit for `None`, correct for a write or a zero-row read. fn meter_gateway_task( state: &crate::control::state::SharedState, identity: &AuthenticatedIdentity, database_id: nodedb_types::id::DatabaseId, - plan: &crate::bridge::envelope::PhysicalPlan, + info: &PlanMeteringInfo, rows: Option, ) { if !state.metering_config.enabled { return; } - let info = PlanMeteringInfo::extract(plan); let scope = RequestAuthScope::builder(identity, state.auth_stores()) .with_session_database(Some(database_id)) .build(); - meter_dispatch(state, &scope, &info, rows); -} - -/// Shape one payload the same way a normal gateway response payload is -/// shaped, and push the result onto `responses`. Shared by the forwarded -/// per-payload loop and the clone-write `Handled` short-circuit, which -/// carries exactly one payload. -fn push_shaped_response( - responses: &mut Vec, - payload: &[u8], - projection: Option<&OutputSchema>, - result_formats: &[FieldFormat], - redaction: &QueryRedaction, - state: &crate::control::state::SharedState, -) -> PgWireResult<()> { - if payload.is_empty() { - responses.push(Response::Execution(Tag::new("OK"))); - return Ok(()); - } - // Gateway forwarding carries no session: a projection with Control-Plane - // computed columns is refused by the shaper rather than NULL-filled. - match compose::shape_payload_no_plan( - payload, - PlanKind::MultiRow, - projection, - Some(redaction.ctx(&state.redaction)), - None, - ) - .map_err(|e| shape_error_to_pg(&e))? - { - ShapeOutcome::Rows(shaped) => { - let (response, notice) = shape_encode::shaped_query_response(shaped, result_formats); - debug_assert!( - notice.is_none(), - "MultiRow gateway response must not carry a NOTICE" - ); - responses.push(response); - } - ShapeOutcome::Passthrough => { - responses.push(multirow_payload_to_response(payload).response); - } - } - Ok(()) + meter_dispatch(state, &scope, info, rows); } /// Everything a gateway dispatch needs besides the tasks themselves. @@ -104,6 +60,8 @@ pub(super) struct GatewayDispatchParams<'a> { pub(super) identity: &'a AuthenticatedIdentity, pub(super) tenant_id: TenantId, pub(super) database_id: nodedb_types::id::DatabaseId, + /// Receives a shaped row set's NOTICE. + pub(super) session_id: SessionId, pub(super) projection: Option<&'a OutputSchema>, pub(super) result_formats: &'a [FieldFormat], /// The requester's resolved context; its roles drive column-level @@ -120,6 +78,10 @@ impl NodeDbPgHandler { /// remote leader's own receive side never re-runs it (see /// `exec_receiver::executor`), so the sending node is the only place the /// copy-up can happen for a task headed off-node. + /// + /// The statement answers exactly as a local dispatch does: rows as they + /// arrive, `RETURNING` rows as one result set, and one command tag + /// folded over every write task. pub(super) async fn dispatch_tasks_via_gateway( &self, tasks: Vec, @@ -129,12 +91,19 @@ impl NodeDbPgHandler { identity, tenant_id, database_id, + session_id, projection, result_formats, auth, } = params; // Resolved once for the whole forwarded task set, before the loop. let redaction = QueryRedaction::for_plans(tenant_id, auth, tasks.iter().map(|t| &t.plan)); + let shaping = GatewayShaping { + projection, + result_formats, + redaction: &redaction, + session_id, + }; let gateway = self.state.gateway.get().ok_or_else(|| { PgWireError::UserError(Box::new(ErrorInfo::new( "ERROR".to_owned(), @@ -143,6 +112,8 @@ impl NodeDbPgHandler { ))) })?; + // Forwarding is autocommit only: an in-block statement never reaches + // here (`maybe_dispatch_tasks_via_gateway`), so no transaction id. let gw_ctx = crate::control::gateway::core::QueryContext { tenant_id, trace_id: TraceId::generate(), @@ -150,9 +121,14 @@ impl NodeDbPgHandler { txn_id: None, }; - let mut responses: Vec = Vec::with_capacity(tasks.len()); + // A derived implicit-edge write beside the user's own never answers + // the statement, exactly as Calvin's deposit rule has it. + let has_user_write = plans_have_user_write(tasks.iter().map(|t| &t.plan)); + let mut fold = GatewayFold::with_capacity(tasks.len()); for task in tasks { - let plan_for_metering = task.plan.clone(); + let plan_kind = describe_plan(&task.plan); + let counts_toward_tag = plan_counts_toward_statement_tag(&task.plan, has_user_write); + let metering_info = PlanMeteringInfo::extract(&task.plan); let emitter = crate::control::security::audit::ArcAuditEmitter(std::sync::Arc::clone( &self.state.audit, )); @@ -182,23 +158,21 @@ impl NodeDbPgHandler { resp, ) => { // The clone-write hook fully handled this task locally — - // never forwarded — so shape its response the same way a - // single-payload gateway response would be. - let payload = resp.payload.to_vec(); - push_shaped_response( - &mut responses, - &payload, - projection, - result_formats, - &redaction, - &self.state, + // never forwarded — so its one payload folds exactly + // like a forwarded one. + let rows = self.fold_gateway_payload( + &mut fold, + resp.payload.as_ref(), + plan_kind, + counts_toward_tag, + &shaping, )?; meter_gateway_task( &self.state, identity, database_id, - &plan_for_metering, - None, + &metering_info, + rows, ); continue; } @@ -215,57 +189,42 @@ impl NodeDbPgHandler { ))) })?; + // One task can yield several payloads (e.g. a multi-page scan). + // Metered once per task, on the total row count across every + // payload — never per payload, which bills a single task + // multiple times. A task with no payload at all folds one empty + // payload: an opaque execution folds as such, a count-bearing + // write is refused by the count reader. + let mut task_rows: Option = None; if payloads.is_empty() { - responses.push(Response::Execution(Tag::new("OK"))); - meter_gateway_task(&self.state, identity, database_id, &plan_for_metering, None); - } else { - // One task can yield several payloads (e.g. a multi-page - // scan). Metered once per task below, on the total row count - // across every payload — never per payload, or a single task - // would be billed multiple times. - let mut task_rows: Option = None; - for payload in &payloads { - match compose::shape_payload_no_plan( - payload, - PlanKind::MultiRow, - projection, - Some(redaction.ctx(&self.state.redaction)), - None, - ) - .map_err(|e| shape_error_to_pg(&e))? - { - ShapeOutcome::Rows(shaped) => { - task_rows = Some(task_rows.unwrap_or(0) + shaped.rows.len() as u64); - let (response, notice) = - shape_encode::shaped_query_response(shaped, result_formats); - // The gateway has no `addr` to route a NOTICE to; the - // MultiRow shape never carries one, so assert loudly - // rather than silently swallowing. - debug_assert!( - notice.is_none(), - "MultiRow gateway response must not carry a NOTICE" - ); - responses.push(response); - } - ShapeOutcome::Passthrough => { - responses.push(multirow_payload_to_response(payload).response); - } - } + task_rows = self.fold_gateway_payload( + &mut fold, + &[], + plan_kind, + counts_toward_tag, + &shaping, + )?; + } + for payload in &payloads { + if let Some(rows) = self.fold_gateway_payload( + &mut fold, + payload, + plan_kind, + counts_toward_tag, + &shaping, + )? { + task_rows = Some(task_rows.unwrap_or(0) + rows); } - meter_gateway_task( - &self.state, - identity, - database_id, - &plan_for_metering, - task_rows, - ); } + meter_gateway_task( + &self.state, + identity, + database_id, + &metering_info, + task_rows, + ); } - if responses.is_empty() { - responses.push(Response::Execution(Tag::new("OK"))); - } - - Ok(responses) + Ok(self.finish_gateway_fold(fold, &shaping)) } } diff --git a/nodedb/src/control/server/pgwire/handler/routing/gateway_fold.rs b/nodedb/src/control/server/pgwire/handler/routing/gateway_fold.rs new file mode 100644 index 000000000..b84981b01 --- /dev/null +++ b/nodedb/src/control/server/pgwire/handler/routing/gateway_fold.rs @@ -0,0 +1,145 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Folds forwarded task payloads into the statement's one answer: rows pushed +//! as they arrive, `RETURNING` rows as one result set, and every write folded +//! through [`StatementTag`] into one command tag — the same tail +//! `dispatch_loop/finish.rs` emits for a locally dispatched statement. + +use pgwire::api::results::{FieldFormat, Response}; +use pgwire::error::PgWireResult; + +use crate::control::server::response_shape::compose::{self, ShapeOutcome}; +use crate::control::server::response_shape::redaction::QueryRedaction; +use crate::control::server::response_shape::schema::OutputSchema; +use crate::control::server::response_shape::types::{ + ShapedRows, StatementTag, payload_to_dml_outcome, +}; +use crate::control::server::shared::session::SessionId; + +use super::super::super::command_tag::push_folded_tag; +use super::super::super::types::{dml_fold_error_to_pg, error_to_pg, shape_error_to_pg}; +use super::super::core::NodeDbPgHandler; +use super::super::plan::PlanKind; +use super::super::shape_encode; + +/// How forwarded rows shape back to the client. +pub(super) struct GatewayShaping<'a> { + pub(super) projection: Option<&'a OutputSchema>, + pub(super) result_formats: &'a [FieldFormat], + /// Resolved once over the whole forwarded task set. + pub(super) redaction: &'a QueryRedaction, + /// Receives a shaped row set's NOTICE. + pub(super) session_id: SessionId, +} + +/// The statement's answer, accumulated across every forwarded task. +pub(super) struct GatewayFold { + responses: Vec, + /// The statement's `RETURNING` rows, folded across every task. + returning_rows: Option, + /// The statement's command tag, folded across every write task. + statement_tag: StatementTag, +} + +impl GatewayFold { + pub(super) fn with_capacity(tasks: usize) -> Self { + Self { + responses: Vec::with_capacity(tasks), + returning_rows: None, + statement_tag: StatementTag::default(), + } + } +} + +impl NodeDbPgHandler { + /// The statement's responses: its rows, then its one command tag. + pub(super) fn finish_gateway_fold( + &self, + fold: GatewayFold, + shaping: &GatewayShaping<'_>, + ) -> Vec { + let GatewayFold { + mut responses, + returning_rows, + statement_tag, + } = fold; + if let Some(shaped) = returning_rows { + let (response, notice) = + shape_encode::shaped_query_response(shaped, shaping.result_formats); + if let Some(n) = notice { + self.sessions.push_notice(shaping.session_id, n); + } + responses.push(response); + } + // A statement never carries both RETURNING rows and a count-bearing + // tag: `RETURNING` classifies as `ReturningRows`, which folds rows + // and no count. + push_folded_tag(&mut responses, statement_tag.finish()); + responses + } + + /// Fold one forwarded payload by its task's plan kind: rows push onto + /// the fold, `RETURNING` rows accumulate into one result set, a + /// passthrough payload folds its count and verb into the statement's + /// tag, and an opaque execution folds as such. A task whose count does + /// not answer the statement (`counts_toward_tag` false: a derived + /// implicit-edge write beside the user's own) folds as opaque. + /// + /// Returns the rows the payload decoded to, `None` for a passthrough + /// payload with no row count. + pub(super) fn fold_gateway_payload( + &self, + fold: &mut GatewayFold, + payload: &[u8], + plan_kind: PlanKind, + counts_toward_tag: bool, + shaping: &GatewayShaping<'_>, + ) -> PgWireResult> { + // Gateway forwarding carries no sequence access: a projection with + // Control-Plane computed columns is refused by the shaper rather + // than NULL-filled. + match compose::shape_payload_no_plan( + payload, + plan_kind, + shaping.projection, + Some(shaping.redaction.ctx(&self.state.redaction)), + None, + ) + .map_err(|e| shape_error_to_pg(&e))? + { + ShapeOutcome::Rows(shaped) => { + let rows = shaped.rows.len() as u64; + if matches!(plan_kind, PlanKind::ReturningRows) { + match &mut fold.returning_rows { + Some(accumulated) => accumulated.append(shaped), + None => fold.returning_rows = Some(shaped), + } + } else { + let (response, notice) = + shape_encode::shaped_query_response(shaped, shaping.result_formats); + if let Some(n) = notice { + self.sessions.push_notice(shaping.session_id, n); + } + fold.responses.push(response); + } + Ok(Some(rows)) + } + ShapeOutcome::Passthrough => { + if !counts_toward_tag { + fold.statement_tag.fold_opaque(); + return Ok(None); + } + // A count-bearing plan with no payload is refused by the + // count reader, never answered without a count. + match payload_to_dml_outcome(payload, plan_kind).map_err(|e| error_to_pg(&e))? { + Some(outcome) => fold + .statement_tag + .fold(outcome) + .map_err(|e| dml_fold_error_to_pg(&e))?, + None => fold.statement_tag.fold_opaque(), + } + Ok(None) + } + } + } +} diff --git a/nodedb/src/control/server/pgwire/handler/routing/mod.rs b/nodedb/src/control/server/pgwire/handler/routing/mod.rs index bbfb76ea5..a72277fad 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/mod.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/mod.rs @@ -9,7 +9,6 @@ //! physical plan via `ExecuteRequest` instead of a raw SQL string. mod calvin_dispatch; -mod calvin_response; mod catalog; mod check_enforcement; mod cluster_array; @@ -18,6 +17,7 @@ pub(in crate::control::server::pgwire::handler) mod execute; mod execute_dml_hooks; mod execute_entry; mod gateway_dispatch; +mod gateway_fold; mod placement; mod planning; mod pre_dispatch; diff --git a/nodedb/src/control/server/pgwire/handler/routing/planning.rs b/nodedb/src/control/server/pgwire/handler/routing/planning.rs index 9e3c16cd4..769a92c39 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/planning.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/planning.rs @@ -3,7 +3,7 @@ //! SQL planning: converts SQL text into physical task lists, and selects the //! read consistency a planned task set requires. //! -//! Calvin batch response shaping lives in `calvin_response.rs`. +//! Calvin batch response shaping lives in `response_shape::calvin_fold`. use std::sync::Arc; diff --git a/nodedb/src/control/server/pgwire/handler/routing/pre_dispatch.rs b/nodedb/src/control/server/pgwire/handler/routing/pre_dispatch.rs index a6978eabf..fbea2701d 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/pre_dispatch.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/pre_dispatch.rs @@ -9,7 +9,7 @@ use nodedb_physical::physical_task::PhysicalTask; use crate::control::planner::calvin::plan_needs_implicit_edge_recon; use crate::control::security::identity::AuthenticatedIdentity; -use crate::control::server::shared::session::SessionId; +use crate::control::server::shared::session::{SessionId, TransactionState}; use crate::types::TenantId; use super::placement::TaskPlacement; @@ -98,7 +98,7 @@ impl NodeDbPgHandler { formats: result_formats, } = shaping; let tx_state = self.sessions.transaction_state(session_id); - if tx_state == crate::control::server::shared::session::TransactionState::InBlock + if tx_state == TransactionState::InBlock || self.state.calvin_completion_registry.get().is_none() { return Ok(None); @@ -138,6 +138,14 @@ impl NodeDbPgHandler { /// /// Unresolved multi-step DML stays local so its orchestrator can resolve /// final plans before authorization. + /// + /// The caller skips this for an in-block statement. A write forwarded + /// here applies durably at once, outside the transaction; the dispatch + /// loop's staging gate stages it on the owner under this transaction + /// instead, through `leader_forward`. A read forwarded here loses the + /// transaction id its gather resolves with and records no read for + /// commit-time validation; the loop's gather reaches the owner carrying + /// the transaction id, so the owner resolves this transaction's overlay. pub(super) async fn maybe_dispatch_tasks_via_gateway( &self, tasks: &[PhysicalTask], @@ -179,6 +187,7 @@ impl NodeDbPgHandler { identity, tenant_id, database_id, + session_id, projection, result_formats, auth, diff --git a/nodedb/src/control/server/pgwire/types/error_map.rs b/nodedb/src/control/server/pgwire/types/error_map.rs index 8beed245b..fbd48e481 100644 --- a/nodedb/src/control/server/pgwire/types/error_map.rs +++ b/nodedb/src/control/server/pgwire/types/error_map.rs @@ -7,6 +7,7 @@ use pgwire::error::{ErrorInfo, PgWireError}; use crate::OllpExhaustedCause; use crate::bridge::envelope::{ErrorCode, Status}; +use crate::control::server::response_shape::types::DmlFoldError; /// Create a pgwire ErrorResponse with a SQLSTATE code. pub fn sqlstate_error(code: &str, message: &str) -> PgWireError { @@ -17,6 +18,24 @@ pub fn sqlstate_error(code: &str, message: &str) -> PgWireError { ))) } +/// Map a statement-tag fold refusal to the pgwire error the client reads. +/// Two tasks of one statement disagreeing on their verb is a planner bug, +/// so it surfaces as an internal error. +pub fn dml_fold_error_to_pg(e: &DmlFoldError) -> PgWireError { + sqlstate_error("XX000", &e.to_string()) +} + +/// Map a NodeDB `Error` to the pgwire error the client reads, through the +/// one SQLSTATE table [`error_to_sqlstate`] owns. +pub fn error_to_pg(err: &crate::Error) -> PgWireError { + let (severity, code, message) = error_to_sqlstate(err); + PgWireError::UserError(Box::new(ErrorInfo::new( + severity.to_owned(), + code.to_owned(), + message, + ))) +} + /// Map an error raised while shaping a response to the pgwire error the /// client reads, with the SQLSTATE its numeric code maps to. A per-row /// sequence accessor refusal (`42704`, `55000`) or a division by zero @@ -194,6 +213,11 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str crate::Error::DataPlane(code) => { crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate(code) } + crate::Error::Shaping(e) => ( + "ERROR", + numeric_code_to_sqlstate(e.code()), + e.message().to_string(), + ), _ => ("ERROR", sqlstate::INTERNAL_ERROR, err.to_string()), } } diff --git a/nodedb/src/control/server/pgwire/types/mod.rs b/nodedb/src/control/server/pgwire/types/mod.rs index e527819fd..b15b76920 100644 --- a/nodedb/src/control/server/pgwire/types/mod.rs +++ b/nodedb/src/control/server/pgwire/types/mod.rs @@ -11,8 +11,8 @@ pub mod parse; pub mod privilege; pub use error_map::{ - error_to_sqlstate, notice_warning, response_status_to_sqlstate, shape_error_to_pg, - sqlstate_error, + dml_fold_error_to_pg, error_to_pg, error_to_sqlstate, notice_warning, + response_status_to_sqlstate, shape_error_to_pg, sqlstate_error, }; pub use field::{ bool_field, bytea_field, float4_array_field, float4_field, float8_array_field, float8_field, diff --git a/nodedb/src/control/server/resp/handler_hash.rs b/nodedb/src/control/server/resp/handler_hash.rs index 828c256e1..e33d19ea7 100644 --- a/nodedb/src/control/server/resp/handler_hash.rs +++ b/nodedb/src/control/server/resp/handler_hash.rs @@ -176,6 +176,7 @@ pub(super) async fn handle_flushdb(session: &RespSession, state: &SharedState) - nodedb_types::DatabaseId::DEFAULT, &session.collection, ), + restart_identity: false, }); match dispatch_kv_write(state, session, plan).await { diff --git a/nodedb/src/control/server/response_shape/calvin_fold.rs b/nodedb/src/control/server/response_shape/calvin_fold.rs new file mode 100644 index 000000000..acbff10fc --- /dev/null +++ b/nodedb/src/control/server/response_shape/calvin_fold.rs @@ -0,0 +1,282 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Fold a completed Calvin batch into the statement's ONE result set and ONE +//! command tag. Protocol-neutral: pgwire and native both render this fold. +//! +//! Calvin deposits ONE applied response for the whole transaction (a second +//! RETURNING-bearing participant fails the statement upstream), and every +//! task reads that same payload: a RETURNING task yields the rows, taken once; +//! every other task folds its count into the tag. + +use nodedb_physical::physical_plan::{DocumentOp, KvOp, PhysicalPlan}; + +use super::compose::{ShapeOutcome, shape_response_materialized}; +use super::redaction::QueryRedaction; +use super::request::MaterializedShapeRequest; +use super::schema::OutputSchema; +use super::types::{ + DmlFoldError, DmlOutcome, FoldedTag, PlanKind, ShapedRows, StatementTag, describe_plan, + dml_outcome_by_op, dml_outcome_from_payload, +}; +use crate::bridge::envelope::Response; +use crate::control::planner::calvin::write_class::{ + plan_counts_toward_statement_tag, plans_have_user_write, +}; +use crate::control::security::auth_context::AuthContext; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId}; + +/// Shared inputs for shaping one completed Calvin batch. +pub struct CalvinFoldCtx<'a> { + /// The statement's announced output columns, when it announced any. A + /// RETURNING write is held to them exactly as the single-shard dispatch + /// loop holds it, so the same statement renders the same row whichever + /// route it took. + pub projection: Option<&'a OutputSchema>, + pub state: &'a SharedState, + pub tenant_id: TenantId, + pub database_id: DatabaseId, + /// The requester's resolved context; its roles drive column-level + /// redaction of any RETURNING rows this batch surfaces. + pub auth: &'a AuthContext, +} + +/// One Calvin task's contribution to the statement's response. +pub enum CalvinTaskOutcome { + /// RETURNING rows, folded into the statement's single result set. + Rows(ShapedRows), + /// A count-bearing outcome, folded into the statement's one tag. + Dml(DmlOutcome), + /// No count and no verb: the statement renders `OK` unless a + /// count-bearing task folds too. + Opaque, +} + +/// Why a batch did not fold. +#[derive(Debug)] +pub enum CalvinFoldError { + /// Two tasks reported verbs that cannot share one tag. + Verb(DmlFoldError), + /// A task's payload could not be read as its plan kind requires. + Shape(crate::Error), +} + +/// The statement's answer: its rows, when a task carried RETURNING, and its tag. +pub struct CalvinBatchFold { + pub rows: Option, + pub tag: Option, +} + +/// Fold every task of a completed batch, in dispatch order. +/// +/// A derived implicit-edge write beside the user's own never answers the +/// statement: it folds as opaque, as it never deposits. The rows are taken +/// once rather than accumulated, which would repeat the identical payload per +/// task. +pub fn fold_calvin_batch( + plans: &[&PhysicalPlan], + apply_resp: Option<&Response>, + ctx: &CalvinFoldCtx<'_>, +) -> Result { + let has_user_write = plans_have_user_write(plans.iter().copied()); + let mut rows: Option = None; + let mut tag = StatementTag::default(); + for plan in plans { + if !plan_counts_toward_statement_tag(plan, has_user_write) { + tag.fold_opaque(); + continue; + } + match calvin_task_outcome(plan, apply_resp, ctx).map_err(CalvinFoldError::Shape)? { + CalvinTaskOutcome::Rows(shaped) => { + rows.get_or_insert(shaped); + } + CalvinTaskOutcome::Dml(outcome) => tag.fold(outcome).map_err(CalvinFoldError::Verb)?, + CalvinTaskOutcome::Opaque => tag.fold_opaque(), + } + } + Ok(CalvinBatchFold { + rows, + tag: tag.finish(), + }) +} + +/// Shape one task of a completed batch. +/// +/// A plan carrying RETURNING yields its rows as protocol-neutral +/// [`ShapedRows`] when the deposited payload shapes; a multi-row write plans +/// one task per row, and the caller folds every such task's rows into ONE +/// result set. Every other task (and a RETURNING task with no shaped payload) +/// contributes to the statement's one command tag. +pub fn calvin_task_outcome( + plan: &PhysicalPlan, + apply_resp: Option<&Response>, + ctx: &CalvinFoldCtx<'_>, +) -> crate::Result { + let plan_kind = describe_plan(plan); + let redaction = QueryRedaction::for_plan(ctx.tenant_id, ctx.auth, plan); + if let (PlanKind::ReturningRows, Some(resp)) = (plan_kind, apply_resp) + && let Ok(ShapeOutcome::Rows(shaped)) = + shape_response_materialized(MaterializedShapeRequest { + payload: resp.payload.as_bytes(), + plan, + plan_kind: PlanKind::ReturningRows, + projection: ctx.projection, + state: ctx.state, + database_id: ctx.database_id, + tenant_id: ctx.tenant_id, + redaction: Some(redaction.ctx(&ctx.state.redaction)), + // A RETURNING list names stored columns only, never a + // Control-Plane computed column. + sequences: None, + }) + { + return Ok(CalvinTaskOutcome::Rows(shaped)); + } + + // Plain write: surface its ACTUAL affected count from the payload. Every + // primary-write participant deposits its applied `Response` before + // proposing the completion ack, so a count-bearing plan ALWAYS has one + // here. If it does not, the deposit path regressed: fail loudly rather + // than synthesise a count, which is what made a delete of an absent row + // report a removed row. + let applied = |tag: &str| { + apply_resp.ok_or_else(|| crate::Error::Internal { + detail: format!( + "Calvin {tag} completed with no applied response to read its affected-row \ + count from" + ), + }) + }; + match plan_kind { + PlanKind::DmlResult(verb) => { + let resp = applied(verb)?; + Ok(CalvinTaskOutcome::Dml(dml_outcome_from_payload( + resp.payload.as_bytes(), + verb, + )?)) + } + // The verb is in the payload; the error text only names the kind. + PlanKind::DmlResultByOp => { + let resp = applied("insert-or-update")?; + Ok(CalvinTaskOutcome::Dml(dml_outcome_by_op( + resp.payload.as_bytes(), + )?)) + } + PlanKind::Execution + | PlanKind::ArraySlice + | PlanKind::ReturningRows + | PlanKind::SingleDocument + | PlanKind::MultiRow => match calvin_tag_for_plan(plan) { + Some(outcome) => Ok(CalvinTaskOutcome::Dml(outcome)), + None => Ok(CalvinTaskOutcome::Opaque), + }, + } +} + +/// The count-bearing outcome a plan answers without a round-trip to the +/// Data Plane, or `None` when its count depends on state the plan never read. +/// +/// Folding is only sound for a write that CANNOT be a no-op — one that either +/// applies exactly one row or fails the statement. Any write whose row count +/// depends on state (a point delete of an absent row, an `ON CONFLICT DO +/// NOTHING` insert onto an existing key, a batch, a predicate) gets its +/// count from the mutation's own response, because a synthesised count is a +/// claim about rows nobody looked at. +/// +/// Foldable: `DocumentOp::PointPut` (`INSERT 0 1`) and `KvOp::Put` +/// (`UPSERT 1`), the two upserts that always write one row. Every other plan, +/// read or write, answers `None`. +pub fn calvin_tag_for_plan(plan: &PhysicalPlan) -> Option { + match plan { + PhysicalPlan::Document(DocumentOp::PointPut { .. }) => Some(DmlOutcome { + verb: "INSERT", + affected: 1, + }), + // The SQL `UPSERT` statement: same tag as its `DocumentOp::Upsert` + // sibling and as `describe_plan`'s `DmlResult("UPSERT")` arm. + PhysicalPlan::Kv(KvOp::Put { .. }) => Some(DmlOutcome { + verb: "UPSERT", + affected: 1, + }), + // The foldable arms above take precedence; every remaining op of each + // engine answers from its payload or folds opaque. + PhysicalPlan::Document(_) + | PhysicalPlan::Kv(_) + | PhysicalPlan::Vector(_) + | PhysicalPlan::Graph(_) + | PhysicalPlan::Text(_) + | PhysicalPlan::Columnar(_) + | PhysicalPlan::Timeseries(_) + | PhysicalPlan::Spatial(_) + | PhysicalPlan::Crdt(_) + | PhysicalPlan::Query(_) + | PhysicalPlan::Meta(_) + | PhysicalPlan::Array(_) + | PhysicalPlan::ClusterArray(_) + | PhysicalPlan::ClusterEvent(_) => None, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_types::{DatabaseId, QualifiedCollection}; + + #[test] + fn a_read_never_folds_to_a_tag() { + let plan = PhysicalPlan::Kv(KvOp::Get { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "items"), + key: Vec::new(), + rls_filters: Vec::new(), + surrogate_ceiling: None, + }); + assert!(calvin_tag_for_plan(&plan).is_none()); + } + + #[test] + fn an_upsert_folds_without_a_round_trip() { + let plan = PhysicalPlan::Kv(KvOp::Put { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "items"), + key: Vec::new(), + value: Vec::new(), + ttl_ms: 0, + surrogate: nodedb_types::Surrogate::ZERO, + returning: None, + rls_filters: Vec::new(), + }); + assert_eq!( + calvin_tag_for_plan(&plan), + Some(DmlOutcome { + verb: "UPSERT", + affected: 1 + }) + ); + } + + /// A write that can legitimately touch nothing never folds: its count + /// is only knowable from the mutation's own response. Folding a delete + /// let a re-delete of an already-deleted key report a removed row. + #[test] + fn no_op_capable_writes_never_fold() { + let delete = PhysicalPlan::Kv(KvOp::Delete { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "items"), + keys: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: None, + rls_filters: Vec::new(), + }); + assert!(calvin_tag_for_plan(&delete).is_none()); + + let point_delete = PhysicalPlan::Document(DocumentOp::PointDelete { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "items"), + document_id: "a".into(), + surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: Vec::new(), + returning: None, + rls_filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + resolved_sum_targets: Vec::new(), + }); + assert!(calvin_tag_for_plan(&point_delete).is_none()); + } +} diff --git a/nodedb/src/control/server/response_shape/compose/materialized.rs b/nodedb/src/control/server/response_shape/compose/materialized.rs index b454ed801..ce00ae8fd 100644 --- a/nodedb/src/control/server/response_shape/compose/materialized.rs +++ b/nodedb/src/control/server/response_shape/compose/materialized.rs @@ -37,9 +37,9 @@ use super::kernel::{empty_shaped, shape_decoded_rows, single_result_row}; /// /// Row-producing plan kinds (`SingleDocument`, `MultiRow`, `ReturningRows`, /// `ArraySlice`) yield `Rows`. Tag/execution kinds (`Execution`, -/// `DmlResult`) yield `Passthrough` — a `ShapedRows` cannot represent a bare -/// `CommandComplete` tag or affected-row count, so callers keep their -/// existing tag / `rows_affected` handling for those. +/// `DmlResult`, `DmlResultByOp`) yield `Passthrough` — a `ShapedRows` cannot +/// represent a bare `CommandComplete` tag or affected-row count, so callers +/// keep their existing tag / `rows_affected` handling for those. pub enum ShapeOutcome { Rows(ShapedRows), Passthrough, @@ -65,7 +65,9 @@ pub fn shape_response_materialized( } = request; match plan_kind { - PlanKind::Execution | PlanKind::DmlResult(_) => return Ok(ShapeOutcome::Passthrough), + PlanKind::Execution | PlanKind::DmlResult(_) | PlanKind::DmlResultByOp => { + return Ok(ShapeOutcome::Passthrough); + } PlanKind::ArraySlice | PlanKind::ReturningRows | PlanKind::SingleDocument @@ -90,7 +92,9 @@ pub fn shape_response_materialized( // Handled by the early return above; kept exhaustive (no catch-all, // no panic) so a future PlanKind desync degrades to passthrough // rather than crashing the connection. - PlanKind::Execution | PlanKind::DmlResult(_) => return Ok(ShapeOutcome::Passthrough), + PlanKind::Execution | PlanKind::DmlResult(_) | PlanKind::DmlResultByOp => { + return Ok(ShapeOutcome::Passthrough); + } }; Ok(ShapeOutcome::Rows(shaped)) } @@ -115,7 +119,9 @@ pub fn shape_payload_no_plan( sequences: Option<&dyn SequenceAccess>, ) -> Result { Ok(match plan_kind { - PlanKind::Execution | PlanKind::DmlResult(_) => ShapeOutcome::Passthrough, + PlanKind::Execution | PlanKind::DmlResult(_) | PlanKind::DmlResultByOp => { + ShapeOutcome::Passthrough + } PlanKind::ArraySlice => ShapeOutcome::Rows(shape_array_slice(payload, redaction)?), PlanKind::ReturningRows => ShapeOutcome::Rows(shape_returning_rows( payload, projection, redaction, sequences, diff --git a/nodedb/src/control/server/response_shape/mod.rs b/nodedb/src/control/server/response_shape/mod.rs index 89735375b..fd8b7ca7a 100644 --- a/nodedb/src/control/server/response_shape/mod.rs +++ b/nodedb/src/control/server/response_shape/mod.rs @@ -2,6 +2,7 @@ //! Shared, protocol-neutral response shaping helpers. +pub mod calvin_fold; pub mod cell; pub mod compose; pub mod kv; diff --git a/nodedb/src/control/server/response_shape/types/dml_outcome.rs b/nodedb/src/control/server/response_shape/types/dml_outcome.rs new file mode 100644 index 000000000..3923d547f --- /dev/null +++ b/nodedb/src/control/server/response_shape/types/dml_outcome.rs @@ -0,0 +1,376 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The count-bearing result of one write task, and the fold of every task's +//! result into the ONE command tag a statement answers with. +//! +//! Carries no pgwire wire types, so every protocol renders it its own way. +//! The payload-to-outcome readers here are the one place a Data Plane +//! count payload becomes a [`DmlOutcome`]; pgwire and native both call them. + +use crate::control::server::shared::sql::staging_predicates::{ + StagedTagKind, extract_kv_conflict_op, require_affected_count, +}; + +use super::PlanKind; + +/// The count-bearing result of one write task, before any protocol renders it. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct DmlOutcome { + /// The command verb, exactly as the tag names it (`INSERT`, `UPDATE`, ...). + pub verb: &'static str, + /// Rows this task affected. + pub affected: u64, +} + +impl DmlOutcome { + /// Whether this outcome reports a row count. See [`Self::verb_carries_count`]. + pub fn carries_count(&self) -> bool { + Self::verb_carries_count(self.verb) + } + + /// Whether `verb` reports a row count. A SQL `TRUNCATE` never does: + /// Postgres tags it bare, and native leaves `rows_affected` unset. Every + /// other verb carries `affected`. Both protocols read this one rule. + pub fn verb_carries_count(verb: &str) -> bool { + verb != "TRUNCATE" + } +} + +/// Two tasks of one statement reported verbs that cannot share one tag. +#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)] +pub enum DmlFoldError { + #[error("one statement reported two command verbs: {first} then {second}")] + VerbMismatch { + first: &'static str, + second: &'static str, + }, +} + +/// The one command tag a statement answers with, folded over its tasks. +/// +/// Fold rules: +/// - Same verb: `affected` sums. +/// - `INSERT` mixed with `UPDATE`: verb `INSERT`, `affected` sums. This is a +/// multi-row `INSERT ... ON CONFLICT DO UPDATE` whose rows resolved +/// differently (`PlanKind::DmlResultByOp`); Postgres tags it `INSERT 0 n`. +/// - Any other verb mix: [`DmlFoldError::VerbMismatch`]. Never a silent pick. +/// - An opaque task (`PlanKind::Execution`) adds nothing when a DML outcome +/// is folded before or after it. Only-opaque folds render as `OK`. +#[derive(Debug, Default)] +pub struct StatementTag { + dml: Option, + opaque: bool, +} + +/// What a statement's fold produced. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum FoldedTag { + Dml(DmlOutcome), + /// Only opaque tasks were folded. Renders as the `OK` tag. + Opaque, +} + +impl StatementTag { + /// Fold one task's count-bearing outcome into the statement's tag. + pub fn fold(&mut self, outcome: DmlOutcome) -> Result<(), DmlFoldError> { + let Some(current) = self.dml else { + self.dml = Some(outcome); + return Ok(()); + }; + let verb = merged_verb(current.verb, outcome.verb).ok_or(DmlFoldError::VerbMismatch { + first: current.verb, + second: outcome.verb, + })?; + self.dml = Some(DmlOutcome { + verb, + affected: current.affected.saturating_add(outcome.affected), + }); + Ok(()) + } + + /// Fold one opaque task: no count, no verb. + pub fn fold_opaque(&mut self) { + self.opaque = true; + } + + /// The statement's tag. `None` when nothing was folded. + pub fn finish(self) -> Option { + match (self.dml, self.opaque) { + (Some(outcome), _) => Some(FoldedTag::Dml(outcome)), + (None, true) => Some(FoldedTag::Opaque), + (None, false) => None, + } + } +} + +/// The count-bearing outcome of a staged write, from the neutral +/// [`StagedTagKind`] the staging gate decided. Verb mapping: `INSERT` / +/// `UPDATE` / `DELETE` by kind, and for a KV `InsertOnConflictUpdate` the +/// verb the stage handler resolved to. +pub(crate) fn staged_dml_outcome(kind: StagedTagKind, affected: usize) -> DmlOutcome { + let verb = match kind { + StagedTagKind::Insert => "INSERT", + StagedTagKind::Update => "UPDATE", + StagedTagKind::Delete => "DELETE", + StagedTagKind::KvUpsert { updated: true } => "UPDATE", + StagedTagKind::KvUpsert { updated: false } => "INSERT", + // Matches the autocommit `DocumentOp::Upsert` / `KvOp::Put` tag + // exactly: always the literal `UPSERT` command, regardless of + // insert-vs-update outcome (see `describe_plan`'s `DmlResult("UPSERT")` + // arms). + StagedTagKind::Upsert => "UPSERT", + // Statement-time in-transaction MERGE: the Postgres command tag for a + // MERGE is `MERGE ` across all arms. + StagedTagKind::Merge => "MERGE", + // Statement-time in-transaction `UPDATE ... FROM`: an UPDATE reports the + // Postgres `UPDATE ` command tag over the matched target rows. + StagedTagKind::UpdateFromJoin => "UPDATE", + // KV `Incr` / `IncrFloat` / `Cas` / `GetSet` never reach a tag: their + // sole SQL surface (`SELECT KV_INCR(..)` and friends, in + // `ddl/neutral/kv_atomic/`) reads `StagedWriteOutcome::payload` + // directly, and both dispatch loops fold `RawPayload` as opaque. This + // arm exists only so the match stays exhaustive against a new + // `PhysicalPlan::Kv` caller; it names the tag a function-call + // `SELECT` renders. + StagedTagKind::RawPayload => "SELECT", + // Staged `TRUNCATE`: `carries_count` is false for this verb, so the + // wire answer is the bare `TRUNCATE` autocommit answers with. + StagedTagKind::Truncate => "TRUNCATE", + }; + DmlOutcome { + verb, + affected: affected as u64, + } +} + +/// The count-bearing outcome a `DmlResult(verb)` payload reports. +/// +/// The count comes from the write, always. There is no "point operations +/// affected exactly 1 row" shortcut: a point delete or a conflicting +/// `ON CONFLICT DO NOTHING` insert is the same plan whether it touched a row +/// or not, so assuming 1 here reported rows that were never there. +pub(crate) fn dml_outcome_from_payload( + payload: &[u8], + verb: &'static str, +) -> crate::Result { + let affected = require_affected_count(payload).map_err(|e| crate::Error::Internal { + detail: format!("{verb} response is missing its affected count: {e}"), + })?; + Ok(DmlOutcome { verb, affected }) +} + +/// The count-bearing outcome a `DmlResultByOp` payload reports. +/// +/// The handler decides insert-vs-update at apply time and reports it as +/// `op`. A missing or unknown verb is a handler bug, never a default tag. +pub(crate) fn dml_outcome_by_op(payload: &[u8]) -> crate::Result { + let affected = require_affected_count(payload).map_err(|e| crate::Error::Internal { + detail: format!("DmlResultByOp response is missing its affected count: {e}"), + })?; + let verb = match extract_kv_conflict_op(payload).as_deref() { + Some("insert") => "INSERT", + Some("update") => "UPDATE", + other => { + return Err(crate::Error::Internal { + detail: format!( + "DmlResultByOp response carries no usable `op` verb \ + (got {other:?}); the handler must report `insert` or `update`" + ), + }); + } + }; + Ok(DmlOutcome { verb, affected }) +} + +/// The count-bearing outcome of a passthrough payload: `Some` for the +/// count-bearing kinds, `None` for an opaque `Execution`, an error for a +/// row-shaped kind (those never reach a tag). +pub(crate) fn payload_to_dml_outcome( + payload: &[u8], + kind: PlanKind, +) -> crate::Result> { + match kind { + PlanKind::Execution => Ok(None), + PlanKind::DmlResult(verb) => dml_outcome_from_payload(payload, verb).map(Some), + PlanKind::DmlResultByOp => dml_outcome_by_op(payload).map(Some), + PlanKind::ArraySlice + | PlanKind::ReturningRows + | PlanKind::SingleDocument + | PlanKind::MultiRow => Err(crate::Error::Internal { + detail: format!("payload_to_dml_outcome cannot handle plan kind {kind:?}"), + }), + } +} + +/// The verb two folded outcomes share, `None` when they cannot share one. +fn merged_verb(first: &'static str, second: &'static str) -> Option<&'static str> { + if first == second { + return Some(first); + } + match (first, second) { + ("INSERT", "UPDATE") | ("UPDATE", "INSERT") => Some("INSERT"), + _ => None, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn outcome(verb: &'static str, affected: u64) -> DmlOutcome { + DmlOutcome { verb, affected } + } + + #[test] + fn empty_fold_finishes_to_none() { + assert_eq!(StatementTag::default().finish(), None); + } + + #[test] + fn same_verb_sums_affected() { + let mut tag = StatementTag::default(); + tag.fold(outcome("INSERT", 1)).expect("first fold"); + tag.fold(outcome("INSERT", 0)).expect("second fold"); + tag.fold(outcome("INSERT", 1)).expect("third fold"); + assert_eq!(tag.finish(), Some(FoldedTag::Dml(outcome("INSERT", 2)))); + } + + #[test] + fn insert_and_update_fold_to_insert_in_either_order() { + let mut tag = StatementTag::default(); + tag.fold(outcome("INSERT", 1)).expect("insert"); + tag.fold(outcome("UPDATE", 2)).expect("update after insert"); + assert_eq!(tag.finish(), Some(FoldedTag::Dml(outcome("INSERT", 3)))); + + let mut tag = StatementTag::default(); + tag.fold(outcome("UPDATE", 2)).expect("update"); + tag.fold(outcome("INSERT", 1)).expect("insert after update"); + assert_eq!(tag.finish(), Some(FoldedTag::Dml(outcome("INSERT", 3)))); + } + + #[test] + fn other_verb_mix_is_an_error() { + let mut tag = StatementTag::default(); + tag.fold(outcome("INSERT", 1)).expect("insert"); + assert_eq!( + tag.fold(outcome("DELETE", 1)), + Err(DmlFoldError::VerbMismatch { + first: "INSERT", + second: "DELETE", + }) + ); + } + + #[test] + fn opaque_never_changes_a_dml_tag() { + let mut tag = StatementTag::default(); + tag.fold_opaque(); + tag.fold(outcome("DELETE", 4)).expect("delete"); + tag.fold_opaque(); + assert_eq!(tag.finish(), Some(FoldedTag::Dml(outcome("DELETE", 4)))); + } + + #[test] + fn only_opaque_finishes_to_opaque() { + let mut tag = StatementTag::default(); + tag.fold_opaque(); + tag.fold_opaque(); + assert_eq!(tag.finish(), Some(FoldedTag::Opaque)); + } + + #[test] + fn truncate_is_the_only_count_less_verb() { + assert!(!outcome("TRUNCATE", 0).carries_count()); + for verb in ["INSERT", "UPDATE", "DELETE", "UPSERT", "MERGE"] { + assert!(outcome(verb, 1).carries_count(), "{verb}"); + } + } + + #[test] + fn passthrough_rejects_precomposed_shapes() { + assert!(payload_to_dml_outcome(&[], PlanKind::ArraySlice).is_err()); + assert!(payload_to_dml_outcome(&[], PlanKind::ReturningRows).is_err()); + assert!(payload_to_dml_outcome(&[], PlanKind::SingleDocument).is_err()); + } + + /// `KvOp::InsertOnConflictUpdate` reports the verb it resolved to; the + /// outcome follows it, and a payload with no verb is refused rather than + /// defaulted. + #[test] + fn dml_result_by_op_follows_the_reported_verb() { + let update = nodedb_types::json_to_msgpack(&serde_json::json!({ + "affected": 1, + "op": "update" + })) + .expect("encode payload"); + assert_eq!( + payload_to_dml_outcome(&update, PlanKind::DmlResultByOp).expect("update outcome"), + Some(outcome("UPDATE", 1)) + ); + + let insert = nodedb_types::json_to_msgpack(&serde_json::json!({ + "affected": 1, + "op": "insert" + })) + .expect("encode payload"); + assert_eq!( + payload_to_dml_outcome(&insert, PlanKind::DmlResultByOp).expect("insert outcome"), + Some(outcome("INSERT", 1)) + ); + + let no_verb = nodedb_types::json_to_msgpack(&serde_json::json!({ "affected": 1 })) + .expect("encode payload"); + assert!(payload_to_dml_outcome(&no_verb, PlanKind::DmlResultByOp).is_err()); + } + + /// A count-bearing response with no count is a handler bug, not a `1`. + #[test] + fn dml_outcome_requires_a_reported_count() { + assert!(payload_to_dml_outcome(&[], PlanKind::DmlResult("DELETE")).is_err()); + let payload = nodedb_types::json_to_msgpack(&serde_json::json!({ "affected": 0 })) + .expect("encode count payload"); + assert_eq!( + payload_to_dml_outcome(&payload, PlanKind::DmlResult("DELETE")).expect("delete"), + Some(outcome("DELETE", 0)) + ); + } + + /// The fold reads the neutral outcome: a count for the count-bearing + /// kinds, nothing for an opaque execution, a refusal for row kinds. + #[test] + fn dml_outcome_follows_plan_kind() { + let payload = nodedb_types::json_to_msgpack(&serde_json::json!({ "affected": 2 })) + .expect("encode count payload"); + assert_eq!( + payload_to_dml_outcome(&payload, PlanKind::DmlResult("INSERT")).expect("insert"), + Some(outcome("INSERT", 2)) + ); + assert_eq!( + payload_to_dml_outcome(&[], PlanKind::Execution).expect("opaque"), + None + ); + assert!(payload_to_dml_outcome(&[], PlanKind::MultiRow).is_err()); + assert!(payload_to_dml_outcome(&[], PlanKind::ReturningRows).is_err()); + } + + /// Staged outcomes carry the verb the staging gate decided. + #[test] + fn staged_outcome_maps_kind_to_verb() { + assert_eq!( + staged_dml_outcome(StagedTagKind::KvUpsert { updated: true }, 1), + outcome("UPDATE", 1) + ); + assert_eq!( + staged_dml_outcome(StagedTagKind::Merge, 4), + outcome("MERGE", 4) + ); + } + + /// A staged `TRUNCATE` is the count-less verb. + #[test] + fn staged_truncate_carries_no_count() { + let staged = staged_dml_outcome(StagedTagKind::Truncate, 0); + assert_eq!(staged, outcome("TRUNCATE", 0)); + assert!(!staged.carries_count()); + } +} diff --git a/nodedb/src/control/server/response_shape/types/mod.rs b/nodedb/src/control/server/response_shape/types/mod.rs index 9b0a0fb2d..0e5cdd436 100644 --- a/nodedb/src/control/server/response_shape/types/mod.rs +++ b/nodedb/src/control/server/response_shape/types/mod.rs @@ -2,8 +2,13 @@ //! Protocol-neutral plan classification and shaped row-set types. +pub mod dml_outcome; pub mod plan_kind; pub mod shaped; +pub use dml_outcome::{DmlFoldError, DmlOutcome, FoldedTag, StatementTag}; +pub(crate) use dml_outcome::{ + dml_outcome_by_op, dml_outcome_from_payload, payload_to_dml_outcome, staged_dml_outcome, +}; pub use plan_kind::{PlanKind, describe_plan}; pub use shaped::{DdlColType, ShapedRow, ShapedRows}; diff --git a/nodedb/src/control/server/response_shape/types/plan_kind.rs b/nodedb/src/control/server/response_shape/types/plan_kind.rs deleted file mode 100644 index 93476319a..000000000 --- a/nodedb/src/control/server/response_shape/types/plan_kind.rs +++ /dev/null @@ -1,478 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Protocol-neutral plan classification types. -//! -//! These operate purely on `PhysicalPlan` and carry no pgwire wire types, -//! so they are shared across any protocol-specific response shaper. - -use crate::bridge::envelope::PhysicalPlan; -use nodedb_physical::physical_plan::{ - ColumnarOp, CrdtOp, DocumentOp, GraphOp, KvOp, QueryOp, SpatialOp, TextOp, TimeseriesOp, - VectorOp, -}; - -#[derive(Debug, Clone, Copy)] -pub enum PlanKind { - SingleDocument, - MultiRow, - /// Array slice result — decoded via `ArraySliceResponse` to surface the - /// `truncated_before_horizon` flag as a pgwire NOTICE when set. - ArraySlice, - Execution, - /// DML operation that returns affected row count. - /// The tag name is used in the pgwire `CommandComplete` message (e.g., "UPDATE", "DELETE"). - DmlResult(&'static str), - /// DML with RETURNING clause — payload is a `RowsPayload` (msgpack). - /// Decoded into one pgwire field per column. - ReturningRows, -} - -pub fn describe_plan(plan: &PhysicalPlan) -> PlanKind { - match plan { - PhysicalPlan::Crdt(CrdtOp::DocUpsert { - returning: Some(_), .. - }) - | PhysicalPlan::Crdt(CrdtOp::DocDelete { - returning: Some(_), .. - }) => PlanKind::ReturningRows, - - // A CRDT delete can legitimately remove nothing, so its count must render - // as a DML count from the write's own response, not a document-shaped read. - PhysicalPlan::Crdt(CrdtOp::DocDelete { .. }) => DmlResult("DELETE"), - - PhysicalPlan::Document(DocumentOp::PointGet { .. }) - | PhysicalPlan::Crdt(CrdtOp::Read { .. }) - | PhysicalPlan::Crdt(CrdtOp::GetPolicy { .. }) - | PhysicalPlan::Crdt(CrdtOp::DocUpsert { .. }) => PlanKind::SingleDocument, - - PhysicalPlan::Vector(VectorOp::Search { .. }) - | PhysicalPlan::Vector(VectorOp::MultiSearch { .. }) - | PhysicalPlan::Vector(VectorOp::MultiVectorScoreSearch { .. }) - | PhysicalPlan::Vector(VectorOp::SparseSearch { .. }) - | PhysicalPlan::Document(DocumentOp::RangeScan { .. }) - | PhysicalPlan::Graph(GraphOp::Hop { .. }) - | PhysicalPlan::Graph(GraphOp::Neighbors { .. }) - | PhysicalPlan::Graph(GraphOp::Path { .. }) - | PhysicalPlan::Graph(GraphOp::Subgraph { .. }) - | PhysicalPlan::Graph(GraphOp::RagFusion { .. }) - | PhysicalPlan::Document(DocumentOp::Scan { .. }) - | PhysicalPlan::Document(DocumentOp::IndexedFetch { .. }) - | PhysicalPlan::Columnar(ColumnarOp::Scan { .. }) - | PhysicalPlan::Timeseries(TimeseriesOp::Scan { .. }) - | PhysicalPlan::Spatial(SpatialOp::Scan { .. }) - | PhysicalPlan::Kv(KvOp::Scan { .. }) - | PhysicalPlan::Kv(KvOp::BatchGet { .. }) - | PhysicalPlan::Query(QueryOp::Aggregate { .. }) - | PhysicalPlan::Query(QueryOp::FacetCounts { .. }) - | PhysicalPlan::Query(QueryOp::HashJoin { .. }) - | PhysicalPlan::Query(QueryOp::RecursiveScan { .. }) - | PhysicalPlan::Query(QueryOp::RecursiveValue { .. }) - | PhysicalPlan::Query(QueryOp::LateralTopK { .. }) - | PhysicalPlan::Query(QueryOp::LateralLoop { .. }) - | PhysicalPlan::Graph(GraphOp::Algo { .. }) - | PhysicalPlan::Graph(GraphOp::Match { .. }) - | PhysicalPlan::Graph(GraphOp::MatchContinuation { .. }) - | PhysicalPlan::Graph(GraphOp::MatchVarLenResume { .. }) - | PhysicalPlan::Graph(GraphOp::BspSuperstep(_)) - | PhysicalPlan::Graph(GraphOp::WccSuperstep(_)) - | PhysicalPlan::Text(TextOp::Search { .. }) - | PhysicalPlan::Text(TextOp::PhraseSearch { .. }) - | PhysicalPlan::Text(TextOp::HybridSearch { .. }) - | PhysicalPlan::Text(TextOp::HybridSearchTriple { .. }) - | PhysicalPlan::Text(TextOp::BM25ScoreScan { .. }) - | PhysicalPlan::Text(TextOp::FtsIndexDoc { .. }) - | PhysicalPlan::Text(TextOp::FtsDeleteDoc { .. }) => PlanKind::MultiRow, - - // Opaque execution results: config write, index teardown status. - PhysicalPlan::Text(TextOp::SetTextConfig { .. }) - | PhysicalPlan::Vector(VectorOp::DropIndex { .. }) - // Internal typed zerompk value, never a client row — decoded by the - // admission caller as `CrdtPreviewResult`. - | PhysicalPlan::Crdt(CrdtOp::PreviewApply { .. }) => PlanKind::Execution, - - PhysicalPlan::Kv(KvOp::Get { .. }) | PhysicalPlan::Kv(KvOp::FieldGet { .. }) => { - PlanKind::SingleDocument - } - - // Constant/catalog-scan expressions compile to ProviderScan; route MultiRow - // so each array element streams as its own pgwire row. - PhysicalPlan::Query(QueryOp::ProviderScan { .. }) => PlanKind::MultiRow, - - // Exchange means the plan wasn't yet resolved — recurse into the child. - PhysicalPlan::Query(QueryOp::Exchange(op)) => describe_plan(&op.child), - - // PostProcess reshapes a multi-row subquery; its kind is the child's. - PhysicalPlan::Query(QueryOp::PostProcess { input, .. }) => describe_plan(input), - - // SetOp resolves to a ProviderScan of merged rows; route MultiRow so - // each row streams as its own pgwire row. - PhysicalPlan::Query(QueryOp::SetOp { .. }) => PlanKind::MultiRow, - - // An insert with a projection returns real stored rows and must be decoded - // and redacted, else it silently leaks unredacted rows like `Merge` did. - PhysicalPlan::Kv( - KvOp::Insert { - returning: Some(_), .. - } - | KvOp::InsertIfAbsent { - returning: Some(_), .. - } - | KvOp::InsertOnConflictUpdate { - returning: Some(_), .. - } - | KvOp::Put { - returning: Some(_), .. - } - | KvOp::BatchPut { - returning: Some(_), .. - }, - ) - | PhysicalPlan::Document(DocumentOp::PointPut { - returning: Some(_), .. - }) - | PhysicalPlan::Document(DocumentOp::PointInsert { - returning: Some(_), .. - }) - | PhysicalPlan::Document(DocumentOp::BatchInsert { - returning: Some(_), .. - }) - | PhysicalPlan::Columnar(ColumnarOp::Insert { - returning: Some(_), .. - }) - | PhysicalPlan::Timeseries(TimeseriesOp::Ingest { - returning: Some(_), .. - }) - | PhysicalPlan::Vector(VectorOp::DirectUpsert { - returning: Some(_), .. - }) => PlanKind::ReturningRows, - - // `PointInsert`/`InsertIfAbsent`: `ON CONFLICT DO NOTHING` makes them - // no-op-capable, so the count must come from the write's response. - PhysicalPlan::Document(DocumentOp::PointPut { .. }) - | PhysicalPlan::Document(DocumentOp::PointInsert { .. }) - | PhysicalPlan::Document(DocumentOp::BatchInsert { .. }) - | PhysicalPlan::Kv(KvOp::InsertIfAbsent { .. }) - | PhysicalPlan::Columnar(ColumnarOp::Insert { .. }) => DmlResult("INSERT"), - - PhysicalPlan::Document(DocumentOp::PointUpdate { - returning: Some(_), .. - }) - | PhysicalPlan::Document(DocumentOp::BulkUpdate { - returning: Some(_), .. - }) => PlanKind::ReturningRows, - PhysicalPlan::Document(DocumentOp::PointUpdate { .. }) - | PhysicalPlan::Document(DocumentOp::BulkUpdate { .. }) => DmlResult("UPDATE"), - - PhysicalPlan::Document(DocumentOp::PointDelete { - returning: Some(_), .. - }) - | PhysicalPlan::Document(DocumentOp::BulkDelete { - returning: Some(_), .. - }) => PlanKind::ReturningRows, - PhysicalPlan::Document(DocumentOp::PointDelete { .. }) - | PhysicalPlan::Document(DocumentOp::BulkDelete { .. }) => DmlResult("DELETE"), - - PhysicalPlan::Document(DocumentOp::UpdateFromJoin { - returning: Some(_), .. - }) => PlanKind::ReturningRows, - PhysicalPlan::Document(DocumentOp::UpdateFromJoin { .. }) => DmlResult("UPDATE"), - - // A MERGE with a projection returns real target rows and must be decoded - // and redacted, else it falls through to unredacted `Execution` passthrough. - PhysicalPlan::Document(DocumentOp::Merge { - returning: Some(_), .. - }) => PlanKind::ReturningRows, - // Postgres tags a plain MERGE `MERGE `, matching the staged path. - PhysicalPlan::Document(DocumentOp::Merge { .. }) => DmlResult("MERGE"), - - PhysicalPlan::Document(DocumentOp::Truncate { .. }) => DmlResult("TRUNCATE"), - - // A KV update/delete with a projection returns real stored rows and - // must be decoded and redacted, exactly like the KV insert ops above. - PhysicalPlan::Kv( - KvOp::FieldSet { - returning: Some(_), .. - } - | KvOp::PredicateUpdate { - returning: Some(_), .. - } - | KvOp::Delete { - returning: Some(_), .. - } - | KvOp::PredicateDelete { - returning: Some(_), .. - }, - ) => PlanKind::ReturningRows, - // KV delete/truncate count the keys removed — `Execution` would discard that. - PhysicalPlan::Kv(KvOp::Delete { .. }) | PhysicalPlan::Kv(KvOp::PredicateDelete { .. }) => { - DmlResult("DELETE") - } - // Reports `{"affected": n}` — `Execution` would discard that count. - // `FieldSet` is the keyed UPDATE, so it tags the same way. - PhysicalPlan::Kv(KvOp::FieldSet { .. }) | PhysicalPlan::Kv(KvOp::PredicateUpdate { .. }) => { - DmlResult("UPDATE") - } - PhysicalPlan::Kv(KvOp::Truncate { .. }) => DmlResult("TRUNCATE"), - - PhysicalPlan::Document(DocumentOp::InsertSelect { .. }) => DmlResult("INSERT"), - - PhysicalPlan::Document(DocumentOp::Upsert { - returning: Some(_), .. - }) => PlanKind::ReturningRows, - PhysicalPlan::Document(DocumentOp::Upsert { .. }) => DmlResult("UPSERT"), - - // Array read/maintenance ops produce a JSON-array payload; route to the - // multi-row decoder so each row streams as its own pgwire field. - PhysicalPlan::Array(nodedb_physical::physical_plan::ArrayOp::Slice { .. }) => { - PlanKind::ArraySlice - } - PhysicalPlan::Array(nodedb_physical::physical_plan::ArrayOp::Project { .. }) - | PhysicalPlan::Array(nodedb_physical::physical_plan::ArrayOp::Aggregate { .. }) - | PhysicalPlan::Array(nodedb_physical::physical_plan::ArrayOp::Elementwise { .. }) => { - PlanKind::MultiRow - } - // Flush/Compact return status JSON — route SingleDocument. - PhysicalPlan::Array(nodedb_physical::physical_plan::ArrayOp::Flush { .. }) - | PhysicalPlan::Array(nodedb_physical::physical_plan::ArrayOp::Compact { .. }) => { - PlanKind::SingleDocument - } - - // Vector write/config ops carry no row payload. Enumerated explicitly (not a - // `Vector(_)` wildcard) so a future read op can't silently strand its hits. - PhysicalPlan::Vector(VectorOp::Insert { .. }) - | PhysicalPlan::Vector(VectorOp::BatchInsert { .. }) - | PhysicalPlan::Vector(VectorOp::Delete { .. }) - | PhysicalPlan::Vector(VectorOp::DeleteBySurrogate { .. }) - | PhysicalPlan::Vector(VectorOp::SetParams { .. }) - | PhysicalPlan::Vector(VectorOp::QueryStats { .. }) - | PhysicalPlan::Vector(VectorOp::Seal { .. }) - | PhysicalPlan::Vector(VectorOp::CompactIndex { .. }) - | PhysicalPlan::Vector(VectorOp::Rebuild { .. }) - | PhysicalPlan::Vector(VectorOp::SparseInsert { .. }) - | PhysicalPlan::Vector(VectorOp::SparseDelete { .. }) - | PhysicalPlan::Vector(VectorOp::MultiVectorInsert { .. }) - | PhysicalPlan::Vector(VectorOp::MultiVectorDelete { .. }) - | PhysicalPlan::Vector(VectorOp::DirectUpsert { .. }) => PlanKind::Execution, - - // Document ops with no row payload. Enumerated explicitly, not a `Document(_)` - // wildcard — that let `Merge` default to unredacted passthrough. - PhysicalPlan::Document(DocumentOp::Register { .. }) - | PhysicalPlan::Document(DocumentOp::IndexLookup { .. }) - | PhysicalPlan::Document(DocumentOp::DropIndex { .. }) - | PhysicalPlan::Document(DocumentOp::BackfillIndex { .. }) - | PhysicalPlan::Document(DocumentOp::EstimateCount { .. }) - | PhysicalPlan::Document(DocumentOp::MaterializeScan { .. }) - // Read-only resolve: payload is the internal classification tuple, never a client row. - | PhysicalPlan::Document(DocumentOp::ResolveWrite(_)) - // A derived balance write answers no client — reports an affected count only. - | PhysicalPlan::Document(DocumentOp::ApplyBalanceDelta { .. }) - // Never reaches this classifier: write-resolve returns the response itself, - // shaped from the intercepted plan whose `returning` slot decides. - | PhysicalPlan::Document(DocumentOp::ResolvedWrite { .. }) - - // Default: opaque execution result. Exhaustive so a new variant forces a decision. - | PhysicalPlan::Graph(_) - | PhysicalPlan::Kv(_) - | PhysicalPlan::Columnar(_) - | PhysicalPlan::Timeseries(_) - | PhysicalPlan::Spatial(_) - | PhysicalPlan::Crdt(_) - | PhysicalPlan::Query(_) - | PhysicalPlan::Meta(_) - | PhysicalPlan::Array(_) - | PhysicalPlan::ClusterArray(_) - | PhysicalPlan::ClusterEvent(_) => PlanKind::Execution, - } -} - -// Bring the variant into scope for brevity in match arms above. -use PlanKind::DmlResult; - -#[cfg(test)] -mod tests { - use super::*; - use nodedb_types::{DatabaseId, QualifiedCollection}; - - #[test] - fn crdt_preview_is_an_opaque_execution_plan() { - let plan = PhysicalPlan::Crdt(CrdtOp::PreviewApply { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "tasks"), - document_id: "task-1".to_string(), - delta: vec![0x92, 0x01], - }); - - assert!(matches!(describe_plan(&plan), PlanKind::Execution)); - } - - fn merge_plan( - returning: Option, - ) -> PhysicalPlan { - PhysicalPlan::Document(DocumentOp::Merge { - target_collection: QualifiedCollection::new(DatabaseId::DEFAULT, "target"), - source_collection: QualifiedCollection::new(DatabaseId::DEFAULT, "source"), - source_alias: "s".to_string(), - target_join_col: "id".to_string(), - source_join_col: "id".to_string(), - clauses: Vec::new(), - returning, - resolved_inserts: None, - source_rows: None, - rls_filters: Vec::new(), - rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), - resolved_sum_targets: Vec::new(), - declared_primary_key: None, - }) - } - - /// A `MERGE ... RETURNING` payload is real target rows — `Execution` would - /// pass them unredacted. - #[test] - fn merge_with_returning_is_returning_rows() { - use nodedb_physical::physical_plan::{ReturningColumns, ReturningSpec}; - - let plan = merge_plan(Some(ReturningSpec { - columns: ReturningColumns::Star, - })); - - assert!(matches!(describe_plan(&plan), PlanKind::ReturningRows)); - } - - /// Every insert-family op with a projection must classify row-returning, - /// else it leaks unredacted like the MERGE case above. - #[test] - fn inserts_with_returning_are_returning_rows() { - use nodedb_physical::physical_plan::{ReturningColumns, ReturningSpec}; - - let spec = || { - Some(ReturningSpec { - columns: ReturningColumns::Star, - }) - }; - let plans = [ - PhysicalPlan::Document(DocumentOp::PointInsert { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), - document_id: "d".into(), - value: Vec::new(), - if_absent: false, - surrogate: nodedb_types::Surrogate::ZERO, - returning: spec(), - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - deferred_sum_targets: Vec::new(), - }), - PhysicalPlan::Document(DocumentOp::PointPut { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), - document_id: "d".into(), - value: Vec::new(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - returning: spec(), - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - }), - PhysicalPlan::Document(DocumentOp::BatchInsert { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), - documents: Vec::new(), - surrogates: Vec::new(), - returning: spec(), - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - deferred_sum_targets: Vec::new(), - }), - PhysicalPlan::Document(DocumentOp::Upsert { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), - document_id: "d".into(), - value: Vec::new(), - on_conflict_updates: Vec::new(), - surrogate: nodedb_types::Surrogate::ZERO, - rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), - returning: spec(), - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - }), - ]; - for plan in &plans { - assert!( - matches!(describe_plan(plan), PlanKind::ReturningRows), - "{plan:?} must shape as rows" - ); - } - } - - /// Every KV insert-family op that can carry a projection must classify as - /// row-returning too — the same passthrough leak, one engine over. - #[test] - fn kv_inserts_with_returning_are_returning_rows() { - use nodedb_physical::physical_plan::{KvOp, ReturningColumns, ReturningSpec}; - - let spec = || { - Some(ReturningSpec { - columns: ReturningColumns::Star, - }) - }; - let plans = [ - PhysicalPlan::Kv(KvOp::Insert { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), - key: b"k".to_vec(), - value: Vec::new(), - ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, - returning: spec(), - rls_filters: Vec::new(), - }), - PhysicalPlan::Kv(KvOp::InsertIfAbsent { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), - key: b"k".to_vec(), - value: Vec::new(), - ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, - returning: spec(), - rls_filters: Vec::new(), - }), - PhysicalPlan::Kv(KvOp::InsertOnConflictUpdate { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), - key: b"k".to_vec(), - value: Vec::new(), - ttl_ms: 0, - updates: Vec::new(), - surrogate: nodedb_types::Surrogate::ZERO, - rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), - returning: spec(), - rls_filters: Vec::new(), - }), - PhysicalPlan::Kv(KvOp::Put { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), - key: b"k".to_vec(), - value: Vec::new(), - ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, - returning: spec(), - rls_filters: Vec::new(), - }), - PhysicalPlan::Kv(KvOp::BatchPut { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), - entries: Vec::new(), - ttl_ms: 0, - surrogates: Vec::new(), - returning: spec(), - rls_filters: Vec::new(), - }), - ]; - for plan in &plans { - assert!( - matches!(describe_plan(plan), PlanKind::ReturningRows), - "{plan:?} must shape as rows" - ); - } - } - - /// A plain MERGE reports its affected count under the Postgres `MERGE` tag, - /// not an opaque `OK`. - #[test] - fn merge_without_returning_is_a_dml_result() { - assert!(matches!( - describe_plan(&merge_plan(None)), - PlanKind::DmlResult("MERGE") - )); - } -} diff --git a/nodedb/src/control/server/response_shape/types/plan_kind/array.rs b/nodedb/src/control/server/response_shape/types/plan_kind/array.rs new file mode 100644 index 000000000..23ec59f37 --- /dev/null +++ b/nodedb/src/control/server/response_shape/types/plan_kind/array.rs @@ -0,0 +1,33 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `ArrayOp` classification. + +use nodedb_physical::physical_plan::ArrayOp; + +use super::kind::PlanKind; + +pub(super) fn describe_array(op: &ArrayOp) -> PlanKind { + match op { + ArrayOp::Slice { .. } => PlanKind::ArraySlice, + + // JSON-array payloads: each row streams as its own pgwire field. + ArrayOp::Project { .. } | ArrayOp::Aggregate { .. } | ArrayOp::Elementwise { .. } => { + PlanKind::MultiRow + } + + // Reports `{"inserted": n}` / `{"deleted": n}`. + ArrayOp::Put { .. } => PlanKind::DmlResult("INSERT"), + ArrayOp::Delete { .. } => PlanKind::DmlResult("DELETE"), + + // Flush/Compact return status JSON — route SingleDocument. + ArrayOp::Flush { .. } | ArrayOp::Compact { .. } => PlanKind::SingleDocument, + + // Array DDL: `{"opened": 1}` / `{"dropped": 1}` status, not a row count. + ArrayOp::OpenArray { .. } + | ArrayOp::DropArray { .. } + | ArrayOp::RestoreArrayDrop { .. } + | ArrayOp::PurgeArrayDrop { .. } + // Internal roaring bitmap for cross-engine prefilter, never a client row. + | ArrayOp::SurrogateBitmapScan { .. } => PlanKind::Execution, + } +} diff --git a/nodedb/src/control/server/response_shape/types/plan_kind/columnar_family.rs b/nodedb/src/control/server/response_shape/types/plan_kind/columnar_family.rs new file mode 100644 index 000000000..ce09c67c3 --- /dev/null +++ b/nodedb/src/control/server/response_shape/types/plan_kind/columnar_family.rs @@ -0,0 +1,64 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `ColumnarOp`, `TimeseriesOp` and `SpatialOp` classification — the three +//! peer engines on the compressed-column storage core. + +use nodedb_physical::physical_plan::{ColumnarOp, SpatialOp, TimeseriesOp}; + +use super::kind::PlanKind; + +pub(super) fn describe_columnar(op: &ColumnarOp) -> PlanKind { + match op { + ColumnarOp::Scan { .. } => PlanKind::MultiRow, + + ColumnarOp::Insert { + returning: Some(_), .. + } => PlanKind::ReturningRows, + + // Reports `{"accepted": n}`. + ColumnarOp::Insert { .. } => PlanKind::DmlResult("INSERT"), + + // Reports `{"affected": n}`. + ColumnarOp::Update { .. } => PlanKind::DmlResult("UPDATE"), + ColumnarOp::Delete { .. } => PlanKind::DmlResult("DELETE"), + // Reports `{"truncated": n}`. + ColumnarOp::Truncate { .. } => PlanKind::DmlResult("TRUNCATE"), + + // Never reach this classifier: write-resolve proposes them and returns + // the response itself, shaped from the intercepted `Update` / `Delete`. + ColumnarOp::ResolvedUpdate { .. } + | ColumnarOp::ResolvedDelete { .. } + // Read-only resolve: payload is the internal row set, never a client row. + | ColumnarOp::ResolveDml { .. } + // Clone materializer payload (`[cursor, entries]`), decoded by its caller. + | ColumnarOp::MaterializeScan { .. } => PlanKind::Execution, + } +} + +pub(super) fn describe_timeseries(op: &TimeseriesOp) -> PlanKind { + match op { + TimeseriesOp::Scan { .. } => PlanKind::MultiRow, + + TimeseriesOp::Ingest { + returning: Some(_), .. + } => PlanKind::ReturningRows, + + // Reports `{"accepted": n}`. + TimeseriesOp::Ingest { .. } => PlanKind::DmlResult("INSERT"), + // Reports `{"truncated": n}`. + TimeseriesOp::Truncate { .. } => PlanKind::DmlResult("TRUNCATE"), + + // Read-only resolve: payload is the internal admission verdict, never a client row. + TimeseriesOp::ResolveIngest(_) => PlanKind::Execution, + } +} + +pub(super) fn describe_spatial(op: &SpatialOp) -> PlanKind { + match op { + SpatialOp::Scan { .. } => PlanKind::MultiRow, + + // Replication-apply plans (sync inbound, WAL dispatch, Raft apply). + // A client statement never lowers to them, so they answer no client. + SpatialOp::Insert { .. } | SpatialOp::Delete { .. } => PlanKind::Execution, + } +} diff --git a/nodedb/src/control/server/response_shape/types/plan_kind/crdt.rs b/nodedb/src/control/server/response_shape/types/plan_kind/crdt.rs new file mode 100644 index 000000000..8d33cbd31 --- /dev/null +++ b/nodedb/src/control/server/response_shape/types/plan_kind/crdt.rs @@ -0,0 +1,55 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `CrdtOp` classification. + +use nodedb_physical::physical_plan::CrdtOp; + +use super::kind::PlanKind; + +pub(super) fn describe_crdt(op: &CrdtOp) -> PlanKind { + match op { + CrdtOp::DocUpsert { + returning: Some(_), .. + } + | CrdtOp::DocDelete { + returning: Some(_), .. + } => PlanKind::ReturningRows, + + // INSERT, UPSERT and UPDATE all lower to `DocUpsert`; the verb the + // statement used decides the tag. + CrdtOp::DocUpsert { verb, .. } => PlanKind::DmlResult(verb.command_tag()), + + // A CRDT delete can legitimately remove nothing, so its count must render + // as a DML count from the write's own response, not a document-shaped read. + CrdtOp::DocDelete { .. } => PlanKind::DmlResult("DELETE"), + + // One document body, or one policy object. + CrdtOp::Read { .. } | CrdtOp::ReadAtVersion { .. } | CrdtOp::GetPolicy { .. } => { + PlanKind::SingleDocument + } + + // Delta application and snapshot import: sync/replication writes with + // no row count. + CrdtOp::Apply { .. } + | CrdtOp::ApplyAuthenticated { .. } + | CrdtOp::ImportSnapshot { .. } + // Constraint and policy DDL. + | CrdtOp::SetConstraints { .. } + | CrdtOp::DropConstraints { .. } + | CrdtOp::SetPolicy { .. } + // History maintenance. + | CrdtOp::RestoreToVersion { .. } + | CrdtOp::CompactAtVersion { .. } + // Block-list edits: the handler reports no count. + | CrdtOp::ListInsert { .. } + | CrdtOp::ListDelete { .. } + | CrdtOp::ListMove { .. } + // Internal typed zerompk payloads, decoded by their own dispatcher: + // the installed constraint set, a version vector, a Loro delta, and + // the admission caller's `CrdtPreviewResult`. + | CrdtOp::ReadConstraints { .. } + | CrdtOp::GetVersionVector { .. } + | CrdtOp::ExportDelta { .. } + | CrdtOp::PreviewApply { .. } => PlanKind::Execution, + } +} diff --git a/nodedb/src/control/server/response_shape/types/plan_kind/describe.rs b/nodedb/src/control/server/response_shape/types/plan_kind/describe.rs new file mode 100644 index 000000000..7c51c000e --- /dev/null +++ b/nodedb/src/control/server/response_shape/types/plan_kind/describe.rs @@ -0,0 +1,342 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `describe_plan`: the one entry point that maps a `PhysicalPlan` to the +//! response shape it produces. Each engine's `*Op` enum is classified in its +//! own file, exhaustively, so a new op is a compile error until it decides. + +use crate::bridge::envelope::PhysicalPlan; + +use super::array::describe_array; +use super::columnar_family::{describe_columnar, describe_spatial, describe_timeseries}; +use super::crdt::describe_crdt; +use super::document::describe_document; +use super::graph::describe_graph; +use super::kind::PlanKind; +use super::kv::describe_kv; +use super::query::describe_query; +use super::search::{describe_text, describe_vector}; + +pub fn describe_plan(plan: &PhysicalPlan) -> PlanKind { + match plan { + PhysicalPlan::Document(op) => describe_document(op), + PhysicalPlan::Kv(op) => describe_kv(op), + PhysicalPlan::Crdt(op) => describe_crdt(op), + PhysicalPlan::Graph(op) => describe_graph(op), + PhysicalPlan::Vector(op) => describe_vector(op), + PhysicalPlan::Text(op) => describe_text(op), + PhysicalPlan::Columnar(op) => describe_columnar(op), + PhysicalPlan::Timeseries(op) => describe_timeseries(op), + PhysicalPlan::Spatial(op) => describe_spatial(op), + PhysicalPlan::Array(op) => describe_array(op), + PhysicalPlan::Query(op) => describe_query(op), + + // Control-plane catalog, session and cluster ops. No `MetaOp` is a + // client DML: none reports a row count, each answers its own caller. + PhysicalPlan::Meta(_) + // Never dispatched through the plan-shaping path: the pgwire cluster + // array router classifies these itself (`routing/cluster_array.rs`). + | PhysicalPlan::ClusterArray(_) + // Event-plane forwarding, answered by its own dispatcher. + | PhysicalPlan::ClusterEvent(_) => PlanKind::Execution, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_physical::physical_plan::{ + ArrayOp, CrdtOp, CrdtWriteVerb, DocumentOp, KvOp, TimeseriesOp, + }; + use nodedb_types::{DatabaseId, QualifiedCollection}; + + #[test] + fn crdt_preview_is_an_opaque_execution_plan() { + let plan = PhysicalPlan::Crdt(CrdtOp::PreviewApply { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "tasks"), + document_id: "task-1".to_string(), + delta: vec![0x92, 0x01], + }); + + assert!(matches!(describe_plan(&plan), PlanKind::Execution)); + } + + fn merge_plan( + returning: Option, + ) -> PhysicalPlan { + PhysicalPlan::Document(DocumentOp::Merge { + target_collection: QualifiedCollection::new(DatabaseId::DEFAULT, "target"), + source_collection: QualifiedCollection::new(DatabaseId::DEFAULT, "source"), + source_alias: "s".to_string(), + target_join_col: "id".to_string(), + source_join_col: "id".to_string(), + clauses: Vec::new(), + returning, + resolved_inserts: None, + resolved_insert_identities: Vec::new(), + source_rows: None, + rls_filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + resolved_sum_targets: Vec::new(), + declared_primary_key: None, + }) + } + + /// A `MERGE ... RETURNING` payload is real target rows — `Execution` would + /// pass them unredacted. + #[test] + fn merge_with_returning_is_returning_rows() { + use nodedb_physical::physical_plan::{ReturningColumns, ReturningSpec}; + + let plan = merge_plan(Some(ReturningSpec { + columns: ReturningColumns::Star, + })); + + assert!(matches!(describe_plan(&plan), PlanKind::ReturningRows)); + } + + /// Every insert-family op with a projection must classify row-returning, + /// else it leaks unredacted like the MERGE case above. + #[test] + fn inserts_with_returning_are_returning_rows() { + use nodedb_physical::physical_plan::{ReturningColumns, ReturningSpec}; + + let spec = || { + Some(ReturningSpec { + columns: ReturningColumns::Star, + }) + }; + let plans = [ + PhysicalPlan::Document(DocumentOp::PointInsert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), + document_id: "d".into(), + value: Vec::new(), + if_absent: false, + surrogate: nodedb_types::Surrogate::ZERO, + returning: spec(), + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + deferred_sum_targets: Vec::new(), + }), + PhysicalPlan::Document(DocumentOp::PointPut { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), + document_id: "d".into(), + value: Vec::new(), + surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: Vec::new(), + returning: spec(), + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + }), + PhysicalPlan::Document(DocumentOp::BatchInsert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), + documents: Vec::new(), + surrogates: Vec::new(), + returning: spec(), + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + deferred_sum_targets: Vec::new(), + }), + PhysicalPlan::Document(DocumentOp::Upsert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), + document_id: "d".into(), + value: Vec::new(), + on_conflict_updates: Vec::new(), + surrogate: nodedb_types::Surrogate::ZERO, + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: spec(), + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + }), + ]; + for plan in &plans { + assert!( + matches!(describe_plan(plan), PlanKind::ReturningRows), + "{plan:?} must shape as rows" + ); + } + } + + /// Every KV insert-family op that can carry a projection must classify as + /// row-returning too — the same passthrough leak, one engine over. + #[test] + fn kv_inserts_with_returning_are_returning_rows() { + use nodedb_physical::physical_plan::{ReturningColumns, ReturningSpec}; + + let spec = || { + Some(ReturningSpec { + columns: ReturningColumns::Star, + }) + }; + let plans = [ + PhysicalPlan::Kv(KvOp::Insert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), + key: b"k".to_vec(), + value: Vec::new(), + ttl_ms: 0, + surrogate: nodedb_types::Surrogate::ZERO, + returning: spec(), + rls_filters: Vec::new(), + }), + PhysicalPlan::Kv(KvOp::InsertIfAbsent { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), + key: b"k".to_vec(), + value: Vec::new(), + ttl_ms: 0, + surrogate: nodedb_types::Surrogate::ZERO, + returning: spec(), + rls_filters: Vec::new(), + }), + PhysicalPlan::Kv(KvOp::InsertOnConflictUpdate { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), + key: b"k".to_vec(), + value: Vec::new(), + ttl_ms: 0, + updates: Vec::new(), + surrogate: nodedb_types::Surrogate::ZERO, + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: spec(), + rls_filters: Vec::new(), + }), + PhysicalPlan::Kv(KvOp::Put { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), + key: b"k".to_vec(), + value: Vec::new(), + ttl_ms: 0, + surrogate: nodedb_types::Surrogate::ZERO, + returning: spec(), + rls_filters: Vec::new(), + }), + PhysicalPlan::Kv(KvOp::BatchPut { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), + entries: Vec::new(), + ttl_ms: 0, + surrogates: Vec::new(), + returning: spec(), + rls_filters: Vec::new(), + }), + ]; + for plan in &plans { + assert!( + matches!(describe_plan(plan), PlanKind::ReturningRows), + "{plan:?} must shape as rows" + ); + } + } + + /// A plain MERGE reports its affected count under the Postgres `MERGE` tag, + /// not an opaque `OK`. + #[test] + fn merge_without_returning_is_a_dml_result() { + assert!(matches!( + describe_plan(&merge_plan(None)), + PlanKind::DmlResult("MERGE") + )); + } + + fn kv_put() -> PhysicalPlan { + PhysicalPlan::Kv(KvOp::Put { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), + key: b"k".to_vec(), + value: Vec::new(), + ttl_ms: 0, + surrogate: nodedb_types::Surrogate::ZERO, + returning: None, + rls_filters: Vec::new(), + }) + } + + /// `KvOp::Put` is the SQL `UPSERT` statement: it tags `UPSERT n`, the + /// same as `DocumentOp::Upsert`, the staged path and the Calvin fold. + #[test] + fn kv_put_is_the_upsert_dml_result() { + assert!(matches!( + describe_plan(&kv_put()), + PlanKind::DmlResult("UPSERT") + )); + } + + /// `KvOp::InsertOnConflictUpdate` resolves insert-vs-update at apply time; + /// the tag must follow the verb the handler reports. + #[test] + fn kv_insert_on_conflict_update_is_decided_by_op() { + let plan = PhysicalPlan::Kv(KvOp::InsertOnConflictUpdate { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), + key: b"k".to_vec(), + value: Vec::new(), + ttl_ms: 0, + updates: Vec::new(), + surrogate: nodedb_types::Surrogate::ZERO, + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: None, + rls_filters: Vec::new(), + }); + assert!(matches!(describe_plan(&plan), PlanKind::DmlResultByOp)); + } + + /// A timeseries ingest reports `{"accepted": n}` under the `INSERT` tag, + /// not an opaque `OK`. + #[test] + fn timeseries_ingest_is_an_insert_dml_result() { + let plan = PhysicalPlan::Timeseries(TimeseriesOp::Ingest { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "metrics"), + payload: Vec::new(), + format: "ilp".to_string(), + wal_lsn: None, + surrogates: Vec::new(), + provenance: None, + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: None, + rls_filters: Vec::new(), + }); + assert!(matches!( + describe_plan(&plan), + PlanKind::DmlResult("INSERT") + )); + } + + fn crdt_doc_upsert(verb: CrdtWriteVerb) -> PhysicalPlan { + PhysicalPlan::Crdt(CrdtOp::DocUpsert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "notes"), + document_id: "d1".into(), + fields_json: "{}".into(), + surrogate: nodedb_types::Surrogate::ZERO, + partial: matches!(verb, CrdtWriteVerb::Update), + verb, + returning: None, + rls_filters: Vec::new(), + }) + } + + /// INSERT, UPSERT and UPDATE all lower to `CrdtOp::DocUpsert`; the tag + /// follows the statement verb carried on the op. + #[test] + fn crdt_doc_upsert_tags_by_verb() { + assert!(matches!( + describe_plan(&crdt_doc_upsert(CrdtWriteVerb::Insert)), + PlanKind::DmlResult("INSERT") + )); + assert!(matches!( + describe_plan(&crdt_doc_upsert(CrdtWriteVerb::Upsert)), + PlanKind::DmlResult("UPSERT") + )); + assert!(matches!( + describe_plan(&crdt_doc_upsert(CrdtWriteVerb::Update)), + PlanKind::DmlResult("UPDATE") + )); + } + + /// `INSERT INTO ARRAY` reports `{"inserted": n}` under the `INSERT` tag. + #[test] + fn array_put_is_an_insert_dml_result() { + let plan = PhysicalPlan::Array(ArrayOp::Put { + array_id: nodedb_array::types::ArrayId::new(nodedb_types::TenantId::new(1), "genome"), + cells_msgpack: Vec::new(), + wal_lsn: 0, + provenance: None, + }); + assert!(matches!( + describe_plan(&plan), + PlanKind::DmlResult("INSERT") + )); + } +} diff --git a/nodedb/src/control/server/response_shape/types/plan_kind/document.rs b/nodedb/src/control/server/response_shape/types/plan_kind/document.rs new file mode 100644 index 000000000..963491c8b --- /dev/null +++ b/nodedb/src/control/server/response_shape/types/plan_kind/document.rs @@ -0,0 +1,88 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `DocumentOp` classification. + +use nodedb_physical::physical_plan::DocumentOp; + +use super::kind::PlanKind; + +pub(super) fn describe_document(op: &DocumentOp) -> PlanKind { + match op { + DocumentOp::PointGet { .. } => PlanKind::SingleDocument, + + DocumentOp::RangeScan { .. } + | DocumentOp::Scan { .. } + | DocumentOp::IndexedFetch { .. } => PlanKind::MultiRow, + + // A write with a projection returns real stored rows and must be + // decoded and redacted, never passed through unshaped. + DocumentOp::PointPut { + returning: Some(_), .. + } + | DocumentOp::PointInsert { + returning: Some(_), .. + } + | DocumentOp::BatchInsert { + returning: Some(_), .. + } + | DocumentOp::PointUpdate { + returning: Some(_), .. + } + | DocumentOp::BulkUpdate { + returning: Some(_), .. + } + | DocumentOp::PointDelete { + returning: Some(_), .. + } + | DocumentOp::BulkDelete { + returning: Some(_), .. + } + | DocumentOp::UpdateFromJoin { + returning: Some(_), .. + } + | DocumentOp::Merge { + returning: Some(_), .. + } + | DocumentOp::Upsert { + returning: Some(_), .. + } => PlanKind::ReturningRows, + + // `PointInsert`: `ON CONFLICT DO NOTHING` makes it no-op-capable, so + // the count must come from the write's response. + DocumentOp::PointPut { .. } + | DocumentOp::PointInsert { .. } + | DocumentOp::BatchInsert { .. } + | DocumentOp::InsertSelect { .. } => PlanKind::DmlResult("INSERT"), + + DocumentOp::PointUpdate { .. } + | DocumentOp::BulkUpdate { .. } + | DocumentOp::UpdateFromJoin { .. } => PlanKind::DmlResult("UPDATE"), + + DocumentOp::PointDelete { .. } | DocumentOp::BulkDelete { .. } => { + PlanKind::DmlResult("DELETE") + } + + // Postgres tags a plain MERGE `MERGE `, matching the staged path. + DocumentOp::Merge { .. } => PlanKind::DmlResult("MERGE"), + + DocumentOp::Truncate { .. } => PlanKind::DmlResult("TRUNCATE"), + + DocumentOp::Upsert { .. } => PlanKind::DmlResult("UPSERT"), + + // Index DDL and catalog maintenance: no row payload, no row count. + DocumentOp::Register { .. } + | DocumentOp::IndexLookup { .. } + | DocumentOp::DropIndex { .. } + | DocumentOp::BackfillIndex { .. } + | DocumentOp::EstimateCount { .. } + // Clone materializer payload (`[cursor, entries]`), decoded by its caller. + | DocumentOp::MaterializeScan { .. } + // Read-only resolve: payload is the internal classification tuple, never a client row. + | DocumentOp::ResolveWrite(_) + // A derived balance write answers no client — reports an affected count only. + | DocumentOp::ApplyBalanceDelta { .. } + // Never reaches this classifier: write-resolve returns the response itself, + // shaped from the intercepted plan whose `returning` slot decides. + | DocumentOp::ResolvedWrite { .. } => PlanKind::Execution, + } +} diff --git a/nodedb/src/control/server/response_shape/types/plan_kind/graph.rs b/nodedb/src/control/server/response_shape/types/plan_kind/graph.rs new file mode 100644 index 000000000..10489a26a --- /dev/null +++ b/nodedb/src/control/server/response_shape/types/plan_kind/graph.rs @@ -0,0 +1,42 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `GraphOp` classification. + +use nodedb_physical::physical_plan::GraphOp; + +use super::kind::PlanKind; + +pub(super) fn describe_graph(op: &GraphOp) -> PlanKind { + match op { + // Traversals, pattern matches, algorithms and stats all return one + // row per hit / node / collection. + GraphOp::Hop { .. } + | GraphOp::Neighbors { .. } + | GraphOp::NeighborsMulti { .. } + | GraphOp::Path { .. } + | GraphOp::Subgraph { .. } + | GraphOp::RagFusion { .. } + | GraphOp::Algo { .. } + | GraphOp::Match { .. } + | GraphOp::MatchContinuation { .. } + | GraphOp::MatchVarLenResume { .. } + | GraphOp::BspSuperstep(_) + | GraphOp::WccSuperstep(_) + | GraphOp::TemporalNeighbors { .. } + | GraphOp::TemporalAlgorithm { .. } + | GraphOp::Stats { .. } => PlanKind::MultiRow, + + GraphOp::EdgePut { .. } | GraphOp::EdgePutBatch { .. } => PlanKind::DmlResult("INSERT"), + + // `ResolveEdgeDelete` reports the same live/absent verdict as the + // delete it wraps, via `response_affected` — matches an edge delete's + // tag even though the resolve pass itself writes nothing. + GraphOp::EdgeDelete { .. } + | GraphOp::EdgeDeleteBatch { .. } + | GraphOp::ResolveEdgeDelete(_) => PlanKind::DmlResult("DELETE"), + + GraphOp::SetNodeLabels { .. } | GraphOp::RemoveNodeLabels { .. } => { + PlanKind::DmlResult("UPDATE") + } + } +} diff --git a/nodedb/src/control/server/response_shape/types/plan_kind/kind.rs b/nodedb/src/control/server/response_shape/types/plan_kind/kind.rs new file mode 100644 index 000000000..3246839a4 --- /dev/null +++ b/nodedb/src/control/server/response_shape/types/plan_kind/kind.rs @@ -0,0 +1,28 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The response shape a physical plan produces. +//! +//! Carries no pgwire wire types, so every protocol-specific shaper shares it. + +#[derive(Debug, Clone, Copy)] +pub enum PlanKind { + SingleDocument, + MultiRow, + /// Array slice result — decoded via `ArraySliceResponse` to surface the + /// `truncated_before_horizon` flag as a pgwire NOTICE when set. + ArraySlice, + /// Opaque execution result: DDL, maintenance, an internal stage, or a + /// function-call payload its dispatcher reads directly. pgwire renders a + /// bare `OK` tag. Never a client-facing row-count DML. + Execution, + /// DML operation that returns affected row count. + /// The tag name is used in the pgwire `CommandComplete` message (e.g., "UPDATE", "DELETE"). + DmlResult(&'static str), + /// DML whose verb is decided by the handler at apply time: the payload + /// carries `affected` plus `op` (`"insert"` or `"update"`), read via + /// `extract_kv_conflict_op`. Renders `INSERT 0 n` or `UPDATE n`. + DmlResultByOp, + /// DML with RETURNING clause — payload is a `RowsPayload` (msgpack). + /// Decoded into one pgwire field per column. + ReturningRows, +} diff --git a/nodedb/src/control/server/response_shape/types/plan_kind/kv.rs b/nodedb/src/control/server/response_shape/types/plan_kind/kv.rs new file mode 100644 index 000000000..12a3baf03 --- /dev/null +++ b/nodedb/src/control/server/response_shape/types/plan_kind/kv.rs @@ -0,0 +1,99 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `KvOp` classification. + +use nodedb_physical::physical_plan::KvOp; + +use super::kind::PlanKind; + +pub(super) fn describe_kv(op: &KvOp) -> PlanKind { + match op { + KvOp::Get { .. } | KvOp::FieldGet { .. } => PlanKind::SingleDocument, + + KvOp::Scan { .. } | KvOp::BatchGet { .. } => PlanKind::MultiRow, + + // A write with a projection returns real stored rows and must be + // decoded and redacted, never passed through unshaped. + KvOp::Insert { + returning: Some(_), .. + } + | KvOp::InsertIfAbsent { + returning: Some(_), .. + } + | KvOp::InsertOnConflictUpdate { + returning: Some(_), .. + } + | KvOp::Put { + returning: Some(_), .. + } + | KvOp::BatchPut { + returning: Some(_), .. + } + | KvOp::FieldSet { + returning: Some(_), .. + } + | KvOp::PredicateUpdate { + returning: Some(_), .. + } + | KvOp::Delete { + returning: Some(_), .. + } + | KvOp::PredicateDelete { + returning: Some(_), .. + } => PlanKind::ReturningRows, + + // The SQL `UPSERT` statement, tagged like `DocumentOp::Upsert`. + KvOp::Put { .. } => PlanKind::DmlResult("UPSERT"), + + // `InsertIfAbsent`: `ON CONFLICT DO NOTHING` makes it no-op-capable, + // so the count must come from the write's response. + KvOp::Insert { .. } | KvOp::InsertIfAbsent { .. } | KvOp::BatchPut { .. } => { + PlanKind::DmlResult("INSERT") + } + + // Insert-vs-update is decided by the handler; the payload says which. + KvOp::InsertOnConflictUpdate { .. } => PlanKind::DmlResultByOp, + + // Reports `{"affected": n}`. `FieldSet` is the keyed UPDATE. + KvOp::FieldSet { .. } | KvOp::PredicateUpdate { .. } => PlanKind::DmlResult("UPDATE"), + + // Counts the keys removed. + KvOp::Delete { .. } | KvOp::PredicateDelete { .. } => PlanKind::DmlResult("DELETE"), + + KvOp::Truncate { .. } => PlanKind::DmlResult("TRUNCATE"), + + // One-object payloads: `{"ttl_ms": n}`, `{"rank": n}`, `{"count": n}`, + // `{"score": ..}`. + KvOp::GetTtl { .. } + | KvOp::SortedIndexRank { .. } + | KvOp::SortedIndexCount { .. } + | KvOp::SortedIndexScore { .. } => PlanKind::SingleDocument, + + // One row per sorted-index entry. + KvOp::SortedIndexTopK { .. } | KvOp::SortedIndexRange { .. } => PlanKind::MultiRow, + + // TTL metadata mutations: no row count. + KvOp::Expire { .. } + | KvOp::Persist { .. } + // Index DDL. + | KvOp::RegisterIndex { .. } + | KvOp::DropIndex { .. } + | KvOp::RegisterSortedIndex { .. } + | KvOp::DropSortedIndex { .. } + // Function-call results (`KV_INCR(..)` and friends): the payload is a + // computed value its dispatcher reads directly, not a row count. + | KvOp::Incr { .. } + | KvOp::IncrFloat { .. } + | KvOp::Cas { .. } + | KvOp::GetSet { .. } + | KvOp::Transfer { .. } + | KvOp::TransferItem { .. } + // Clone materializer payload (`[cursor, entries]`), decoded by its caller. + | KvOp::MaterializeScan { .. } + // Read-only resolve: payload is the internal mutation list, never a client row. + | KvOp::ResolveWrite(_) + // Never reaches this classifier: write-resolve returns the response itself, + // shaped from the intercepted plan whose `returning` slot decides. + | KvOp::ResolvedWrite { .. } => PlanKind::Execution, + } +} diff --git a/nodedb/src/control/server/response_shape/types/plan_kind/mod.rs b/nodedb/src/control/server/response_shape/types/plan_kind/mod.rs new file mode 100644 index 000000000..07757bacf --- /dev/null +++ b/nodedb/src/control/server/response_shape/types/plan_kind/mod.rs @@ -0,0 +1,20 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Protocol-neutral plan classification. +//! +//! One file per engine so every `*Op` enum is matched exhaustively in one +//! place and a new variant fails to compile until it is classified. + +mod array; +mod columnar_family; +mod crdt; +mod describe; +mod document; +mod graph; +mod kind; +mod kv; +mod query; +mod search; + +pub use describe::describe_plan; +pub use kind::PlanKind; diff --git a/nodedb/src/control/server/response_shape/types/plan_kind/query.rs b/nodedb/src/control/server/response_shape/types/plan_kind/query.rs new file mode 100644 index 000000000..23c33e958 --- /dev/null +++ b/nodedb/src/control/server/response_shape/types/plan_kind/query.rs @@ -0,0 +1,40 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `QueryOp` classification. + +use nodedb_physical::physical_plan::QueryOp; + +use super::describe::describe_plan; +use super::kind::PlanKind; + +pub(super) fn describe_query(op: &QueryOp) -> PlanKind { + match op { + // Exchange means the plan wasn't yet resolved — recurse into the child. + QueryOp::Exchange(exchange) => describe_plan(&exchange.child), + + // PostProcess reshapes a multi-row subquery; its kind is the child's. + QueryOp::PostProcess { input, .. } => describe_plan(input), + + // Constant/catalog-scan expressions compile to ProviderScan, and a + // SetOp resolves to a ProviderScan of merged rows: each element + // streams as its own pgwire row. + QueryOp::ProviderScan { .. } + | QueryOp::SetOp { .. } + | QueryOp::Aggregate { .. } + | QueryOp::FacetCounts { .. } + | QueryOp::HashJoin { .. } + | QueryOp::NestedLoopJoin { .. } + | QueryOp::SortMergeJoin { .. } + | QueryOp::RecursiveScan { .. } + | QueryOp::RecursiveValue { .. } + | QueryOp::LateralTopK { .. } + | QueryOp::LateralLoop { .. } => PlanKind::MultiRow, + + // Intra-plan stages: their payload feeds the next stage of the same + // plan and never reaches a client. + QueryOp::PartialAggregate { .. } + | QueryOp::PartialAggregateState { .. } + | QueryOp::ShuffleJoinConsume { .. } + | QueryOp::ShuffleAggregateConsume { .. } => PlanKind::Execution, + } +} diff --git a/nodedb/src/control/server/response_shape/types/plan_kind/search.rs b/nodedb/src/control/server/response_shape/types/plan_kind/search.rs new file mode 100644 index 000000000..341bb8aac --- /dev/null +++ b/nodedb/src/control/server/response_shape/types/plan_kind/search.rs @@ -0,0 +1,85 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `VectorOp` and `TextOp` classification. + +use nodedb_physical::physical_plan::{TextOp, VectorOp}; + +use super::kind::PlanKind; + +pub(super) fn describe_vector(op: &VectorOp) -> PlanKind { + match op { + VectorOp::DirectUpsert { + returning: Some(_), .. + } + | VectorOp::DirectInsert { + returning: Some(_), .. + } + | VectorOp::DirectInsertIfAbsent { + returning: Some(_), .. + } + | VectorOp::DirectDelete { + returning: Some(_), .. + } + | VectorOp::DirectUpdate { + returning: Some(_), .. + } => PlanKind::ReturningRows, + // The vector-primary write family: each handler reports its affected + // row count. An `ON CONFLICT DO UPDATE` upsert reports the verb it + // applied, exactly as `KvOp::InsertOnConflictUpdate` does. + VectorOp::DirectInsert { .. } | VectorOp::DirectInsertIfAbsent { .. } => { + PlanKind::DmlResult("INSERT") + } + VectorOp::DirectUpsert { + on_conflict_updates, + .. + } if !on_conflict_updates.is_empty() => PlanKind::DmlResultByOp, + VectorOp::DirectUpsert { .. } => PlanKind::DmlResult("UPSERT"), + VectorOp::DirectDelete { .. } => PlanKind::DmlResult("DELETE"), + VectorOp::DirectTruncate { .. } => PlanKind::DmlResult("TRUNCATE"), + VectorOp::DirectUpdate { .. } => PlanKind::DmlResult("UPDATE"), + + // Read-only resolve: payload is the internal mutation list, never a client row. + VectorOp::ResolveDirectWrite(_) + // Never reaches this classifier: write-resolve returns the response itself, + // shaped from the intercepted plan whose `returning` slot decides. + | VectorOp::ResolvedDirectWrite { .. } => PlanKind::Execution, + + VectorOp::Search { .. } + | VectorOp::MultiSearch { .. } + | VectorOp::MultiVectorScoreSearch { .. } + | VectorOp::SparseSearch { .. } => PlanKind::MultiRow, + + // Index-maintenance and config ops: none is a client DML statement, + // and none carries a row payload. Enumerated explicitly so a future + // read op can't silently strand its hits. + VectorOp::Insert { .. } + | VectorOp::BatchInsert { .. } + | VectorOp::Delete { .. } + | VectorOp::DeleteBySurrogate { .. } + | VectorOp::SetParams { .. } + | VectorOp::DropIndex { .. } + | VectorOp::QueryStats { .. } + | VectorOp::Seal { .. } + | VectorOp::CompactIndex { .. } + | VectorOp::Rebuild { .. } + | VectorOp::SparseInsert { .. } + | VectorOp::SparseDelete { .. } + | VectorOp::MultiVectorInsert { .. } + | VectorOp::MultiVectorDelete { .. } => PlanKind::Execution, + } +} + +pub(super) fn describe_text(op: &TextOp) -> PlanKind { + match op { + TextOp::Search { .. } + | TextOp::PhraseSearch { .. } + | TextOp::HybridSearch { .. } + | TextOp::HybridSearchTriple { .. } + | TextOp::BM25ScoreScan { .. } + | TextOp::FtsIndexDoc { .. } + | TextOp::FtsDeleteDoc { .. } => PlanKind::MultiRow, + + // Config write: opaque status. + TextOp::SetTextConfig { .. } => PlanKind::Execution, + } +} diff --git a/nodedb/src/control/server/shared/cluster_array_dispatch.rs b/nodedb/src/control/server/shared/cluster_array_dispatch.rs new file mode 100644 index 000000000..527cdebd5 --- /dev/null +++ b/nodedb/src/control/server/shared/cluster_array_dispatch.rs @@ -0,0 +1,163 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Protocol-neutral `ClusterArray` plan dispatch, shared by pgwire and native. +//! +//! `ClusterArrayOp` plans are handled entirely on the Control Plane by the +//! `ArrayCoordinator` — they must never reach the SPSC bridge or the +//! trigger/DML machinery. Each protocol's dispatch loop intercepts a +//! `PhysicalPlan::ClusterArray` task right after its own in-transaction +//! routing gate, calls [`execute_cluster_array`], then renders the +//! [`ClusterArrayShaped`] outcome in its own wire format. pgwire's adapter +//! lives in `pgwire::handler::routing::cluster_array`; native's lives in +//! `native::dispatch::cluster_array`. + +use std::sync::Arc; + +use nodedb_physical::physical_plan::{ClusterArrayOp, PhysicalPlan}; + +use crate::control::cluster::ClusterArrayExecutor; +use crate::control::security::auth_context::AuthContext; +use crate::control::server::dispatch_utils::publish_cluster_array_change_events; +use crate::control::server::response_shape::compose::{self, ShapeOutcome}; +use crate::control::server::response_shape::redaction::QueryRedaction; +use crate::control::server::response_shape::schema::OutputSchema; +use crate::control::server::response_shape::types::{DmlOutcome, PlanKind, ShapedRows}; +use crate::control::server::shared::authorization::AuthorizedTask; +use crate::control::server::shared::sql::staging_predicates::require_affected_count; +use crate::control::state::SharedState; + +/// What one `ClusterArrayOp` answers with, before any protocol renders it. +pub(crate) enum ClusterArrayShaped { + /// A read's rows (`Slice` / `Agg`). Any carried client-facing notice + /// lives on `ShapedRows::notice`. + Rows(ShapedRows), + /// A write's count (`Put` / `Delete`), with the verb the tag names it. + Affected(DmlOutcome), +} + +/// Execute one authorized `ClusterArrayOp` via the `ArrayCoordinator` and +/// shape its payload into protocol-neutral rows or a count-bearing outcome. +/// +/// On a successful `Put`/`Delete` (writes; `Slice`/`Agg` are reads and +/// publish nothing), publishes a CDC change event keyed by the op's own +/// `wal_lsn` — this path never touches the SPSC bridge, so there is no +/// Data-Plane `Response::watermark_lsn` to read the LSN from the way the +/// normal dispatch funnel does (see `publish_cluster_array_change_events`'s +/// own doc comment). +pub(crate) async fn execute_cluster_array( + state: &Arc, + auth: &AuthContext, + authorized: AuthorizedTask, + projection: Option<&OutputSchema>, +) -> crate::Result { + // Read before the task is consumed: an in-transaction `Slice`/`Agg` + // carries the session's transaction id, and each shard folds that + // transaction's staged cells into its result. + let txn_id = authorized.txn_id(); + let task = authorized.into_physical_task(); + let tenant_id = task.tenant_id; + let database_id = task.database_id; + let PhysicalPlan::ClusterArray(cluster_op) = task.plan else { + return Err(crate::Error::Internal { + detail: "authorized task is not a ClusterArray operation".to_owned(), + }); + }; + + let transport = state + .cluster_transport + .as_ref() + .ok_or_else(|| crate::Error::Internal { + detail: "cluster transport not available for ClusterArray dispatch".to_owned(), + })?; + let routing = state + .cluster_routing + .as_ref() + .ok_or_else(|| crate::Error::Internal { + detail: "cluster routing not available for ClusterArray dispatch".to_owned(), + })?; + let executor = ClusterArrayExecutor::new( + Arc::clone(transport), + Arc::clone(routing), + state.node_id, + Arc::clone(state), + ); + let op = match &cluster_op { + ClusterArrayOp::Slice { .. } => "slice", + ClusterArrayOp::Agg { .. } => "agg", + ClusterArrayOp::Put { .. } => "put", + ClusterArrayOp::Delete { .. } => "delete", + }; + tracing::debug!( + op, + ?txn_id, + array = %cluster_op.array_id().name, + "cluster array dispatch" + ); + let payload_bytes = executor.execute(&cluster_op, txn_id).await?; + + // Publish CDC change event(s) for a successful write. `Slice`/`Agg` are + // reads and publish nothing; `Put`/`Delete` carry their own + // Control-Plane-allocated `wal_lsn` since there is no Data-Plane + // `Response::watermark_lsn` on this coordinator-only path. + let write_lsn = match &cluster_op { + ClusterArrayOp::Put { wal_lsn, .. } | ClusterArrayOp::Delete { wal_lsn, .. } => { + Some(*wal_lsn) + } + ClusterArrayOp::Slice { .. } | ClusterArrayOp::Agg { .. } => None, + }; + if let Some(lsn) = write_lsn { + publish_cluster_array_change_events(state, tenant_id, database_id, &cluster_op, lsn); + } + + let cluster_plan_kind = match &cluster_op { + ClusterArrayOp::Slice { .. } => PlanKind::ArraySlice, + ClusterArrayOp::Agg { .. } => PlanKind::MultiRow, + // The coordinator reports `{"inserted": n}` / `{"deleted": n}`, the + // same count map the local array handlers emit. + ClusterArrayOp::Put { .. } => PlanKind::DmlResult("INSERT"), + ClusterArrayOp::Delete { .. } => PlanKind::DmlResult("DELETE"), + }; + // This coordinator path never builds a `PhysicalPlan`, so the source + // collection comes straight off the op's array name. A single source + // means bare-key matching, which is what an array's cell rows carry. + let array_name = match &cluster_op { + ClusterArrayOp::Slice { array_id, .. } + | ClusterArrayOp::Agg { array_id, .. } + | ClusterArrayOp::Put { array_id, .. } + | ClusterArrayOp::Delete { array_id, .. } => array_id.name.clone(), + }; + let redaction = + QueryRedaction::for_collections(tenant_id, auth, vec![(String::new(), array_name)]); + // A cluster array plan projects attribute names only, never a + // Control-Plane computed column, so no session sequence access. + match compose::shape_payload_no_plan( + &payload_bytes, + cluster_plan_kind, + projection, + Some(redaction.ctx(&state.redaction)), + None, + )? { + ShapeOutcome::Rows(shaped) => Ok(ClusterArrayShaped::Rows(shaped)), + ShapeOutcome::Passthrough => match cluster_plan_kind { + PlanKind::DmlResult(verb) => { + let affected = require_affected_count(&payload_bytes)?; + Ok(ClusterArrayShaped::Affected(DmlOutcome { verb, affected })) + } + // `cluster_plan_kind` above is only ever `ArraySlice`, `MultiRow` + // or `DmlResult(_)`; the remaining `PlanKind` variants can never + // reach this arm. Kept exhaustive (no `_ =>`) so a future + // `PlanKind` desync surfaces as a typed error rather than a panic. + PlanKind::DmlResultByOp + | PlanKind::Execution + | PlanKind::ArraySlice + | PlanKind::ReturningRows + | PlanKind::SingleDocument + | PlanKind::MultiRow => Err(crate::Error::Internal { + detail: format!( + "ClusterArray dispatch produced an unreachable passthrough plan kind: \ + {cluster_plan_kind:?}" + ), + }), + }, + } +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs index 955583ee8..23e63430d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs @@ -12,6 +12,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::planner::calvin::{build_static_tx_class, submit_calvin_routed}; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::session::{DmlTxnCtx, TransactionState}; +use crate::control::server::shared::sql::staging_predicates::require_affected_count; use crate::control::server::surrogate_exchange::assign_surrogate_routed; use crate::control::state::SharedState; use crate::types::{DatabaseId, TraceId, VShardId}; @@ -22,6 +23,17 @@ use super::super::super::result::{DdlError, DdlResult}; use super::edge_parse::{properties_to_json, validate_edge_label}; use super::support::{data_plane_verdict, ddl_err}; +/// Read the affected count off a Data-Plane response, mapping a missing count +/// to a [`DdlError`] via `ddl_err` — never a default. +fn response_affected(response: &crate::bridge::envelope::Response) -> Result { + require_affected_count(response.payload.as_bytes()).map_err(|e| { + ddl_err( + "XX000", + format!("edge write response is missing its affected count: {e}"), + ) + }) +} + /// `GRAPH INSERT EDGE IN '' FROM '' TO '' TYPE '