From 617b6e2c3cdfdbf0c5fec6858f9363849d9a5f78 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Thu, 17 Sep 2026 20:38:25 +0800 Subject: [PATCH 01/15] test(sql): cover computed columns and errors over derived tables Add wire-level tests for SELECT over derived tables covering constant, aggregate, grouped, and UNION ALL bodies: computed-column projection, aggregate-of-aggregate, window functions, and division-by-zero propagation through projections, aggregate arguments, GROUP BY keys, and window PARTITION BY. --- nodedb/tests/wire/cases/sql_subquery_from.rs | 275 +++++++++++++++++++ 1 file changed, 275 insertions(+) diff --git a/nodedb/tests/wire/cases/sql_subquery_from.rs b/nodedb/tests/wire/cases/sql_subquery_from.rs index 4b516bd4c..016d888a8 100644 --- a/nodedb/tests/wire/cases/sql_subquery_from.rs +++ b/nodedb/tests/wire/cases/sql_subquery_from.rs @@ -101,3 +101,278 @@ async fn derived_group_by_in_from_is_supported() { assert_eq!(totals.get("b"), Some(&7.0), "category b total should be 7"); assert_eq!(totals.get("c"), Some(&5.0), "category c total should be 5"); } + +/// A computed column over a constant derived table must evaluate. The +/// constant body lowers to a provider row, and the outer projection must +/// run over that row instead of resolving `x * 2` by name to NULL. +#[tokio::test] +async fn computed_column_over_constant_derived_table_evaluates() { + let srv = TestServer::start().await; + + let rows = srv + .query_rows("SELECT x * 2 AS doubled FROM (SELECT 1 AS x) AS s") + .await + .expect("computed column over a constant derived table must plan"); + + assert_eq!(rows, vec![vec!["2".to_string()]], "got {rows:?}"); +} + +/// A computed column over a derived table whose body is an aggregate must +/// evaluate over the aggregate's output row. +#[tokio::test] +async fn computed_column_over_aggregate_derived_table_evaluates() { + let srv = TestServer::start().await; + create_items(&srv).await; + + let rows = srv + .query_rows("SELECT total * 2 AS doubled FROM (SELECT SUM(qty) AS total FROM items) AS s") + .await + .expect("computed column over an aggregate derived table must plan"); + + assert_eq!(rows, vec![vec!["30".to_string()]], "got {rows:?}"); +} + +/// A computed column over a grouped derived table must evaluate per output +/// row and keep every projected column. +#[tokio::test] +async fn computed_column_over_grouped_derived_table_evaluates() { + let srv = TestServer::start().await; + create_items(&srv).await; + + let rows = srv + .query_rows( + "SELECT category, total * 2 AS doubled \ + FROM (SELECT category, SUM(qty) AS total FROM items GROUP BY category) AS agg", + ) + .await + .expect("computed column over a grouped derived table must plan"); + + let mut got: Vec<(String, String)> = + rows.iter().map(|r| (r[0].clone(), r[1].clone())).collect(); + got.sort(); + assert_eq!( + got, + vec![ + ("a".to_string(), "6".to_string()), + ("b".to_string(), "14".to_string()), + ("c".to_string(), "10".to_string()), + ], + "got {rows:?}" + ); +} + +/// A computed column over a UNION ALL derived table must evaluate per row. +#[tokio::test] +async fn computed_column_over_union_derived_table_evaluates() { + let srv = TestServer::start().await; + + let rows = srv + .query_rows("SELECT x * 2 AS doubled FROM (SELECT 1 AS x UNION ALL SELECT 2 AS x) AS s") + .await + .expect("computed column over a UNION ALL derived table must plan"); + + let mut got: Vec = rows.iter().map(|r| r[0].clone()).collect(); + got.sort(); + assert_eq!(got, vec!["2".to_string(), "4".to_string()], "got {rows:?}"); +} + +/// Division by zero in the projection over a constant derived table must +/// raise `22012`, never fold to a NULL row. +#[tokio::test] +async fn projection_division_by_zero_over_constant_derived_table_errors_22012() { + let srv = TestServer::start().await; + + srv.expect_error("SELECT x / 0 FROM (SELECT 1 AS x) AS s", "22012") + .await; +} + +/// Division by zero in an aggregate argument over a constant derived table +/// must raise `22012`, never fold to a NULL aggregate. +#[tokio::test] +async fn aggregate_argument_division_by_zero_over_constant_derived_table_errors_22012() { + let srv = TestServer::start().await; + + srv.expect_error("SELECT SUM(x / 0) FROM (SELECT 1 AS x) AS s", "22012") + .await; +} + +/// Division by zero in a GROUP BY key over a constant derived table must +/// raise `22012`, never return an empty result with a missing column. +#[tokio::test] +async fn group_by_key_division_by_zero_over_constant_derived_table_errors_22012() { + let srv = TestServer::start().await; + + srv.expect_error( + "SELECT x, COUNT(*) FROM (SELECT 1 AS x) AS s GROUP BY x / 0", + "22012", + ) + .await; +} + +/// Division by zero in a window PARTITION BY over a constant derived table +/// must raise `22012`, never fold to a NULL window value. +#[tokio::test] +async fn window_partition_division_by_zero_over_constant_derived_table_errors_22012() { + let srv = TestServer::start().await; + + srv.expect_error( + "SELECT SUM(x) OVER (PARTITION BY x / 0) FROM (SELECT 1 AS x) AS s", + "22012", + ) + .await; +} + +/// Division by zero in the projection over an aggregate derived table must +/// raise `22012`. +#[tokio::test] +async fn projection_division_by_zero_over_aggregate_derived_table_errors_22012() { + let srv = TestServer::start().await; + create_items(&srv).await; + + srv.expect_error( + "SELECT total / 0 FROM (SELECT SUM(qty) AS total FROM items) AS s", + "22012", + ) + .await; +} + +/// A grouped query over a constant derived table must keep every projected +/// column in the result, not drop the non-aggregate column. +#[tokio::test] +async fn group_by_over_constant_derived_table_keeps_projected_columns() { + let srv = TestServer::start().await; + + let rows = srv + .query_rows("SELECT x, COUNT(*) AS n FROM (SELECT 1 AS x) AS s GROUP BY x") + .await + .expect("GROUP BY over a constant derived table must plan"); + + assert_eq!( + rows, + vec![vec!["1".to_string(), "1".to_string()]], + "got {rows:?}" + ); +} + +/// Control: the same projection over a derived table that scans a +/// collection raises `22012`. The derived table itself is not the trigger. +#[tokio::test] +async fn projection_division_by_zero_over_scan_derived_table_errors_22012() { + let srv = TestServer::start().await; + create_items(&srv).await; + + srv.expect_error( + "SELECT x / 0 FROM (SELECT qty AS x FROM items) AS s", + "22012", + ) + .await; +} + +/// A computed column that references an inner alias (`qty AS x`) over a +/// scanning derived table must resolve through the alias. Merging the outer +/// projection onto the inner scan must not discard the inner rename. +#[tokio::test] +async fn computed_column_over_aliased_scan_derived_table_evaluates() { + let srv = TestServer::start().await; + create_items(&srv).await; + + let rows = srv + .query_rows( + "SELECT x * 2 AS doubled FROM (SELECT qty AS x FROM items WHERE id = 'i3') AS s", + ) + .await + .expect("computed column over an aliased scan derived table must plan"); + + assert_eq!(rows, vec![vec!["6".to_string()]], "got {rows:?}"); +} + +/// A computed column over a grouped derived table with an outer ORDER BY +/// (which routes through the row post-processor) must evaluate per row. +#[tokio::test] +async fn computed_column_over_grouped_derived_table_with_order_by_evaluates() { + let srv = TestServer::start().await; + create_items(&srv).await; + + let rows = srv + .query_rows( + "SELECT category, total * 2 AS doubled \ + FROM (SELECT category, SUM(qty) AS total FROM items GROUP BY category) AS agg \ + ORDER BY category", + ) + .await + .expect("computed column over a grouped derived table with ORDER BY must plan"); + + assert_eq!( + rows, + vec![ + vec!["a".to_string(), "6".to_string()], + vec!["b".to_string(), "14".to_string()], + vec!["c".to_string(), "10".to_string()], + ], + "got {rows:?}" + ); +} + +/// An aggregate over a grouped derived table (aggregate of aggregates) must +/// run over the inner group rows, not over an empty collection. +#[tokio::test] +async fn aggregate_over_grouped_derived_table_evaluates() { + let srv = TestServer::start().await; + create_items(&srv).await; + + let rows = srv + .query_rows( + "SELECT SUM(total) AS grand, COUNT(*) AS groups \ + FROM (SELECT category, SUM(qty) AS total FROM items GROUP BY category) AS agg", + ) + .await + .expect("aggregate over a grouped derived table must plan"); + + assert_eq!( + rows, + vec![vec!["15".to_string(), "3".to_string()]], + "got {rows:?}" + ); +} + +/// An aggregate over a UNION ALL derived table must run over the union rows. +#[tokio::test] +async fn aggregate_over_union_derived_table_evaluates() { + let srv = TestServer::start().await; + + let rows = srv + .query_rows("SELECT SUM(x) AS total FROM (SELECT 1 AS x UNION ALL SELECT 2 AS x) AS s") + .await + .expect("aggregate over a UNION ALL derived table must plan"); + + assert_eq!(rows, vec![vec!["3".to_string()]], "got {rows:?}"); +} + +/// A window function over a grouped derived table must rank the inner group +/// rows. +#[tokio::test] +async fn window_over_grouped_derived_table_evaluates() { + let srv = TestServer::start().await; + create_items(&srv).await; + + let rows = srv + .query_rows( + "SELECT category, RANK() OVER (ORDER BY total DESC) AS rnk \ + FROM (SELECT category, SUM(qty) AS total FROM items GROUP BY category) AS agg", + ) + .await + .expect("window function over a grouped derived table must plan"); + + let mut got: Vec<(String, String)> = + rows.iter().map(|r| (r[0].clone(), r[1].clone())).collect(); + got.sort(); + assert_eq!( + got, + vec![ + ("a".to_string(), "3".to_string()), + ("b".to_string(), "1".to_string()), + ("c".to_string(), "2".to_string()), + ], + "got {rows:?}" + ); +} From 8180eb276084b1dbcac407a8c138aa7d971a5c41 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Thu, 17 Sep 2026 20:56:48 +0800 Subject: [PATCH 02/15] refactor(planner): split sql_plan_convert/expr.rs by concern Move bridge-expression conversion, CTE inlining, and sort-key conversion into separate files under expr/, with mod.rs limited to module declarations and re-exports. --- .../sql_plan_convert/expr/bridge_expr.rs | 265 ++++++++++++++++ .../{expr.rs => expr/inline_cte.rs} | 286 +----------------- .../planner/sql_plan_convert/expr/mod.rs | 11 + .../sql_plan_convert/expr/sort_keys.rs | 23 ++ 4 files changed, 305 insertions(+), 280 deletions(-) create mode 100644 nodedb/src/control/planner/sql_plan_convert/expr/bridge_expr.rs rename nodedb/src/control/planner/sql_plan_convert/{expr.rs => expr/inline_cte.rs} (57%) create mode 100644 nodedb/src/control/planner/sql_plan_convert/expr/mod.rs create mode 100644 nodedb/src/control/planner/sql_plan_convert/expr/sort_keys.rs diff --git a/nodedb/src/control/planner/sql_plan_convert/expr/bridge_expr.rs b/nodedb/src/control/planner/sql_plan_convert/expr/bridge_expr.rs new file mode 100644 index 000000000..4924eba8b --- /dev/null +++ b/nodedb/src/control/planner/sql_plan_convert/expr/bridge_expr.rs @@ -0,0 +1,265 @@ +// SPDX-License-Identifier: BUSL-1.1 + +use nodedb_sql::types::SqlExpr; + +use super::super::value::sql_value_to_nodedb_value; + +/// Convert a `nodedb_sql::types::SqlExpr` (parser AST) to a +/// `nodedb_query::expr::SqlExpr` (bridge evaluation type). +/// +/// Column references use the **bare** name (no table qualifier) for +/// single-collection evaluation contexts (WHERE, CHECK, GENERATED). +/// For join contexts where the merged document uses qualified keys +/// (`"t1.col"`), use [`sql_expr_to_bridge_expr_qualified`] instead. +pub(in crate::control::planner::sql_plan_convert) fn sql_expr_to_bridge_expr( + expr: &SqlExpr, +) -> crate::bridge::expr_eval::SqlExpr { + convert_expr_inner(expr, false) +} + +/// Like [`sql_expr_to_bridge_expr`] but qualifies column references +/// with their table name (`t.col` → `"t.col"`) for join merged docs. +pub(in crate::control::planner::sql_plan_convert) fn sql_expr_to_bridge_expr_qualified( + expr: &SqlExpr, +) -> crate::bridge::expr_eval::SqlExpr { + convert_expr_inner(expr, true) +} + +fn convert_expr_inner(expr: &SqlExpr, qualify: bool) -> crate::bridge::expr_eval::SqlExpr { + use crate::bridge::expr_eval::SqlExpr as BExpr; + match expr { + SqlExpr::Column { table, name } => { + // `EXCLUDED.col` references the row proposed for insertion in + // `INSERT ... ON CONFLICT DO UPDATE`. Emit the dedicated + // variant so the upsert handler can resolve against the + // incoming row via `eval_with_excluded`. The table qualifier + // comes in already-normalized (lowercased) from the parser. + if table + .as_deref() + .is_some_and(|t| t.eq_ignore_ascii_case("excluded")) + { + return BExpr::ExcludedColumn(name.clone()); + } + if qualify { + BExpr::Column(nodedb_sql::planner::qualified_name(table.as_deref(), name)) + } else { + BExpr::Column(name.clone()) + } + } + SqlExpr::Literal(v) => BExpr::Literal(sql_value_to_nodedb_value(v)), + SqlExpr::BinaryOp { left, op, right } => BExpr::BinaryOp { + left: Box::new(convert_expr_inner(left, qualify)), + op: match op { + nodedb_sql::types::BinaryOp::Add => crate::bridge::expr_eval::BinaryOp::Add, + nodedb_sql::types::BinaryOp::Sub => crate::bridge::expr_eval::BinaryOp::Sub, + nodedb_sql::types::BinaryOp::Mul => crate::bridge::expr_eval::BinaryOp::Mul, + nodedb_sql::types::BinaryOp::Div => crate::bridge::expr_eval::BinaryOp::Div, + nodedb_sql::types::BinaryOp::Mod => crate::bridge::expr_eval::BinaryOp::Mod, + nodedb_sql::types::BinaryOp::Eq => crate::bridge::expr_eval::BinaryOp::Eq, + nodedb_sql::types::BinaryOp::Ne => crate::bridge::expr_eval::BinaryOp::NotEq, + nodedb_sql::types::BinaryOp::Gt => crate::bridge::expr_eval::BinaryOp::Gt, + nodedb_sql::types::BinaryOp::Ge => crate::bridge::expr_eval::BinaryOp::GtEq, + nodedb_sql::types::BinaryOp::Lt => crate::bridge::expr_eval::BinaryOp::Lt, + nodedb_sql::types::BinaryOp::Le => crate::bridge::expr_eval::BinaryOp::LtEq, + nodedb_sql::types::BinaryOp::And => crate::bridge::expr_eval::BinaryOp::And, + nodedb_sql::types::BinaryOp::Or => crate::bridge::expr_eval::BinaryOp::Or, + nodedb_sql::types::BinaryOp::Concat => crate::bridge::expr_eval::BinaryOp::Concat, + }, + right: Box::new(convert_expr_inner(right, qualify)), + }, + SqlExpr::Function { name, args, .. } => BExpr::Function { + name: name.clone(), + args: args + .iter() + .map(|a| convert_expr_inner(a, qualify)) + .collect(), + }, + SqlExpr::Case { + operand, + when_then, + else_expr, + } => BExpr::Case { + operand: operand + .as_ref() + .map(|e| Box::new(convert_expr_inner(e, qualify))), + when_thens: when_then + .iter() + .map(|(w, t)| { + ( + convert_expr_inner(w, qualify), + convert_expr_inner(t, qualify), + ) + }) + .collect(), + else_expr: else_expr + .as_ref() + .map(|e| Box::new(convert_expr_inner(e, qualify))), + }, + SqlExpr::Cast { expr, to_type } => { + let cast_type = match to_type.to_uppercase().as_str() { + "INT" | "INTEGER" | "BIGINT" | "SMALLINT" => { + crate::bridge::expr_eval::CastType::Int + } + "FLOAT" | "DOUBLE" | "REAL" | "NUMERIC" | "DECIMAL" => { + crate::bridge::expr_eval::CastType::Float + } + "BOOL" | "BOOLEAN" => crate::bridge::expr_eval::CastType::Bool, + _ => crate::bridge::expr_eval::CastType::String, + }; + BExpr::Cast { + expr: Box::new(convert_expr_inner(expr, qualify)), + to_type: cast_type, + } + } + SqlExpr::Wildcard => BExpr::Column("*".into()), + + // NOT e / -e → evaluator's Negate (handles both bool and numeric). + SqlExpr::UnaryOp { expr, .. } => BExpr::Negate(Box::new(convert_expr_inner(expr, qualify))), + + // `e IS NULL` / `e IS NOT NULL` — direct passthrough. + SqlExpr::IsNull { expr, negated } => BExpr::IsNull { + expr: Box::new(convert_expr_inner(expr, qualify)), + negated: *negated, + }, + + // `e BETWEEN low AND high` desugars to `e >= low AND e <= high` + // (or `e < low OR e > high` when negated). The evaluator has no + // native Between variant, so the planner must lower it here. + SqlExpr::Between { + expr, + low, + high, + negated, + } => { + let e = convert_expr_inner(expr, qualify); + let l = convert_expr_inner(low, qualify); + let h = convert_expr_inner(high, qualify); + if *negated { + let lt = BExpr::BinaryOp { + left: Box::new(e.clone()), + op: crate::bridge::expr_eval::BinaryOp::Lt, + right: Box::new(l), + }; + let gt = BExpr::BinaryOp { + left: Box::new(e), + op: crate::bridge::expr_eval::BinaryOp::Gt, + right: Box::new(h), + }; + BExpr::BinaryOp { + left: Box::new(lt), + op: crate::bridge::expr_eval::BinaryOp::Or, + right: Box::new(gt), + } + } else { + let ge = BExpr::BinaryOp { + left: Box::new(e.clone()), + op: crate::bridge::expr_eval::BinaryOp::GtEq, + right: Box::new(l), + }; + let le = BExpr::BinaryOp { + left: Box::new(e), + op: crate::bridge::expr_eval::BinaryOp::LtEq, + right: Box::new(h), + }; + BExpr::BinaryOp { + left: Box::new(ge), + op: crate::bridge::expr_eval::BinaryOp::And, + right: Box::new(le), + } + } + } + + // `e IN (a, b, c)` desugars to `e = a OR e = b OR e = c` — each + // element may itself be a non-literal expression, so we must + // recursively convert and OR the comparisons together. `NOT IN` + // is `e <> a AND e <> b AND e <> c`. + SqlExpr::InList { + expr, + list, + negated, + } => { + let target = convert_expr_inner(expr, qualify); + if list.is_empty() { + // Empty list: `e IN ()` = false, `e NOT IN ()` = true. + return BExpr::Literal(nodedb_types::Value::Bool(*negated)); + } + let (eq_op, combine_op) = if *negated { + ( + crate::bridge::expr_eval::BinaryOp::NotEq, + crate::bridge::expr_eval::BinaryOp::And, + ) + } else { + ( + crate::bridge::expr_eval::BinaryOp::Eq, + crate::bridge::expr_eval::BinaryOp::Or, + ) + }; + // Empty list is handled above, so `list` is guaranteed non-empty + // here: we reduce `(target eq list[0]) op (target eq list[1]) op ...` + // without touching `.unwrap()` or `.expect()`. + list.iter() + .map(|item| BExpr::BinaryOp { + left: Box::new(target.clone()), + op: eq_op, + right: Box::new(convert_expr_inner(item, qualify)), + }) + .reduce(|acc, next| BExpr::BinaryOp { + left: Box::new(acc), + op: combine_op, + right: Box::new(next), + }) + // Unreachable: `list.is_empty()` returns early above. + .unwrap_or(BExpr::Literal(nodedb_types::Value::Bool(*negated))) + } + + // `e LIKE pattern` — no direct evaluator variant; route through a + // function call so the shared function dispatcher handles it. + SqlExpr::Like { + expr, + pattern, + negated, + case_insensitive, + } => { + let fn_name = if *case_insensitive { "ilike" } else { "like" }; + let call = BExpr::Function { + name: fn_name.into(), + args: vec![ + convert_expr_inner(expr, qualify), + convert_expr_inner(pattern, qualify), + ], + }; + if *negated { + BExpr::Negate(Box::new(call)) + } else { + call + } + } + + // `ARRAY['a', 'b', ...]` — lower each element and, when all resolve to + // `BExpr::Literal`, fold into a single `Value::Array` literal so that + // functions like `pg_json_has_any_key` / `pg_json_has_all_keys` receive + // a proper `Value::Array` argument rather than `Value::Null`. + SqlExpr::ArrayLiteral(elems) => { + let mut values = Vec::with_capacity(elems.len()); + let mut all_literal = true; + for elem in elems { + match convert_expr_inner(elem, qualify) { + BExpr::Literal(v) => values.push(v), + other => { + all_literal = false; + // Non-literal element: fall back to Null for that slot. + let _ = other; + values.push(nodedb_types::Value::Null); + } + } + } + if all_literal { + BExpr::Literal(nodedb_types::Value::Array(values)) + } else { + BExpr::Literal(nodedb_types::Value::Null) + } + } + + _ => BExpr::Literal(nodedb_types::Value::Null), + } +} diff --git a/nodedb/src/control/planner/sql_plan_convert/expr.rs b/nodedb/src/control/planner/sql_plan_convert/expr/inline_cte.rs similarity index 57% rename from nodedb/src/control/planner/sql_plan_convert/expr.rs rename to nodedb/src/control/planner/sql_plan_convert/expr/inline_cte.rs index 80119f4ff..ad7302897 100644 --- a/nodedb/src/control/planner/sql_plan_convert/expr.rs +++ b/nodedb/src/control/planner/sql_plan_convert/expr/inline_cte.rs @@ -1,284 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Expression conversion and CTE inlining. - -use nodedb_physical::physical_plan::SortKeySpec; -use nodedb_sql::types::{SortKey, SqlExpr, SqlPlan}; - -use super::value::sql_value_to_nodedb_value; - -/// Convert a `nodedb_sql::types::SqlExpr` (parser AST) to a -/// `nodedb_query::expr::SqlExpr` (bridge evaluation type). -/// -/// Column references use the **bare** name (no table qualifier) for -/// single-collection evaluation contexts (WHERE, CHECK, GENERATED). -/// For join contexts where the merged document uses qualified keys -/// (`"t1.col"`), use [`sql_expr_to_bridge_expr_qualified`] instead. -pub(super) fn sql_expr_to_bridge_expr(expr: &SqlExpr) -> crate::bridge::expr_eval::SqlExpr { - convert_expr_inner(expr, false) -} - -/// Like [`sql_expr_to_bridge_expr`] but qualifies column references -/// with their table name (`t.col` → `"t.col"`) for join merged docs. -pub(super) fn sql_expr_to_bridge_expr_qualified( - expr: &SqlExpr, -) -> crate::bridge::expr_eval::SqlExpr { - convert_expr_inner(expr, true) -} - -fn convert_expr_inner(expr: &SqlExpr, qualify: bool) -> crate::bridge::expr_eval::SqlExpr { - use crate::bridge::expr_eval::SqlExpr as BExpr; - match expr { - SqlExpr::Column { table, name } => { - // `EXCLUDED.col` references the row proposed for insertion in - // `INSERT ... ON CONFLICT DO UPDATE`. Emit the dedicated - // variant so the upsert handler can resolve against the - // incoming row via `eval_with_excluded`. The table qualifier - // comes in already-normalized (lowercased) from the parser. - if table - .as_deref() - .is_some_and(|t| t.eq_ignore_ascii_case("excluded")) - { - return BExpr::ExcludedColumn(name.clone()); - } - if qualify { - BExpr::Column(nodedb_sql::planner::qualified_name(table.as_deref(), name)) - } else { - BExpr::Column(name.clone()) - } - } - SqlExpr::Literal(v) => BExpr::Literal(sql_value_to_nodedb_value(v)), - SqlExpr::BinaryOp { left, op, right } => BExpr::BinaryOp { - left: Box::new(convert_expr_inner(left, qualify)), - op: match op { - nodedb_sql::types::BinaryOp::Add => crate::bridge::expr_eval::BinaryOp::Add, - nodedb_sql::types::BinaryOp::Sub => crate::bridge::expr_eval::BinaryOp::Sub, - nodedb_sql::types::BinaryOp::Mul => crate::bridge::expr_eval::BinaryOp::Mul, - nodedb_sql::types::BinaryOp::Div => crate::bridge::expr_eval::BinaryOp::Div, - nodedb_sql::types::BinaryOp::Mod => crate::bridge::expr_eval::BinaryOp::Mod, - nodedb_sql::types::BinaryOp::Eq => crate::bridge::expr_eval::BinaryOp::Eq, - nodedb_sql::types::BinaryOp::Ne => crate::bridge::expr_eval::BinaryOp::NotEq, - nodedb_sql::types::BinaryOp::Gt => crate::bridge::expr_eval::BinaryOp::Gt, - nodedb_sql::types::BinaryOp::Ge => crate::bridge::expr_eval::BinaryOp::GtEq, - nodedb_sql::types::BinaryOp::Lt => crate::bridge::expr_eval::BinaryOp::Lt, - nodedb_sql::types::BinaryOp::Le => crate::bridge::expr_eval::BinaryOp::LtEq, - nodedb_sql::types::BinaryOp::And => crate::bridge::expr_eval::BinaryOp::And, - nodedb_sql::types::BinaryOp::Or => crate::bridge::expr_eval::BinaryOp::Or, - nodedb_sql::types::BinaryOp::Concat => crate::bridge::expr_eval::BinaryOp::Concat, - }, - right: Box::new(convert_expr_inner(right, qualify)), - }, - SqlExpr::Function { name, args, .. } => BExpr::Function { - name: name.clone(), - args: args - .iter() - .map(|a| convert_expr_inner(a, qualify)) - .collect(), - }, - SqlExpr::Case { - operand, - when_then, - else_expr, - } => BExpr::Case { - operand: operand - .as_ref() - .map(|e| Box::new(convert_expr_inner(e, qualify))), - when_thens: when_then - .iter() - .map(|(w, t)| { - ( - convert_expr_inner(w, qualify), - convert_expr_inner(t, qualify), - ) - }) - .collect(), - else_expr: else_expr - .as_ref() - .map(|e| Box::new(convert_expr_inner(e, qualify))), - }, - SqlExpr::Cast { expr, to_type } => { - let cast_type = match to_type.to_uppercase().as_str() { - "INT" | "INTEGER" | "BIGINT" | "SMALLINT" => { - crate::bridge::expr_eval::CastType::Int - } - "FLOAT" | "DOUBLE" | "REAL" | "NUMERIC" | "DECIMAL" => { - crate::bridge::expr_eval::CastType::Float - } - "BOOL" | "BOOLEAN" => crate::bridge::expr_eval::CastType::Bool, - _ => crate::bridge::expr_eval::CastType::String, - }; - BExpr::Cast { - expr: Box::new(convert_expr_inner(expr, qualify)), - to_type: cast_type, - } - } - SqlExpr::Wildcard => BExpr::Column("*".into()), - - // NOT e / -e → evaluator's Negate (handles both bool and numeric). - SqlExpr::UnaryOp { expr, .. } => BExpr::Negate(Box::new(convert_expr_inner(expr, qualify))), - - // `e IS NULL` / `e IS NOT NULL` — direct passthrough. - SqlExpr::IsNull { expr, negated } => BExpr::IsNull { - expr: Box::new(convert_expr_inner(expr, qualify)), - negated: *negated, - }, - - // `e BETWEEN low AND high` desugars to `e >= low AND e <= high` - // (or `e < low OR e > high` when negated). The evaluator has no - // native Between variant, so the planner must lower it here. - SqlExpr::Between { - expr, - low, - high, - negated, - } => { - let e = convert_expr_inner(expr, qualify); - let l = convert_expr_inner(low, qualify); - let h = convert_expr_inner(high, qualify); - if *negated { - let lt = BExpr::BinaryOp { - left: Box::new(e.clone()), - op: crate::bridge::expr_eval::BinaryOp::Lt, - right: Box::new(l), - }; - let gt = BExpr::BinaryOp { - left: Box::new(e), - op: crate::bridge::expr_eval::BinaryOp::Gt, - right: Box::new(h), - }; - BExpr::BinaryOp { - left: Box::new(lt), - op: crate::bridge::expr_eval::BinaryOp::Or, - right: Box::new(gt), - } - } else { - let ge = BExpr::BinaryOp { - left: Box::new(e.clone()), - op: crate::bridge::expr_eval::BinaryOp::GtEq, - right: Box::new(l), - }; - let le = BExpr::BinaryOp { - left: Box::new(e), - op: crate::bridge::expr_eval::BinaryOp::LtEq, - right: Box::new(h), - }; - BExpr::BinaryOp { - left: Box::new(ge), - op: crate::bridge::expr_eval::BinaryOp::And, - right: Box::new(le), - } - } - } - - // `e IN (a, b, c)` desugars to `e = a OR e = b OR e = c` — each - // element may itself be a non-literal expression, so we must - // recursively convert and OR the comparisons together. `NOT IN` - // is `e <> a AND e <> b AND e <> c`. - SqlExpr::InList { - expr, - list, - negated, - } => { - let target = convert_expr_inner(expr, qualify); - if list.is_empty() { - // Empty list: `e IN ()` = false, `e NOT IN ()` = true. - return BExpr::Literal(nodedb_types::Value::Bool(*negated)); - } - let (eq_op, combine_op) = if *negated { - ( - crate::bridge::expr_eval::BinaryOp::NotEq, - crate::bridge::expr_eval::BinaryOp::And, - ) - } else { - ( - crate::bridge::expr_eval::BinaryOp::Eq, - crate::bridge::expr_eval::BinaryOp::Or, - ) - }; - // Empty list is handled above, so `list` is guaranteed non-empty - // here: we reduce `(target eq list[0]) op (target eq list[1]) op ...` - // without touching `.unwrap()` or `.expect()`. - list.iter() - .map(|item| BExpr::BinaryOp { - left: Box::new(target.clone()), - op: eq_op, - right: Box::new(convert_expr_inner(item, qualify)), - }) - .reduce(|acc, next| BExpr::BinaryOp { - left: Box::new(acc), - op: combine_op, - right: Box::new(next), - }) - // Unreachable: `list.is_empty()` returns early above. - .unwrap_or(BExpr::Literal(nodedb_types::Value::Bool(*negated))) - } - - // `e LIKE pattern` — no direct evaluator variant; route through a - // function call so the shared function dispatcher handles it. - SqlExpr::Like { - expr, - pattern, - negated, - case_insensitive, - } => { - let fn_name = if *case_insensitive { "ilike" } else { "like" }; - let call = BExpr::Function { - name: fn_name.into(), - args: vec![ - convert_expr_inner(expr, qualify), - convert_expr_inner(pattern, qualify), - ], - }; - if *negated { - BExpr::Negate(Box::new(call)) - } else { - call - } - } - - // `ARRAY['a', 'b', ...]` — lower each element and, when all resolve to - // `BExpr::Literal`, fold into a single `Value::Array` literal so that - // functions like `pg_json_has_any_key` / `pg_json_has_all_keys` receive - // a proper `Value::Array` argument rather than `Value::Null`. - SqlExpr::ArrayLiteral(elems) => { - let mut values = Vec::with_capacity(elems.len()); - let mut all_literal = true; - for elem in elems { - match convert_expr_inner(elem, qualify) { - BExpr::Literal(v) => values.push(v), - other => { - all_literal = false; - // Non-literal element: fall back to Null for that slot. - let _ = other; - values.push(nodedb_types::Value::Null); - } - } - } - if all_literal { - BExpr::Literal(nodedb_types::Value::Array(values)) - } else { - BExpr::Literal(nodedb_types::Value::Null) - } - } - - _ => BExpr::Literal(nodedb_types::Value::Null), - } -} - -/// Lower planner sort keys to their physical form. -/// -/// Every key is carried, expression and all. Dropping a key the Data Plane -/// cannot name as a stored column would silently answer -/// `ORDER BY 100 / weight` with rows in storage order. -pub(super) fn convert_sort_keys(keys: &[SortKey]) -> Vec { - keys.iter() - .map(|k| SortKeySpec { - expr: sql_expr_to_bridge_expr(&k.expr), - ascending: k.ascending, - nulls_first: k.nulls_first, - }) - .collect() -} +use nodedb_sql::types::SqlPlan; /// Replace scans on `cte_name` with the CTE's actual subquery plan. /// @@ -287,7 +9,11 @@ pub(super) fn convert_sort_keys(keys: &[SortKey]) -> Vec { /// `VectorSearch` body takes filters, projection, and an unordered LIMIT (as /// `top_k`). Constraints a body has no slot for — an outer `ORDER BY`, OFFSET, /// or DISTINCT over a non-`Scan` body — are not applied. -pub(super) fn inline_cte(plan: &SqlPlan, cte_name: &str, cte_plan: &SqlPlan) -> SqlPlan { +pub(in crate::control::planner::sql_plan_convert) fn inline_cte( + plan: &SqlPlan, + cte_name: &str, + cte_plan: &SqlPlan, +) -> SqlPlan { match plan { // Direct scan on CTE name → replace with CTE plan. SqlPlan::Scan { diff --git a/nodedb/src/control/planner/sql_plan_convert/expr/mod.rs b/nodedb/src/control/planner/sql_plan_convert/expr/mod.rs new file mode 100644 index 000000000..a2e40b80f --- /dev/null +++ b/nodedb/src/control/planner/sql_plan_convert/expr/mod.rs @@ -0,0 +1,11 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Expression conversion and CTE inlining. + +mod bridge_expr; +mod inline_cte; +mod sort_keys; + +pub(super) use bridge_expr::{sql_expr_to_bridge_expr, sql_expr_to_bridge_expr_qualified}; +pub(super) use inline_cte::inline_cte; +pub(super) use sort_keys::convert_sort_keys; diff --git a/nodedb/src/control/planner/sql_plan_convert/expr/sort_keys.rs b/nodedb/src/control/planner/sql_plan_convert/expr/sort_keys.rs new file mode 100644 index 000000000..2f80dbf6a --- /dev/null +++ b/nodedb/src/control/planner/sql_plan_convert/expr/sort_keys.rs @@ -0,0 +1,23 @@ +// SPDX-License-Identifier: BUSL-1.1 + +use nodedb_physical::physical_plan::SortKeySpec; +use nodedb_sql::types::SortKey; + +use super::bridge_expr::sql_expr_to_bridge_expr; + +/// Lower planner sort keys to their physical form. +/// +/// Every key is carried, expression and all. Dropping a key the Data Plane +/// cannot name as a stored column would silently answer +/// `ORDER BY 100 / weight` with rows in storage order. +pub(in crate::control::planner::sql_plan_convert) fn convert_sort_keys( + keys: &[SortKey], +) -> Vec { + keys.iter() + .map(|k| SortKeySpec { + expr: sql_expr_to_bridge_expr(&k.expr), + ascending: k.ascending, + nulls_first: k.nulls_first, + }) + .collect() +} From f6bc12630dd51d9ca6d35b3bd6782375fd5fb1fa Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Thu, 17 Sep 2026 21:50:10 +0800 Subject: [PATCH 03/15] feat(query): evaluate window functions and computed columns in ProviderScan Carry serialized computed-column and window-function specs on the ScanProvider and Scan QueryOp variants, threading the new fields through every planner conversion path, exchange resolution, clone rewriting, and the cluster shuffle test fixtures that construct these plan nodes. Add provider_scan_compute to decode the row set to JSON, evaluate window functions over the full set, apply computed columns per row, and re-encode to msgpack; the executor runs this step after sort and before distinct, skipping it entirely when both byte slices are empty so a plain relational scan stays on the zero-decode msgpack path. Reorder the ProviderScan pipeline so offset runs after sort instead of before, matching ORDER BY ... OFFSET semantics. --- .../cases/shuffle_aggregate_cross_node.rs | 2 + .../cases/shuffle_consume_cross_node.rs | 2 + .../cases/shuffle_produce_cross_node.rs | 2 + nodedb-physical/src/physical_plan/query.rs | 16 +++ .../src/physical_plan/streaming.rs | 3 +- nodedb/src/control/clone/resolver/rewrite.rs | 6 ++ .../control/planner/redaction_refusal/plan.rs | 2 + .../rls_injection/permission_tree/plan.rs | 2 + .../src/control/planner/rls_injection/plan.rs | 2 + .../sql_plan_convert/aggregate/plan.rs | 2 + .../planner/sql_plan_convert/convert.rs | 2 + .../planner/sql_plan_convert/scan/core.rs | 7 +- .../planner/sql_plan_convert/set_ops.rs | 4 + .../exchange/resolve/exchange/dispatch.rs | 4 + .../resolve/exchange/post_process_arm.rs | 6 ++ .../server/exchange/resolve/join_input.rs | 6 ++ .../server/exchange/resolve/materialize.rs | 8 ++ .../shared/authorization/requirements.rs | 2 + .../predicate/txn_buffering/classify.rs | 2 + nodedb/src/data/executor/dispatch/query.rs | 4 + nodedb/src/data/executor/handlers/mod.rs | 1 + .../data/executor/handlers/provider_scan.rs | 60 +++++++---- .../handlers/provider_scan_compute.rs | 100 ++++++++++++++++++ .../test_cross_type_join/inline_hash_join.rs | 4 + .../test_cross_type_join/multi_core_joins.rs | 6 ++ 25 files changed, 234 insertions(+), 21 deletions(-) create mode 100644 nodedb/src/data/executor/handlers/provider_scan_compute.rs diff --git a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_aggregate_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_aggregate_cross_node.rs index d534d18ba..08beb3d1a 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_aggregate_cross_node.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_aggregate_cross_node.rs @@ -107,6 +107,8 @@ fn producer_plan(rows: &[&Row]) -> Vec { rows: msgpack_array(rows), filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, diff --git a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_consume_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_consume_cross_node.rs index 205772766..f343e19fe 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_consume_cross_node.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_consume_cross_node.rs @@ -104,6 +104,8 @@ fn provider_scan_plan(rows: &[&Row]) -> Vec { rows: msgpack_array(rows), filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, diff --git a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_produce_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_produce_cross_node.rs index 3f0602c28..49f8bf830 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_produce_cross_node.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_produce_cross_node.rs @@ -108,6 +108,8 @@ fn provider_scan_plan(rows: &[Vec]) -> Vec { rows: msgpack_array(rows), filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, diff --git a/nodedb-physical/src/physical_plan/query.rs b/nodedb-physical/src/physical_plan/query.rs index 96f087d3c..001c5d459 100644 --- a/nodedb-physical/src/physical_plan/query.rs +++ b/nodedb-physical/src/physical_plan/query.rs @@ -78,6 +78,14 @@ pub enum QueryOp { /// Output column names to keep. Empty = emit all columns. #[serde(default)] projection: Vec, + /// Serialized `Vec` (MessagePack), same encoding as + /// `DocumentOp::Scan::computed_columns`. Empty = none. + #[serde(default)] + computed_columns: Vec, + /// Serialized `Vec` (MessagePack), same encoding as + /// `DocumentOp::Scan::window_functions`. Empty = none. + #[serde(default)] + window_functions: Vec, /// ORDER BY terms, each an expression. Empty = unordered. #[serde(default)] sort_keys: Vec, @@ -118,6 +126,14 @@ pub enum QueryOp { /// Output column names to keep. Empty = emit all columns. #[serde(default)] projection: Vec, + /// Serialized `Vec` (MessagePack), same encoding as + /// `DocumentOp::Scan::computed_columns`. Empty = none. + #[serde(default)] + computed_columns: Vec, + /// Serialized `Vec` (MessagePack), same encoding as + /// `DocumentOp::Scan::window_functions`. Empty = none. + #[serde(default)] + window_functions: Vec, /// ORDER BY terms, each an expression. Empty = unordered. #[serde(default)] sort_keys: Vec, diff --git a/nodedb-physical/src/physical_plan/streaming.rs b/nodedb-physical/src/physical_plan/streaming.rs index 76d92634b..cd728fc93 100644 --- a/nodedb-physical/src/physical_plan/streaming.rs +++ b/nodedb-physical/src/physical_plan/streaming.rs @@ -42,8 +42,9 @@ impl PhysicalPlan { sort_keys, offset, distinct, + window_functions, .. - }) => sort_keys.is_empty() && *offset == 0 && !*distinct, + }) => sort_keys.is_empty() && *offset == 0 && !*distinct && window_functions.is_empty(), // Every other Document / Kv / Columnar / Timeseries op, plus all // other engines and query ops, are not unordered-streamable. diff --git a/nodedb/src/control/clone/resolver/rewrite.rs b/nodedb/src/control/clone/resolver/rewrite.rs index 237ee8f23..c2f7df732 100644 --- a/nodedb/src/control/clone/resolver/rewrite.rs +++ b/nodedb/src/control/clone/resolver/rewrite.rs @@ -117,6 +117,8 @@ pub fn rewrite_plan_for_source(params: RewriteForSourceParams<'_>) -> crate::Res input, filters, projection, + computed_columns, + window_functions, sort_keys, limit, offset, @@ -139,6 +141,8 @@ pub fn rewrite_plan_for_source(params: RewriteForSourceParams<'_>) -> crate::Res input: child, filters: filters.clone(), projection: projection.clone(), + computed_columns: computed_columns.clone(), + window_functions: window_functions.clone(), sort_keys: sort_keys.clone(), limit: *limit, offset: *offset, @@ -626,6 +630,8 @@ mod tests { input: Box::new(gather(plan)), filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, diff --git a/nodedb/src/control/planner/redaction_refusal/plan.rs b/nodedb/src/control/planner/redaction_refusal/plan.rs index dd71f71eb..4e1f08fe0 100644 --- a/nodedb/src/control/planner/redaction_refusal/plan.rs +++ b/nodedb/src/control/planner/redaction_refusal/plan.rs @@ -461,6 +461,8 @@ mod tests { input: Box::new(aggregate_plan("users", vec![agg_spec("min", "ssn")])), filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/plan.rs b/nodedb/src/control/planner/rls_injection/permission_tree/plan.rs index 742d8716f..0ada83e52 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/plan.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/plan.rs @@ -414,6 +414,8 @@ mod tests { input: Box::new(columnar_scan("events")), filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, diff --git a/nodedb/src/control/planner/rls_injection/plan.rs b/nodedb/src/control/planner/rls_injection/plan.rs index 57cf47739..671884b88 100644 --- a/nodedb/src/control/planner/rls_injection/plan.rs +++ b/nodedb/src/control/planner/rls_injection/plan.rs @@ -412,6 +412,8 @@ mod tests { input: Box::new(rag_fusion("docs")), filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, diff --git a/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs b/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs index 2f9c4c3bc..aad2dc7a9 100644 --- a/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs +++ b/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs @@ -166,6 +166,8 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_aggregate( // before the rows reach the aggregate. filters: filter_bytes.clone(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, diff --git a/nodedb/src/control/planner/sql_plan_convert/convert.rs b/nodedb/src/control/planner/sql_plan_convert/convert.rs index 1b38821ea..03a901dfc 100644 --- a/nodedb/src/control/planner/sql_plan_convert/convert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/convert.rs @@ -251,6 +251,8 @@ pub fn convert( rows: Vec::new(), filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/core.rs b/nodedb/src/control/planner/sql_plan_convert/scan/core.rs index 5e56c2697..5f6ff318a 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/core.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/core.rs @@ -43,6 +43,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_scan( // `rows` is left empty here; the coordinator fills it post-cache via // `materialize_providers`. Using an empty-coordinator vshard (empty // collection string) keeps the task coordinator-local. + let computed_bytes = extract_computed_columns(projection, window_functions)?; + let window_bytes = serialize_window_functions(window_functions)?; + if crate::control::server::pgwire::catalog::schema::catalog_collection_info(collection) .is_some() { @@ -58,6 +61,8 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_scan( rows: Vec::new(), filters: filter_bytes, projection: proj_names, + computed_columns: computed_bytes, + window_functions: window_bytes, sort_keys: sort, limit: *limit, offset: *offset, @@ -75,8 +80,6 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_scan( let proj_names = extract_projection_names(projection, window_functions); let sort = convert_sort_keys(sort_keys); let vshard = VShardId::from_collection_in_database(database_id, collection); - let computed_bytes = extract_computed_columns(projection, window_functions)?; - let window_bytes = serialize_window_functions(window_functions)?; let physical = match engine { EngineType::Timeseries => { diff --git a/nodedb/src/control/planner/sql_plan_convert/set_ops.rs b/nodedb/src/control/planner/sql_plan_convert/set_ops.rs index 9e43b00ef..4a9a81e7f 100644 --- a/nodedb/src/control/planner/sql_plan_convert/set_ops.rs +++ b/nodedb/src/control/planner/sql_plan_convert/set_ops.rs @@ -47,6 +47,8 @@ pub(super) fn convert_constant_result( rows: payload, filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, @@ -317,6 +319,8 @@ pub(super) fn convert_subquery( input: Box::new(child), filters: super::filter::serialize_filters(filters)?, projection: lower_subquery_projection(projection)?, + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: lower_subquery_sort_keys(sort_keys, merged_doc_body), limit, offset, diff --git a/nodedb/src/control/server/exchange/resolve/exchange/dispatch.rs b/nodedb/src/control/server/exchange/resolve/exchange/dispatch.rs index c312c65c4..fcdccdade 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/dispatch.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/dispatch.rs @@ -157,6 +157,8 @@ pub(super) async fn resolve_exchange( input, filters, projection, + computed_columns, + window_functions, sort_keys, limit, offset, @@ -170,6 +172,8 @@ pub(super) async fn resolve_exchange( input, filters, projection, + computed_columns, + window_functions, sort_keys, limit, offset, diff --git a/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs b/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs index 9e4920724..0bc66c706 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs @@ -29,6 +29,8 @@ pub(super) struct PostProcessFields { pub input: Box, pub filters: Vec, pub projection: Vec, + pub computed_columns: Vec, + pub window_functions: Vec, pub sort_keys: Vec, pub limit: Option, pub offset: usize, @@ -107,6 +109,8 @@ pub(super) async fn resolve_post_process( input, filters, projection, + computed_columns, + window_functions, sort_keys, limit, offset, @@ -223,6 +227,8 @@ pub(super) async fn resolve_post_process( rows, filters, projection, + computed_columns, + window_functions, sort_keys, limit, offset, diff --git a/nodedb/src/control/server/exchange/resolve/join_input.rs b/nodedb/src/control/server/exchange/resolve/join_input.rs index 16a50d81b..2d3e1ab32 100644 --- a/nodedb/src/control/server/exchange/resolve/join_input.rs +++ b/nodedb/src/control/server/exchange/resolve/join_input.rs @@ -60,6 +60,8 @@ pub(super) async fn resolve_join_input( rows: flatten_to_relational_rows(&outcome.merged_array), filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, @@ -134,6 +136,8 @@ pub(super) async fn resolve_join_input( rows: flatten_to_relational_rows(&merged), filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, @@ -236,6 +240,8 @@ pub(super) async fn gather_join_build_side( rows: flatten_to_relational_rows(&outcome.merged_array), filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, diff --git a/nodedb/src/control/server/exchange/resolve/materialize.rs b/nodedb/src/control/server/exchange/resolve/materialize.rs index 1d2a2c417..a4b53518d 100644 --- a/nodedb/src/control/server/exchange/resolve/materialize.rs +++ b/nodedb/src/control/server/exchange/resolve/materialize.rs @@ -34,6 +34,8 @@ pub(super) async fn materialize_providers( rows: _, filters, projection, + computed_columns, + window_functions, sort_keys, limit, offset, @@ -46,6 +48,8 @@ pub(super) async fn materialize_providers( rows: encoded, filters, projection, + computed_columns, + window_functions, sort_keys, limit, offset, @@ -223,6 +227,8 @@ pub(super) async fn materialize_providers( input, filters, projection, + computed_columns, + window_functions, sort_keys, limit, offset, @@ -233,6 +239,8 @@ pub(super) async fn materialize_providers( input: Box::new(input), filters, projection, + computed_columns, + window_functions, sort_keys, limit, offset, diff --git a/nodedb/src/control/server/shared/authorization/requirements.rs b/nodedb/src/control/server/shared/authorization/requirements.rs index dc7b1e6c6..7655de50f 100644 --- a/nodedb/src/control/server/shared/authorization/requirements.rs +++ b/nodedb/src/control/server/shared/authorization/requirements.rs @@ -73,6 +73,8 @@ mod tests { rows: Vec::new(), filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, diff --git a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs index 29c9350d4..fe9387d2c 100644 --- a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs +++ b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs @@ -1520,6 +1520,8 @@ mod tests { rows: Vec::new(), filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, diff --git a/nodedb/src/data/executor/dispatch/query.rs b/nodedb/src/data/executor/dispatch/query.rs index 180a112d0..ac9d9115d 100644 --- a/nodedb/src/data/executor/dispatch/query.rs +++ b/nodedb/src/data/executor/dispatch/query.rs @@ -71,6 +71,8 @@ impl CoreLoop { rows, filters, projection, + computed_columns, + window_functions, sort_keys, limit, offset, @@ -82,6 +84,8 @@ impl CoreLoop { rows_bytes: rows, filters_bytes: filters, projection, + computed_columns_bytes: computed_columns, + window_functions_bytes: window_functions, sort_keys, limit: *limit, offset: *offset, diff --git a/nodedb/src/data/executor/handlers/mod.rs b/nodedb/src/data/executor/handlers/mod.rs index 40935615e..3e839f49a 100644 --- a/nodedb/src/data/executor/handlers/mod.rs +++ b/nodedb/src/data/executor/handlers/mod.rs @@ -40,6 +40,7 @@ pub(super) mod merge_helpers; pub(super) mod merge_orchestrated; pub mod point; pub(super) mod provider_scan; +pub(super) mod provider_scan_compute; pub mod purge; pub mod query_collection_size; pub mod reclaim; diff --git a/nodedb/src/data/executor/handlers/provider_scan.rs b/nodedb/src/data/executor/handlers/provider_scan.rs index b48f92e9d..51af25c83 100644 --- a/nodedb/src/data/executor/handlers/provider_scan.rs +++ b/nodedb/src/data/executor/handlers/provider_scan.rs @@ -2,15 +2,19 @@ //! Executor handler for `QueryOp::ProviderScan`. //! -//! Decodes the pre-materialized msgpack row array, applies predicate filtering, -//! offset, sort, distinct deduplication, column projection, and limit — in that -//! order — then emits the resulting rows via `response_with_payload`. +//! Decodes the pre-materialized msgpack row array, applies predicate +//! filtering, sort, window functions + computed columns, distinct +//! deduplication, offset, column projection, and limit — in that order — +//! then emits the resulting rows via `response_with_payload`. Sort runs +//! before offset so `ORDER BY ... OFFSET n` skips the first `n` rows of the +//! sorted set, not the decoded set. use nodedb_query::msgpack_scan; use crate::bridge::envelope::{ErrorCode, Response}; use crate::bridge::scan_filter::ScanFilter; use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::provider_scan_compute::apply_windows_and_computed; use crate::data::executor::handlers::sort_utils::sort_msgpack_rows; use crate::data::executor::msgpack_utils::write_str; use crate::data::executor::response_codec::encode_binary_rows; @@ -21,6 +25,8 @@ pub(in crate::data::executor) struct ProviderScanParams<'a> { pub rows_bytes: &'a [u8], pub filters_bytes: &'a [u8], pub projection: &'a [String], + pub computed_columns_bytes: &'a [u8], + pub window_functions_bytes: &'a [u8], pub sort_keys: &'a [nodedb_physical::physical_plan::SortKeySpec], pub limit: Option, pub offset: usize, @@ -30,8 +36,8 @@ pub(in crate::data::executor) struct ProviderScanParams<'a> { impl CoreLoop { /// Execute a `ProviderScan` plan node. /// - /// Processing order: decode rows → filter → offset → sort → distinct → - /// project → limit → emit. + /// Processing order: decode rows → filter → sort → windows + computed + /// columns → distinct → offset → project → limit → emit. pub(in crate::data::executor) fn execute_provider_scan( &mut self, task: &ExecutionTask, @@ -41,6 +47,8 @@ impl CoreLoop { rows_bytes, filters_bytes, projection, + computed_columns_bytes, + window_functions_bytes, sort_keys, limit, offset, @@ -92,22 +100,29 @@ impl CoreLoop { } } - // ── 3. Offset. ──────────────────────────────────────────────────────── - if offset > 0 { - if offset >= rows.len() { - rows.clear(); - } else { - rows.drain(..offset); - } - } - - // ── 4. Sort. ────────────────────────────────────────────────────────── + // ── 3. Sort. ────────────────────────────────────────────────────────── + // Runs before offset: `ORDER BY ... OFFSET n` skips the first `n` rows + // of the SORTED set, not the decoded set. if !sort_keys.is_empty() && let Err(e) = sort_msgpack_rows(&mut rows, sort_keys) { return self.response_error(task, crate::Error::from(e)); } + // ── 4. Window functions + computed columns. ───────────────────────── + // Skipped entirely (zero-decode msgpack path) when both byte slices + // are empty. + if !window_functions_bytes.is_empty() || !computed_columns_bytes.is_empty() { + rows = match apply_windows_and_computed( + rows, + window_functions_bytes, + computed_columns_bytes, + ) { + Ok(r) => r, + Err(e) => return self.response_error(task, e), + }; + } + // ── 5. Distinct (on the would-be projected row). ────────────────────── // Deduplicate on the projected shape so SQL DISTINCT semantics are // honoured: two rows with the same projected columns but different @@ -124,7 +139,16 @@ impl CoreLoop { }); } - // ── 6. Project. ─────────────────────────────────────────────────────── + // ── 6. Offset. ──────────────────────────────────────────────────────── + if offset > 0 { + if offset >= rows.len() { + rows.clear(); + } else { + rows.drain(..offset); + } + } + + // ── 7. Project. ─────────────────────────────────────────────────────── let rows: Vec> = if projection.is_empty() { rows } else { @@ -133,14 +157,14 @@ impl CoreLoop { .collect() }; - // ── 7. Limit. ───────────────────────────────────────────────────────── + // ── 8. Limit. ───────────────────────────────────────────────────────── let rows = if let Some(n) = limit { rows.into_iter().take(n).collect() } else { rows }; - // ── 8. Emit. ────────────────────────────────────────────────────────── + // ── 9. Emit. ────────────────────────────────────────────────────────── let payload = encode_binary_rows(&rows); self.response_with_payload(task, payload) } diff --git a/nodedb/src/data/executor/handlers/provider_scan_compute.rs b/nodedb/src/data/executor/handlers/provider_scan_compute.rs new file mode 100644 index 000000000..64284f960 --- /dev/null +++ b/nodedb/src/data/executor/handlers/provider_scan_compute.rs @@ -0,0 +1,100 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Window-function and computed-column evaluation for `QueryOp::ProviderScan`. +//! +//! Runs after sort and before distinct/offset/project/limit in the +//! `ProviderScan` pipeline. Each msgpack row decodes to a `serde_json::Value`, +//! windows evaluate over the full row set, computed columns evaluate +//! per-row, and the result re-encodes to msgpack. Skipped entirely when both +//! byte slices are empty, so the zero-decode msgpack path stays untouched for +//! a plain relational scan. + +use crate::bridge::envelope::ErrorCode; +use crate::bridge::expr_eval::ComputedColumn; +use crate::bridge::window_func::{WindowFuncSpec, evaluate_window_functions}; + +/// Decode a `Vec` from MessagePack, tagging decode failures with which +/// byte slice (`kind`, e.g. `"window"` or `"computed"`) failed. +fn decode_bytes<'a, T: zerompk::FromMessagePack<'a>>( + bytes: &'a [u8], + kind: &str, +) -> crate::Result { + zerompk::from_msgpack(bytes).map_err(|e| { + crate::Error::DataPlane(ErrorCode::Internal { + detail: format!("ProviderScan: malformed {kind} bytes: {e}"), + }) + }) +} + +/// Apply window functions then computed columns to `rows`, both optional and +/// independently controlled by `window_bytes` / `computed_bytes` being +/// non-empty. Returns `rows` unchanged, still msgpack-encoded, when both are +/// empty. +pub(in crate::data::executor) fn apply_windows_and_computed( + rows: Vec>, + window_bytes: &[u8], + computed_bytes: &[u8], +) -> crate::Result>> { + if window_bytes.is_empty() && computed_bytes.is_empty() { + return Ok(rows); + } + + let window_specs: Vec = if window_bytes.is_empty() { + Vec::new() + } else { + decode_bytes(window_bytes, "window")? + }; + let computed_cols: Vec = if computed_bytes.is_empty() { + Vec::new() + } else { + decode_bytes(computed_bytes, "computed")? + }; + + let mut json_rows: Vec<(String, serde_json::Value)> = Vec::with_capacity(rows.len()); + for (idx, row) in rows.iter().enumerate() { + let value = nodedb_types::value_from_msgpack(row).map_err(|e| { + crate::Error::DataPlane(ErrorCode::Internal { + detail: format!("ProviderScan: malformed row for window/computed evaluation: {e}"), + }) + })?; + json_rows.push((idx.to_string(), serde_json::Value::from(value))); + } + + if !window_specs.is_empty() { + evaluate_window_functions(&mut json_rows, &window_specs).map_err(crate::Error::from)?; + } + + for (_, row_json) in &mut json_rows { + if computed_cols.is_empty() { + continue; + } + // Every computed column evaluates against the row as it stood before + // this loop, matching `apply_projection`'s semantics: later computed + // columns never observe earlier ones' results. + let doc_val = nodedb_types::Value::from(row_json.clone()); + for cc in &computed_cols { + let already_present = matches!(row_json.get(&cc.alias), Some(v) if !v.is_null()); + if already_present { + continue; + } + let v = cc.expr.eval(&doc_val)?; + if let serde_json::Value::Object(obj) = row_json { + obj.insert(cc.alias.clone(), serde_json::Value::from(v)); + } + } + } + + let mut out = Vec::with_capacity(json_rows.len()); + for (_, row_json) in json_rows { + let bytes = nodedb_types::json_to_msgpack(&row_json).map_err(|e| { + crate::Error::DataPlane(ErrorCode::Internal { + detail: format!( + "ProviderScan: failed to re-encode row after window/computed evaluation: {e}" + ), + }) + })?; + out.push(bytes); + } + + Ok(out) +} diff --git a/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/inline_hash_join.rs b/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/inline_hash_join.rs index 8ce8e1420..5f30bf18d 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/inline_hash_join.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/inline_hash_join.rs @@ -171,6 +171,8 @@ fn inline_hash_join_honors_qualified_left_keys() { rows: response_codec::flatten_to_relational_rows(&left_data), filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, @@ -181,6 +183,8 @@ fn inline_hash_join_honors_qualified_left_keys() { rows: response_codec::flatten_to_relational_rows(&right_data), filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, diff --git a/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/multi_core_joins.rs b/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/multi_core_joins.rs index 0452152fc..321116d8b 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/multi_core_joins.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/multi_core_joins.rs @@ -126,6 +126,8 @@ fn multi_core_broadcast_inner_join() { rows: response_codec::flatten_to_relational_rows(&phase1_payload), filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, @@ -268,6 +270,8 @@ fn multi_core_broadcast_left_join() { rows: response_codec::flatten_to_relational_rows(&phase1_payload), filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, @@ -450,6 +454,8 @@ fn multi_core_broadcast_merge_simulation() { rows: response_codec::flatten_to_relational_rows(&data), filters: Vec::new(), projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), sort_keys: Vec::new(), limit: None, offset: 0, From 1699789ee5a4d9a3087f3e668941ff1b19bc13a2 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Thu, 17 Sep 2026 22:08:59 +0800 Subject: [PATCH 04/15] fix(query): order window partitions by each spec's own ORDER BY Window functions previously ranked/aggregated rows in row-array arrival order, ignoring the spec's ORDER BY when it differed from how rows were already sorted. build_partitions and build_value_partitions now sort each partition's row indices by the spec's ORDER BY (with PostgreSQL-style NULL placement) after grouping, while leaving the row array itself untouched. The Value-native partition builder moves into its own value_partition module alongside the shared null_order helper. --- nodedb-query/src/window/eval.rs | 40 ++++++- nodedb-query/src/window/helpers.rs | 115 +++++++++++++++++---- nodedb-query/src/window/mod.rs | 1 + nodedb-query/src/window/value_eval.rs | 61 ++++------- nodedb-query/src/window/value_partition.rs | 102 ++++++++++++++++++ 5 files changed, 252 insertions(+), 67 deletions(-) create mode 100644 nodedb-query/src/window/value_partition.rs diff --git a/nodedb-query/src/window/eval.rs b/nodedb-query/src/window/eval.rs index cf6085ad6..ad741cdf3 100644 --- a/nodedb-query/src/window/eval.rs +++ b/nodedb-query/src/window/eval.rs @@ -15,7 +15,8 @@ use super::spec::WindowFuncSpec; /// /// `rows` is the sorted result set. Each row is a `(doc_id, serde_json::Value)`. /// The same rows are mutated in place with window columns appended to each -/// document. +/// document. The row array keeps its input order; each spec's partitions are +/// ordered by that spec's own ORDER BY, independent of the row array order. /// /// Unknown window function names must be rejected by the planner before /// reaching this dispatcher; an unrecognised name here is an internal bug @@ -29,7 +30,7 @@ pub fn evaluate_window_functions( specs: &[WindowFuncSpec], ) -> Result<(), crate::expr::EvalError> { for spec in specs { - let partitions = build_partitions(rows, &spec.partition_by)?; + let partitions = build_partitions(rows, &spec.partition_by, &spec.order_by)?; for partition_indices in &partitions { match spec.func_name.as_str() { @@ -146,9 +147,11 @@ mod tests { frame: WindowFrame::default(), }; evaluate_window_functions(&mut rows, &[spec]).unwrap(); - assert_eq!(rows[0].1["running_total"], json!(100.0)); - assert_eq!(rows[1].1["running_total"], json!(220.0)); - assert_eq!(rows[2].1["running_total"], json!(310.0)); + // The frame runs in salary order within each dept, not in row + // arrival order: eng = Carol(90) → Alice(100) → Bob(120). + assert_eq!(rows[0].1["running_total"], json!(190.0)); + assert_eq!(rows[1].1["running_total"], json!(310.0)); + assert_eq!(rows[2].1["running_total"], json!(90.0)); assert_eq!(rows[3].1["running_total"], json!(80.0)); assert_eq!(rows[4].1["running_total"], json!(190.0)); } @@ -262,6 +265,33 @@ mod tests { assert_eq!(rows[4].1["nv"], json!(2)); } + #[test] + fn rank_orders_by_spec_order_by_not_row_arrival_order() { + // Rows arrive as Alice(100), Bob(120), Carol(90) within dept "eng" — + // not sorted by salary. RANK() OVER (ORDER BY salary DESC) must rank + // by salary, and the row array order must stay unchanged. + let mut rows = make_rows(); + let spec = WindowFuncSpec { + alias: "rnk".into(), + func_name: "rank".into(), + args: vec![], + partition_by: vec![SqlExpr::Column("dept".into())], + order_by: vec![(SqlExpr::Column("salary".into()), false)], + frame: WindowFrame::default(), + }; + evaluate_window_functions(&mut rows, &[spec]).unwrap(); + assert_eq!(rows[0].1["name"], json!("Alice")); + assert_eq!(rows[1].1["name"], json!("Bob")); + assert_eq!(rows[2].1["name"], json!("Carol")); + assert_eq!(rows[3].1["name"], json!("Dave")); + assert_eq!(rows[4].1["name"], json!("Eve")); + assert_eq!(rows[0].1["rnk"], json!(2)); // Alice, salary 100 + assert_eq!(rows[1].1["rnk"], json!(1)); // Bob, salary 120 + assert_eq!(rows[2].1["rnk"], json!(3)); // Carol, salary 90 + assert_eq!(rows[3].1["rnk"], json!(2)); // Dave, salary 80 + assert_eq!(rows[4].1["rnk"], json!(1)); // Eve, salary 110 + } + #[test] #[should_panic(expected = "should have been rejected at planning time")] fn unknown_function_panics_at_evaluator() { diff --git a/nodedb-query/src/window/helpers.rs b/nodedb-query/src/window/helpers.rs index 3bd7b36fb..487d30c4f 100644 --- a/nodedb-query/src/window/helpers.rs +++ b/nodedb-query/src/window/helpers.rs @@ -6,35 +6,112 @@ use std::collections::HashMap; use crate::expr::types::SqlExpr; -/// Group row indices by partition key, preserving first-seen partition order. +/// Group row indices by partition key, preserving first-seen partition order, +/// then sort each partition's indices by the spec's ORDER BY. /// -/// A division/modulo-by-zero in a PARTITION BY expression propagates as -/// `Err(EvalError::DivisionByZero)` rather than being folded to NULL. +/// The returned index lists are ordered by `order_by`; the `rows` array +/// itself keeps its input order — only the per-partition index lists move. +/// +/// A division/modulo-by-zero in a PARTITION BY or ORDER BY expression +/// propagates as `Err(EvalError::DivisionByZero)` rather than being folded to +/// NULL. pub(super) fn build_partitions( rows: &[(String, serde_json::Value)], partition_by: &[SqlExpr], + order_by: &[(SqlExpr, bool)], ) -> Result>, crate::expr::EvalError> { - if partition_by.is_empty() { - return Ok(vec![(0..rows.len()).collect()]); - } + let mut partitions = if partition_by.is_empty() { + vec![(0..rows.len()).collect::>()] + } else { + let mut groups: HashMap> = HashMap::new(); + let mut order = Vec::new(); + + for (i, (_id, doc)) in rows.iter().enumerate() { + let key: String = partition_by + .iter() + .map(|expr| eval_expr_on_json(expr, doc).map(|v| v.to_string())) + .collect::, _>>()? + .join("\x00"); + let entry = groups.entry(key.clone()).or_default(); + if entry.is_empty() { + order.push(key); + } + entry.push(i); + } + + order.iter().filter_map(|k| groups.remove(k)).collect() + }; - let mut groups: HashMap> = HashMap::new(); - let mut order = Vec::new(); + if !order_by.is_empty() { + let mut keys: Vec> = Vec::with_capacity(rows.len()); + for (_id, doc) in rows.iter() { + keys.push( + order_by + .iter() + .map(|(expr, _)| eval_expr_on_json(expr, doc)) + .collect::, _>>()?, + ); + } - for (i, (_id, doc)) in rows.iter().enumerate() { - let key: String = partition_by - .iter() - .map(|expr| eval_expr_on_json(expr, doc).map(|v| v.to_string())) - .collect::, _>>()? - .join("\x00"); - let entry = groups.entry(key.clone()).or_default(); - if entry.is_empty() { - order.push(key); + for partition in &mut partitions { + partition.sort_by(|&a, &b| compare_order_keys(&keys[a], &keys[b], order_by)); } - entry.push(i); } - Ok(order.iter().filter_map(|k| groups.remove(k)).collect()) + Ok(partitions) +} + +/// Decide NULL placement for one ORDER BY column, shared by every window +/// evaluator's `compare_order_keys`. +/// +/// NULL placement follows PostgreSQL's default: ASC places NULLs last, DESC +/// places NULLs first. A window spec carries no explicit NULLS FIRST/LAST +/// override, so this default is fixed by direction alone. Returns `None` +/// when neither value is NULL, leaving the non-null comparison to the +/// caller. +pub(super) fn null_order( + a_null: bool, + b_null: bool, + ascending: bool, +) -> Option { + use std::cmp::Ordering; + let nulls_first = !ascending; + match (a_null, b_null) { + (true, true) => Some(Ordering::Equal), + (true, false) => Some(if nulls_first { + Ordering::Less + } else { + Ordering::Greater + }), + (false, true) => Some(if nulls_first { + Ordering::Greater + } else { + Ordering::Less + }), + (false, false) => None, + } +} + +/// Compare two rows' pre-evaluated ORDER BY keys. +fn compare_order_keys( + a: &[serde_json::Value], + b: &[serde_json::Value], + order_by: &[(SqlExpr, bool)], +) -> std::cmp::Ordering { + use std::cmp::Ordering; + for (idx, (_, ascending)) in order_by.iter().enumerate() { + let (Some(va), Some(vb)) = (a.get(idx), b.get(idx)) else { + continue; + }; + let ord = null_order(va.is_null(), vb.is_null(), *ascending).unwrap_or_else(|| { + let c = crate::json_expr::compare_json(va, vb); + if *ascending { c } else { c.reverse() } + }); + if ord != Ordering::Equal { + return ord; + } + } + Ordering::Equal } pub(super) fn set_window_col(row: &mut serde_json::Value, alias: &str, val: serde_json::Value) { diff --git a/nodedb-query/src/window/mod.rs b/nodedb-query/src/window/mod.rs index 47cfe48b0..f1e4f5dda 100644 --- a/nodedb-query/src/window/mod.rs +++ b/nodedb-query/src/window/mod.rs @@ -16,6 +16,7 @@ pub mod running; pub mod spec; pub mod value_agg; pub mod value_eval; +pub mod value_partition; pub use eval::evaluate_window_functions; pub use spec::{FrameBound, WindowFrame, WindowFuncSpec}; diff --git a/nodedb-query/src/window/value_eval.rs b/nodedb-query/src/window/value_eval.rs index c22230d94..1ba2da874 100644 --- a/nodedb-query/src/window/value_eval.rs +++ b/nodedb-query/src/window/value_eval.rs @@ -12,6 +12,7 @@ use nodedb_types::Value; use super::spec::WindowFuncSpec; use super::value_agg::apply_v_aggregate; +use super::value_partition::build_value_partitions; use crate::expr::types::SqlExpr; use crate::value_ops::compare_values; @@ -35,7 +36,9 @@ pub enum WindowError { /// /// `column_index` maps column name → position in each row slice. /// For each spec, one `Value` is appended to every row. Returns the list of -/// new column names, one per spec in spec order. +/// new column names, one per spec in spec order. `rows` keeps its input +/// order; each spec's partitions are ordered by that spec's own ORDER BY, +/// independent of the row order. pub fn evaluate_window_functions_value( rows: &mut [Vec], column_index: &HashMap, @@ -91,48 +94,7 @@ pub fn evaluate_window_functions_value( Ok(new_cols) } -// ── Partition building ──────────────────────────────────────────────────────── - -fn build_value_partitions( - rows: &[Vec], - column_index: &HashMap, - spec: &WindowFuncSpec, -) -> Result>, WindowError> { - if spec.partition_by.is_empty() { - return Ok(vec![(0..rows.len()).collect()]); - } - - let mut groups: HashMap> = HashMap::new(); - let mut order: Vec = Vec::new(); - - for (i, row) in rows.iter().enumerate() { - let key = partition_key(row, column_index, &spec.partition_by)?; - let entry = groups.entry(key.clone()).or_default(); - if entry.is_empty() { - order.push(key); - } - entry.push(i); - } - - Ok(order.iter().filter_map(|k| groups.remove(k)).collect()) -} - -fn partition_key( - row: &[Value], - column_index: &HashMap, - partition_by: &[SqlExpr], -) -> Result { - Ok(partition_by - .iter() - .map(|expr| { - let v = eval_arg_for_row(expr, row, column_index)?; - Ok(format!("{v:?}")) - }) - .collect::, WindowError>>()? - .join("\x00")) -} - -// ── Value comparison helpers (pub(super) for value_agg) ─────────────────────── +// ── Value comparison helpers (pub(super) for value_agg, value_partition) ───── pub(super) fn cmp_values(a: &Value, b: &Value) -> std::cmp::Ordering { match (a, b) { @@ -619,6 +581,19 @@ mod tests { assert_eq!(out_int(&rows, 2), vec![1, 2, 1]); } + #[test] + fn rank_orders_by_spec_order_by_not_row_arrival_order() { + // Rows arrive as 10, 30, 20 — not sorted by value. RANK() OVER + // (ORDER BY v DESC) must rank 30(1), 20(2), 10(3); the row array + // order must stay 10, 30, 20. + let mut rows = rows_v(&[10, 30, 20]); + let cols = ci(&["v"]); + let s = spec("rank", vec![], vec![], vec![(col("v"), false)]); + evaluate_window_functions_value(&mut rows, &cols, &[s]).unwrap(); + assert_eq!(out_int(&rows, 0), vec![10, 30, 20]); + assert_eq!(out_int(&rows, 1), vec![3, 1, 2]); + } + #[test] fn unknown_function_errors() { let mut rows = rows_v(&[1]); diff --git a/nodedb-query/src/window/value_partition.rs b/nodedb-query/src/window/value_partition.rs new file mode 100644 index 000000000..3a7a5d0c3 --- /dev/null +++ b/nodedb-query/src/window/value_partition.rs @@ -0,0 +1,102 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Partition building and per-partition ORDER BY sort for the Value-native +//! window evaluator. + +use std::collections::HashMap; + +use nodedb_types::Value; + +use super::helpers::null_order; +use super::spec::WindowFuncSpec; +use super::value_eval::{WindowError, cmp_values, eval_arg_for_row}; +use crate::expr::types::SqlExpr; + +/// Group row indices by partition key, preserving first-seen partition +/// order, then sort each partition's indices by the spec's ORDER BY. +/// +/// The returned index lists are ordered by `spec.order_by`; `rows` itself +/// keeps its input order — only the per-partition index lists move. +pub(super) fn build_value_partitions( + rows: &[Vec], + column_index: &HashMap, + spec: &WindowFuncSpec, +) -> Result>, WindowError> { + let mut partitions = if spec.partition_by.is_empty() { + vec![(0..rows.len()).collect::>()] + } else { + let mut groups: HashMap> = HashMap::new(); + let mut order: Vec = Vec::new(); + + for (i, row) in rows.iter().enumerate() { + let key = partition_key(row, column_index, &spec.partition_by)?; + let entry = groups.entry(key.clone()).or_default(); + if entry.is_empty() { + order.push(key); + } + entry.push(i); + } + + order.iter().filter_map(|k| groups.remove(k)).collect() + }; + + if !spec.order_by.is_empty() { + let mut keys: Vec> = Vec::with_capacity(rows.len()); + for row in rows.iter() { + keys.push( + spec.order_by + .iter() + .map(|(expr, _)| eval_arg_for_row(expr, row, column_index)) + .collect::, _>>()?, + ); + } + + for partition in &mut partitions { + partition.sort_by(|&a, &b| compare_order_keys(&keys[a], &keys[b], &spec.order_by)); + } + } + + Ok(partitions) +} + +fn partition_key( + row: &[Value], + column_index: &HashMap, + partition_by: &[SqlExpr], +) -> Result { + Ok(partition_by + .iter() + .map(|expr| { + let v = eval_arg_for_row(expr, row, column_index)?; + Ok(format!("{v:?}")) + }) + .collect::, WindowError>>()? + .join("\x00")) +} + +/// Compare two rows' pre-evaluated ORDER BY keys. +fn compare_order_keys( + a: &[Value], + b: &[Value], + order_by: &[(SqlExpr, bool)], +) -> std::cmp::Ordering { + use std::cmp::Ordering; + for (idx, (_, ascending)) in order_by.iter().enumerate() { + let (Some(va), Some(vb)) = (a.get(idx), b.get(idx)) else { + continue; + }; + let ord = null_order( + matches!(va, Value::Null), + matches!(vb, Value::Null), + *ascending, + ) + .unwrap_or_else(|| { + let c = cmp_values(va, vb); + if *ascending { c } else { c.reverse() } + }); + if ord != Ordering::Equal { + return ord; + } + } + Ordering::Equal +} From 2cbe9bf786402bcb8cdbc0b4ee527553af2b2ee2 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Thu, 17 Sep 2026 23:02:00 +0800 Subject: [PATCH 05/15] feat(query): evaluate window functions over subquery and CTE tails Add a window_functions field to SqlPlan::Subquery and thread it through catalog folding/validation, the plan visitor, and CTE inlining, so a derived table or CTE reference carrying window specs merges or wraps correctly instead of dropping them silently. Subquery-to-physical-plan lowering now serializes computed columns and window specs for the tail, selecting qualified vs unqualified column references based on whether the body emits merged join/lateral documents. The ProviderScan executor evaluates windows and computed columns before sort instead of after, so ORDER BY can reference a window or computed alias. --- nodedb-sql/src/planner/catalog_fold.rs | 3 + .../src/planner/catalog_plan_validate.rs | 2 + nodedb-sql/src/planner/select/entry.rs | 2 + nodedb-sql/src/planner/select/post_process.rs | 2 + nodedb-sql/src/types/plan/variants.rs | 2 + nodedb-sql/src/visitor/plan_visitor/args.rs | 1 + .../src/visitor/plan_visitor/dispatch_rest.rs | 2 + .../sql_plan_convert/aggregate/plan.rs | 4 +- .../sql_plan_convert/aggregate/projection.rs | 32 +- .../sql_plan_convert/expr/inline_cte.rs | 572 ++++++++++++++---- .../planner/sql_plan_convert/scan/core.rs | 4 +- .../sql_plan_convert/scan/timeseries.rs | 2 +- .../planner/sql_plan_convert/set_ops.rs | 48 +- .../data/executor/handlers/provider_scan.rs | 39 +- .../handlers/provider_scan_compute.rs | 2 +- 15 files changed, 544 insertions(+), 173 deletions(-) diff --git a/nodedb-sql/src/planner/catalog_fold.rs b/nodedb-sql/src/planner/catalog_fold.rs index d29db868b..5aca0895d 100644 --- a/nodedb-sql/src/planner/catalog_fold.rs +++ b/nodedb-sql/src/planner/catalog_fold.rs @@ -112,6 +112,7 @@ fn walk_plan( input, mut filters, mut projection, + mut window_functions, mut sort_keys, offset, distinct, @@ -121,11 +122,13 @@ fn walk_plan( fold_filter(f, catalog, database_id, tenant_id); } fold_projection(&mut projection, catalog, database_id, tenant_id); + fold_windows(&mut window_functions, catalog, database_id, tenant_id); fold_sort_keys(&mut sort_keys, catalog, database_id, tenant_id); SqlPlan::Subquery { input: Box::new(walk_plan(*input, catalog, database_id, tenant_id)), filters, projection, + window_functions, sort_keys, offset, distinct, diff --git a/nodedb-sql/src/planner/catalog_plan_validate.rs b/nodedb-sql/src/planner/catalog_plan_validate.rs index 8f947256e..fb0a24f18 100644 --- a/nodedb-sql/src/planner/catalog_plan_validate.rs +++ b/nodedb-sql/src/planner/catalog_plan_validate.rs @@ -106,12 +106,14 @@ pub(super) fn validate_catalog_exprs( input, filters, projection, + window_functions, sort_keys, .. } => { validate_catalog_exprs(input, catalog, database_id, tenant_id)?; validate_filters(filters, catalog, database_id, tenant_id)?; validate_projection(projection, catalog, database_id, tenant_id)?; + validate_windows(window_functions, catalog, database_id, tenant_id)?; validate_sort_keys(sort_keys, catalog, database_id, tenant_id)?; } SqlPlan::Join { diff --git a/nodedb-sql/src/planner/select/entry.rs b/nodedb-sql/src/planner/select/entry.rs index 48a41f8f6..70a2dc426 100644 --- a/nodedb-sql/src/planner/select/entry.rs +++ b/nodedb-sql/src/planner/select/entry.rs @@ -173,6 +173,7 @@ pub fn plan_query( SqlPlan::Subquery { filters, projection, + window_functions, sort_keys, offset, distinct, @@ -182,6 +183,7 @@ pub fn plan_query( input: Box::new(upgraded_leaf), filters, projection, + window_functions, sort_keys, offset, distinct, diff --git a/nodedb-sql/src/planner/select/post_process.rs b/nodedb-sql/src/planner/select/post_process.rs index 3d351cefa..fa3022587 100644 --- a/nodedb-sql/src/planner/select/post_process.rs +++ b/nodedb-sql/src/planner/select/post_process.rs @@ -37,6 +37,8 @@ pub(in crate::planner::select) fn post_process( input: Box::new(input), filters: Vec::new(), projection, + // The body keeps its own window specs; the tail evaluates none. + window_functions: Vec::new(), sort_keys, offset, distinct: false, diff --git a/nodedb-sql/src/types/plan/variants.rs b/nodedb-sql/src/types/plan/variants.rs index 7439de518..87ed75533 100644 --- a/nodedb-sql/src/types/plan/variants.rs +++ b/nodedb-sql/src/types/plan/variants.rs @@ -537,6 +537,8 @@ pub enum SqlPlan { filters: Vec, /// Outer projection (target list). Empty = inherit the body's columns. projection: Vec, + /// Window functions evaluated over the post-processed rows. Empty = none. + window_functions: Vec, /// Outer `ORDER BY` keys applied over the materialized rows. sort_keys: Vec, /// Outer `OFFSET` (0 = none). diff --git a/nodedb-sql/src/visitor/plan_visitor/args.rs b/nodedb-sql/src/visitor/plan_visitor/args.rs index f3678a436..2c89436e7 100644 --- a/nodedb-sql/src/visitor/plan_visitor/args.rs +++ b/nodedb-sql/src/visitor/plan_visitor/args.rs @@ -37,6 +37,7 @@ pub struct SubqueryVisitArgs<'a> { pub input: &'a SqlPlan, pub filters: &'a [Filter], pub projection: &'a [Projection], + pub window_functions: &'a [WindowSpec], pub sort_keys: &'a [SortKey], pub offset: usize, pub distinct: bool, diff --git a/nodedb-sql/src/visitor/plan_visitor/dispatch_rest.rs b/nodedb-sql/src/visitor/plan_visitor/dispatch_rest.rs index 0f5f61452..1a24ee07b 100644 --- a/nodedb-sql/src/visitor/plan_visitor/dispatch_rest.rs +++ b/nodedb-sql/src/visitor/plan_visitor/dispatch_rest.rs @@ -26,6 +26,7 @@ pub(super) fn dispatch_rest( input, filters, projection, + window_functions, sort_keys, offset, distinct, @@ -34,6 +35,7 @@ pub(super) fn dispatch_rest( input, filters, projection, + window_functions, sort_keys, offset: *offset, distinct: *distinct, diff --git a/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs b/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs index aad2dc7a9..e9bb60684 100644 --- a/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs +++ b/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs @@ -401,8 +401,8 @@ mod tests { vec!["name".to_string(), "rn".to_string()] ); - let computed_bytes = - extract_computed_columns(&projection, &window_functions).expect("serialize computed"); + let computed_bytes = extract_computed_columns(&projection, &window_functions, false) + .expect("serialize computed"); let computed: Vec = zerompk::from_msgpack(&computed_bytes).expect("deserialize computed"); diff --git a/nodedb/src/control/planner/sql_plan_convert/aggregate/projection.rs b/nodedb/src/control/planner/sql_plan_convert/aggregate/projection.rs index 2591a3d9f..bdb5152a5 100644 --- a/nodedb/src/control/planner/sql_plan_convert/aggregate/projection.rs +++ b/nodedb/src/control/planner/sql_plan_convert/aggregate/projection.rs @@ -125,10 +125,28 @@ fn encode_computed_columns( }) } +/// Pick the bridge-expression converter for the row shape the expression runs +/// against. A join / lateral body emits merged documents keyed by `t.col`, so +/// its column references keep their table qualifier. +fn bridge_expr_converter(qualified: bool) -> fn(&SqlExpr) -> crate::bridge::expr_eval::SqlExpr { + if qualified { + sql_expr_to_bridge_expr_qualified + } else { + sql_expr_to_bridge_expr + } +} + +/// Serialize the computed (non-window) projection entries. +/// +/// `qualified` selects the column-key convention of the rows the columns are +/// evaluated over: `true` for a join / lateral merged document, `false` for a +/// single-collection row. pub(in crate::control::planner::sql_plan_convert) fn extract_computed_columns( proj: &[Projection], window_functions: &[WindowSpec], + qualified: bool, ) -> crate::Result> { + let convert = bridge_expr_converter(qualified); let computed: Vec = proj .iter() .filter_map(|p| match p { @@ -137,7 +155,7 @@ pub(in crate::control::planner::sql_plan_convert) fn extract_computed_columns( { Some(crate::bridge::expr_eval::ComputedColumn { alias: alias.clone(), - expr: sql_expr_to_bridge_expr(expr), + expr: convert(expr), }) } _ => None, @@ -151,23 +169,27 @@ pub(in crate::control::planner::sql_plan_convert) fn extract_computed_columns( }) } +/// Serialize window specs. `qualified` follows the same convention as +/// [`extract_computed_columns`]. pub(in crate::control::planner::sql_plan_convert) fn serialize_window_functions( - specs: &[nodedb_sql::types::WindowSpec], + specs: &[WindowSpec], + qualified: bool, ) -> crate::Result> { if specs.is_empty() { return Ok(Vec::new()); } + let convert = bridge_expr_converter(qualified); let bridge_specs: Vec = specs .iter() .map(|s| crate::bridge::window_func::WindowFuncSpec { alias: s.alias.clone(), func_name: s.function.clone(), - args: s.args.iter().map(sql_expr_to_bridge_expr).collect(), - partition_by: s.partition_by.iter().map(sql_expr_to_bridge_expr).collect(), + args: s.args.iter().map(convert).collect(), + partition_by: s.partition_by.iter().map(convert).collect(), order_by: s .order_by .iter() - .map(|k| (sql_expr_to_bridge_expr(&k.expr), k.ascending)) + .map(|k| (convert(&k.expr), k.ascending)) .collect(), frame: s.frame.clone(), }) diff --git a/nodedb/src/control/planner/sql_plan_convert/expr/inline_cte.rs b/nodedb/src/control/planner/sql_plan_convert/expr/inline_cte.rs index ad7302897..bcd334df5 100644 --- a/nodedb/src/control/planner/sql_plan_convert/expr/inline_cte.rs +++ b/nodedb/src/control/planner/sql_plan_convert/expr/inline_cte.rs @@ -1,14 +1,23 @@ // SPDX-License-Identifier: BUSL-1.1 -use nodedb_sql::types::SqlPlan; +use nodedb_sql::types::{Filter, Projection, SortKey, SqlPlan, WindowSpec}; /// Replace scans on `cte_name` with the CTE's actual subquery plan. /// -/// Outer constraints on the CTE reference are merged onto the CTE body as far -/// as the body can carry them: a `Scan` body takes all of them; a -/// `VectorSearch` body takes filters, projection, and an unordered LIMIT (as -/// `top_k`). Constraints a body has no slot for — an outer `ORDER BY`, OFFSET, -/// or DISTINCT over a non-`Scan` body — are not applied. +/// Outer constraints on the CTE reference merge onto the body only where the +/// body can carry them without changing the rows it produces: +/// +/// - A plain filtered `Scan` body (no computed projection, window functions, +/// LIMIT, OFFSET, DISTINCT, or ORDER BY) takes every outer constraint. +/// - A `VectorSearch` body takes filters, a column-only projection, and an +/// unordered LIMIT (as `top_k`). +/// - A column-only outer projection over any other body resolves by output +/// schema, so the body is returned as is. +/// +/// Every other combination lowers into a `Subquery` post-processor over the +/// body's materialized rows. Computed projection entries and window functions +/// always take that path: the body does not evaluate them, and a name lookup +/// at the response boundary yields NULL. pub(in crate::control::planner::sql_plan_convert) fn inline_cte( plan: &SqlPlan, cte_name: &str, @@ -24,129 +33,21 @@ pub(in crate::control::planner::sql_plan_convert) fn inline_cte( limit, offset, distinct, + window_functions, .. - } if collection == cte_name => { - // If the outer query adds filters/sort/limit, wrap the CTE plan. - // For simple SELECT * FROM cte, just return the CTE plan directly. - if filters.is_empty() - && sort_keys.is_empty() - && limit.is_none() - && !distinct - && projection.is_empty() - { - cte_plan.clone() - } else { - // Merge outer constraints onto the CTE plan if it's also a Scan. - if let SqlPlan::Scan { - collection: inner_col, - alias: inner_alias, - engine: inner_eng, - filters: inner_f, - projection: inner_p, - sort_keys: inner_s, - limit: inner_l, - offset: inner_o, - distinct: inner_d, - window_functions: inner_w, - temporal: inner_t, - } = cte_plan - { - let mut merged_filters = inner_f.clone(); - merged_filters.extend(filters.iter().cloned()); - SqlPlan::Scan { - collection: inner_col.clone(), - alias: inner_alias.clone(), - engine: *inner_eng, - filters: merged_filters, - // Outer projection overrides inner; empty means "inherit from CTE". - projection: if projection.is_empty() { - inner_p.clone() - } else { - projection.clone() - }, - sort_keys: if sort_keys.is_empty() { - inner_s.clone() - } else { - sort_keys.clone() - }, - limit: limit.or(*inner_l), - // offset 0 = unspecified → inherit CTE's offset. - offset: if *offset > 0 { *offset } else { *inner_o }, - distinct: *distinct || *inner_d, - window_functions: inner_w.clone(), - temporal: *inner_t, - } - } else if let SqlPlan::VectorSearch { .. } = cte_plan { - // A k-NN body carries its own post-filter list and top-k. An - // outer `WHERE` merges into the engine post-filter so the cut - // counts MATCHING rows, and — when nothing reorders the - // result — an unordered `LIMIT` folds into `top_k` and the - // projection rides along. An outer `ORDER BY` / `OFFSET` / - // `DISTINCT` reorders the k rows, which the search leaf has no - // slot for; those (and a `LIMIT` that must apply after the - // reorder) run in a `Subquery` post-processor over the k rows. - let needs_reorder = !sort_keys.is_empty() || *offset > 0 || *distinct; - let mut leaf = cte_plan.clone(); - if let SqlPlan::VectorSearch { - filters: body_filters, - projection: body_projection, - top_k, - .. - } = &mut leaf - { - body_filters.extend(filters.iter().cloned()); - if !needs_reorder { - if !projection.is_empty() { - body_projection.clone_from(projection); - } - if let Some(outer_limit) = limit { - *top_k = (*top_k).min(*outer_limit); - } - } - } - if needs_reorder { - // Filters already run in the engine; the tail applies the - // reorder-dependent constraints over the k rows. It sorts - // before projecting, so ORDER BY may reference any column. - SqlPlan::Subquery { - input: Box::new(leaf), - filters: Vec::new(), - projection: projection.clone(), - sort_keys: sort_keys.clone(), - offset: *offset, - distinct: *distinct, - limit: *limit, - } - } else { - leaf - } - } else if filters.is_empty() - && sort_keys.is_empty() - && *offset == 0 - && !*distinct - && limit.is_none() - { - // Any other non-`Scan` body (Aggregate, Join, TextSearch, - // HybridSearch, SparseSearch, SpatialScan, MultiVectorSearch, - // ...) with only an outer projection: the response boundary - // projects by output schema, so no post-processor is needed. - cte_plan.clone() - } else { - // The body has no slot for these outer constraints. Apply - // them over its materialized rows in a `Subquery` - // post-processor — previously they were silently dropped. - SqlPlan::Subquery { - input: Box::new(cte_plan.clone()), - filters: filters.clone(), - projection: projection.clone(), - sort_keys: sort_keys.clone(), - offset: *offset, - distinct: *distinct, - limit: *limit, - } - } - } - } + } if collection == cte_name => inline_cte_scan_ref( + ScanRef { + filters, + projection, + sort_keys, + limit: *limit, + offset: *offset, + distinct: *distinct, + window_functions, + has_computed: has_computed_projection(projection), + }, + cte_plan, + ), // Aggregate referencing CTE → inline into the input. SqlPlan::Aggregate { @@ -234,6 +135,7 @@ pub(in crate::control::planner::sql_plan_convert) fn inline_cte( input, filters, projection, + window_functions, sort_keys, offset, distinct, @@ -242,6 +144,7 @@ pub(in crate::control::planner::sql_plan_convert) fn inline_cte( input: Box::new(inline_cte(input, cte_name, cte_plan)), filters: filters.clone(), projection: projection.clone(), + window_functions: window_functions.clone(), sort_keys: sort_keys.clone(), offset: *offset, distinct: *distinct, @@ -253,6 +156,191 @@ pub(in crate::control::planner::sql_plan_convert) fn inline_cte( } } +/// `true` if any projection entry is a computed expression (`price * qty AS +/// total`) rather than a bare column or star. +fn has_computed_projection(projection: &[Projection]) -> bool { + projection + .iter() + .any(|p| matches!(p, Projection::Computed { .. })) +} + +/// The outer constraints carried on a `Scan` that references the CTE by +/// name. `has_computed` is precomputed once so the sub-cases below don't +/// each re-walk `projection`. +#[derive(Clone, Copy)] +struct ScanRef<'a> { + filters: &'a Vec, + projection: &'a Vec, + sort_keys: &'a Vec, + limit: Option, + offset: usize, + distinct: bool, + window_functions: &'a Vec, + has_computed: bool, +} + +impl ScanRef<'_> { + /// No filter, sort, limit, offset, distinct, window, or computed entry — + /// the outer reference adds nothing the body's output schema doesn't + /// already answer. + fn is_unconstrained(&self) -> bool { + self.filters.is_empty() + && self.sort_keys.is_empty() + && self.limit.is_none() + && !self.distinct + && self.offset == 0 + && self.window_functions.is_empty() + && !self.has_computed + } +} + +/// Wrap `input` in a `Subquery` post-processor carrying the outer +/// constraints that `input` has no slot for. +fn wrap_in_subquery(input: SqlPlan, filters: Vec, outer: ScanRef<'_>) -> SqlPlan { + SqlPlan::Subquery { + input: Box::new(input), + filters, + projection: outer.projection.clone(), + window_functions: outer.window_functions.clone(), + sort_keys: outer.sort_keys.clone(), + offset: outer.offset, + distinct: outer.distinct, + limit: outer.limit, + } +} + +/// Resolve a CTE reference for a `Scan { collection: cte_name, .. }` node, +/// merging the outer constraints onto `cte_plan` as far as its body kind +/// can carry them. +fn inline_cte_scan_ref(outer: ScanRef<'_>, cte_plan: &SqlPlan) -> SqlPlan { + // A column-only projection resolves by the body's output schema. + if outer.is_unconstrained() { + return cte_plan.clone(); + } + + if let Some(merged) = merge_into_scan_body(outer, cte_plan) { + return merged; + } + + if let Some(merged) = merge_into_vector_search_body(outer, cte_plan) { + return merged; + } + + // Any other body (Aggregate, Join, TextSearch, HybridSearch, + // SparseSearch, SpatialScan, MultiVectorSearch, a constrained Scan, + // ...) has no slot for the outer constraints reaching this point — the + // unconstrained case already returned at the top of the function. Apply + // them over the body's materialized rows in a `Subquery` + // post-processor, which evaluates computed columns and window + // functions itself. + wrap_in_subquery(cte_plan.clone(), outer.filters.clone(), outer) +} + +/// A plain filtered `Scan` body takes every outer constraint. A body that +/// limits, offsets, dedups, orders, computes, or windows changes which rows +/// the outer constraints see if they merge into it: an outer WHERE inside an +/// inner LIMIT changes the cut, and an inner `qty AS x` alias replaced by the +/// outer projection makes `x` NULL. +fn merge_into_scan_body(outer: ScanRef<'_>, cte_plan: &SqlPlan) -> Option { + let SqlPlan::Scan { + collection: inner_col, + alias: inner_alias, + engine: inner_eng, + filters: inner_f, + projection: inner_p, + sort_keys: inner_s, + limit: inner_l, + offset: inner_o, + distinct: inner_d, + window_functions: inner_w, + temporal: inner_t, + } = cte_plan + else { + return None; + }; + if has_computed_projection(inner_p) + || !inner_w.is_empty() + || inner_l.is_some() + || *inner_o != 0 + || *inner_d + || !inner_s.is_empty() + { + return None; + } + + let mut merged_filters = inner_f.clone(); + merged_filters.extend(outer.filters.iter().cloned()); + Some(SqlPlan::Scan { + collection: inner_col.clone(), + alias: inner_alias.clone(), + engine: *inner_eng, + filters: merged_filters, + // A named outer projection overrides the inner one. An empty or + // star-only outer projection inherits the CTE's own column list, so + // `SELECT * FROM (SELECT a FROM t)` emits `a` alone. + projection: if outer + .projection + .iter() + .all(|p| matches!(p, Projection::Star | Projection::QualifiedStar(_))) + { + inner_p.clone() + } else { + outer.projection.clone() + }, + sort_keys: outer.sort_keys.clone(), + limit: outer.limit, + offset: outer.offset, + distinct: outer.distinct, + window_functions: outer.window_functions.clone(), + temporal: *inner_t, + }) +} + +/// A k-NN body carries its own post-filter list and top-k. An outer `WHERE` +/// merges into the engine post-filter so the cut counts MATCHING rows. When +/// nothing reorders the result and the outer projection is column-only, an +/// unordered `LIMIT` folds into `top_k` and the projection rides along. An +/// outer `ORDER BY` / `OFFSET` / `DISTINCT` reorders the k rows, and a +/// computed projection or window function evaluates over them; the search +/// leaf has no slot for any of those, so they (and a `LIMIT` that must apply +/// after the reorder) run in a `Subquery` post-processor over the k rows. +fn merge_into_vector_search_body(outer: ScanRef<'_>, cte_plan: &SqlPlan) -> Option { + if !matches!(cte_plan, SqlPlan::VectorSearch { .. }) { + return None; + } + + let needs_reorder = !outer.sort_keys.is_empty() + || outer.offset > 0 + || outer.distinct + || outer.has_computed + || !outer.window_functions.is_empty(); + let mut leaf = cte_plan.clone(); + if let SqlPlan::VectorSearch { + filters: body_filters, + projection: body_projection, + top_k, + .. + } = &mut leaf + { + body_filters.extend(outer.filters.iter().cloned()); + if !needs_reorder { + if !outer.projection.is_empty() { + body_projection.clone_from(outer.projection); + } + if let Some(outer_limit) = outer.limit { + *top_k = (*top_k).min(outer_limit); + } + } + } + if !needs_reorder { + return Some(leaf); + } + // Filters already run in the engine; the tail applies the + // reorder-dependent constraints over the k rows. It sorts before + // projecting, so ORDER BY may reference any column. + Some(wrap_in_subquery(leaf, Vec::new(), outer)) +} + #[cfg(test)] mod tests { use super::*; @@ -439,4 +527,232 @@ mod tests { "an unordered LIMIT must fold into top_k, not wrap: {plan:?}" ); } + + fn doubled_x() -> Projection { + Projection::Computed { + expr: nodedb_sql::types::SqlExpr::BinaryOp { + left: Box::new(nodedb_sql::types::SqlExpr::Column { + table: None, + name: "x".to_string(), + }), + op: nodedb_sql::types::BinaryOp::Mul, + right: Box::new(nodedb_sql::types::SqlExpr::Literal(SqlValue::Int(2))), + }, + alias: "y".to_string(), + } + } + + fn row_number_spec() -> nodedb_sql::types::WindowSpec { + nodedb_sql::types::WindowSpec { + function: "row_number".to_string(), + args: Vec::new(), + partition_by: Vec::new(), + order_by: Vec::new(), + alias: "rn".to_string(), + frame: Default::default(), + } + } + + /// A CTE-referencing scan carrying an outer projection and window list. + fn scan_on_cte_projected( + projection: Vec, + window_functions: Vec, + ) -> SqlPlan { + SqlPlan::Scan { + collection: "knn".to_string(), + alias: None, + engine: EngineType::DocumentSchemaless, + filters: Vec::new(), + projection, + sort_keys: Vec::new(), + limit: None, + offset: 0, + distinct: false, + window_functions, + temporal: nodedb_sql::TemporalScope::default(), + } + } + + /// A `SELECT 1 AS x` body: a constant result, not a scan. + fn constant_body() -> SqlPlan { + SqlPlan::ConstantResult { + columns: vec!["x".to_string()], + values: vec![SqlValue::Int(1)], + volatile: false, + } + } + + #[test] + fn computed_projection_over_vector_search_wraps_in_subquery() { + match inline_cte( + &scan_on_cte_projected(vec![doubled_x()], Vec::new()), + "knn", + &vector_search_body(), + ) { + SqlPlan::Subquery { + input, projection, .. + } => { + assert_eq!(projection.len(), 1, "the computed entry rides the wrapper"); + assert!( + matches!(&*input, SqlPlan::VectorSearch { projection, .. } if projection.is_empty()), + "a computed projection never folds into the search leaf" + ); + } + other => panic!("expected Subquery, got {other:?}"), + } + } + + #[test] + fn column_only_projection_over_non_scan_body_returns_body() { + let plan = inline_cte( + &scan_on_cte_projected(vec![Projection::Column("x".to_string())], Vec::new()), + "knn", + &constant_body(), + ); + assert!( + matches!(plan, SqlPlan::ConstantResult { .. }), + "a column-only projection resolves by output schema: {plan:?}" + ); + } + + #[test] + fn computed_projection_over_constant_body_wraps_in_subquery() { + match inline_cte( + &scan_on_cte_projected(vec![doubled_x()], Vec::new()), + "knn", + &constant_body(), + ) { + SqlPlan::Subquery { + input, + projection, + window_functions, + .. + } => { + assert!(matches!(*input, SqlPlan::ConstantResult { .. })); + assert_eq!(projection.len(), 1); + assert!(window_functions.is_empty()); + } + other => panic!("expected Subquery, got {other:?}"), + } + } + + #[test] + fn window_function_over_constant_body_rides_the_subquery() { + match inline_cte( + &scan_on_cte_projected(Vec::new(), vec![row_number_spec()]), + "knn", + &constant_body(), + ) { + SqlPlan::Subquery { + window_functions, .. + } => assert_eq!(window_functions[0].alias, "rn"), + other => panic!("expected Subquery, got {other:?}"), + } + } + + fn limited_scan_body() -> SqlPlan { + SqlPlan::Scan { + collection: "orders".to_string(), + alias: None, + engine: EngineType::DocumentSchemaless, + filters: Vec::new(), + projection: Vec::new(), + sort_keys: Vec::new(), + limit: Some(5), + offset: 0, + distinct: false, + window_functions: Vec::new(), + temporal: nodedb_sql::TemporalScope::default(), + } + } + + #[test] + fn outer_filter_over_limited_scan_body_wraps_instead_of_merging() { + // Merging the WHERE under the inner LIMIT changes which rows the + // limit sees; the filter must run over the limited rows instead. + match inline_cte( + &scan_on_cte(vec![tag_filter()], None), + "knn", + &limited_scan_body(), + ) { + SqlPlan::Subquery { input, filters, .. } => { + assert_eq!(filters.len(), 1); + assert!(matches!(*input, SqlPlan::Scan { limit: Some(5), .. })); + } + other => panic!("expected Subquery, got {other:?}"), + } + } + + #[test] + fn outer_filter_over_plain_scan_body_merges() { + let body = SqlPlan::Scan { + collection: "orders".to_string(), + alias: None, + engine: EngineType::DocumentSchemaless, + filters: Vec::new(), + projection: vec![Projection::Column("qty".to_string())], + sort_keys: Vec::new(), + limit: None, + offset: 0, + distinct: false, + window_functions: Vec::new(), + temporal: nodedb_sql::TemporalScope::default(), + }; + match inline_cte(&scan_on_cte(vec![tag_filter()], Some(2)), "knn", &body) { + SqlPlan::Scan { + collection, + filters, + projection, + limit, + .. + } => { + assert_eq!(collection, "orders"); + assert_eq!(filters.len(), 1); + assert_eq!(limit, Some(2)); + assert_eq!(projection.len(), 1, "the inner column list is inherited"); + } + other => panic!("expected merged Scan, got {other:?}"), + } + } + + #[test] + fn aliased_inner_scan_body_keeps_alias_under_outer_computed() { + // `SELECT x * 2 AS y FROM (SELECT qty AS x FROM orders) s`: the inner + // alias must be evaluated by the body before the outer expression reads it. + let body = SqlPlan::Scan { + collection: "orders".to_string(), + alias: None, + engine: EngineType::DocumentSchemaless, + filters: Vec::new(), + projection: vec![Projection::Computed { + expr: nodedb_sql::types::SqlExpr::Column { + table: None, + name: "qty".to_string(), + }, + alias: "x".to_string(), + }], + sort_keys: Vec::new(), + limit: None, + offset: 0, + distinct: false, + window_functions: Vec::new(), + temporal: nodedb_sql::TemporalScope::default(), + }; + match inline_cte( + &scan_on_cte_projected(vec![doubled_x()], Vec::new()), + "knn", + &body, + ) { + SqlPlan::Subquery { + input, projection, .. + } => { + assert!(matches!( + &*input, + SqlPlan::Scan { projection, .. } if projection.len() == 1 + )); + assert_eq!(projection.len(), 1); + } + other => panic!("expected Subquery, got {other:?}"), + } + } } diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/core.rs b/nodedb/src/control/planner/sql_plan_convert/scan/core.rs index 5f6ff318a..4367a9719 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/core.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/core.rs @@ -43,8 +43,8 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_scan( // `rows` is left empty here; the coordinator fills it post-cache via // `materialize_providers`. Using an empty-coordinator vshard (empty // collection string) keeps the task coordinator-local. - let computed_bytes = extract_computed_columns(projection, window_functions)?; - let window_bytes = serialize_window_functions(window_functions)?; + let computed_bytes = extract_computed_columns(projection, window_functions, false)?; + let window_bytes = serialize_window_functions(window_functions, false)?; if crate::control::server::pgwire::catalog::schema::catalog_collection_info(collection) .is_some() diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs b/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs index 8333cd819..0d8521248 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs @@ -64,7 +64,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_timeseries_scan( } let proj_names = extract_projection_names(projection, &[]); - let computed_bytes = extract_computed_columns(projection, &[])?; + let computed_bytes = extract_computed_columns(projection, &[], false)?; let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); Ok(vec![PhysicalTask { tenant_id, diff --git a/nodedb/src/control/planner/sql_plan_convert/set_ops.rs b/nodedb/src/control/planner/sql_plan_convert/set_ops.rs index 4a9a81e7f..383c9b78f 100644 --- a/nodedb/src/control/planner/sql_plan_convert/set_ops.rs +++ b/nodedb/src/control/planner/sql_plan_convert/set_ops.rs @@ -2,7 +2,7 @@ //! Set operations and miscellaneous plan conversions (UNION, INTERSECT, EXCEPT, CTE, etc.). -use nodedb_sql::types::{Projection, SortKey, SqlExpr, SqlPlan, SqlValue}; +use nodedb_sql::types::{Projection, SortKey, SqlExpr, SqlPlan, SqlValue, WindowSpec}; use crate::bridge::envelope::PhysicalPlan; use crate::types::{TenantId, VShardId}; @@ -255,6 +255,7 @@ pub(super) fn convert_subquery( input, filters, projection, + window_functions, sort_keys, offset, distinct, @@ -278,10 +279,11 @@ pub(super) fn convert_subquery( // A join / lateral body emits ONE merged document per output row whose // columns keep their table prefix (`a.attnum`), which is why the response - // shaper looks those rows up by the qualified name. The tail's sort keys - // must address the same shape — an unqualified key resolves to NULL on - // every merged row, and a sort where every key is NULL is a no-op that - // silently answers an ordered query in the body's own order. + // shaper looks those rows up by the qualified name. The tail's sort keys, + // computed columns, and window specs must address the same shape — an + // unqualified key resolves to NULL on every merged row, and a sort where + // every key is NULL is a no-op that silently answers an ordered query in + // the body's own order. let merged_doc_body = matches!( child, PhysicalPlan::Query( @@ -318,9 +320,16 @@ pub(super) fn convert_subquery( plan: PhysicalPlan::Query(QueryOp::PostProcess { input: Box::new(child), filters: super::filter::serialize_filters(filters)?, - projection: lower_subquery_projection(projection)?, - computed_columns: Vec::new(), - window_functions: Vec::new(), + projection: lower_subquery_projection(projection, window_functions)?, + computed_columns: super::aggregate::extract_computed_columns( + projection, + window_functions, + merged_doc_body, + )?, + window_functions: super::aggregate::serialize_window_functions( + window_functions, + merged_doc_body, + )?, sort_keys: lower_subquery_sort_keys(sort_keys, merged_doc_body), limit, offset, @@ -336,14 +345,16 @@ pub(super) fn convert_subquery( /// A bare column keeps its unqualified name (the flattened row's column key); a /// star selects every column, so no column pruning is applied (empty = all). /// -/// A computed item is projected under its alias: the body evaluates the -/// expression and emits the value under that name before the tail runs, which -/// is the same key the response shaper reads it back by. Erroring here instead -/// would reject `SELECT a, f(b) … ORDER BY c` outright, since the wrapper -/// carries the original SELECT list whenever the body had to be widened to keep -/// the sort column. -fn lower_subquery_projection(projection: &[Projection]) -> crate::Result> { - let mut names = Vec::with_capacity(projection.len()); +/// A computed item is projected under its alias: the tail evaluates the +/// expression over the materialized rows and emits the value under that name, +/// which is the same key the response shaper reads it back by. Every window +/// alias is kept too, so a window output the SELECT list does not repeat as a +/// computed entry survives the column pruning. +fn lower_subquery_projection( + projection: &[Projection], + window_functions: &[WindowSpec], +) -> crate::Result> { + let mut names = Vec::with_capacity(projection.len() + window_functions.len()); for p in projection { match p { Projection::Column(qname) => { @@ -353,6 +364,11 @@ fn lower_subquery_projection(projection: &[Projection]) -> crate::Result names.push(alias.clone()), } } + for spec in window_functions { + if !names.contains(&spec.alias) { + names.push(spec.alias.clone()); + } + } Ok(names) } diff --git a/nodedb/src/data/executor/handlers/provider_scan.rs b/nodedb/src/data/executor/handlers/provider_scan.rs index 51af25c83..799f3acf1 100644 --- a/nodedb/src/data/executor/handlers/provider_scan.rs +++ b/nodedb/src/data/executor/handlers/provider_scan.rs @@ -3,11 +3,12 @@ //! Executor handler for `QueryOp::ProviderScan`. //! //! Decodes the pre-materialized msgpack row array, applies predicate -//! filtering, sort, window functions + computed columns, distinct +//! filtering, window functions + computed columns, sort, distinct //! deduplication, offset, column projection, and limit — in that order — -//! then emits the resulting rows via `response_with_payload`. Sort runs -//! before offset so `ORDER BY ... OFFSET n` skips the first `n` rows of the -//! sorted set, not the decoded set. +//! then emits the resulting rows via `response_with_payload`. Windows and +//! computed columns run before sort so `ORDER BY` can name their aliases. +//! Sort runs before offset so `ORDER BY ... OFFSET n` skips the first `n` +//! rows of the sorted set, not the decoded set. use nodedb_query::msgpack_scan; @@ -36,8 +37,8 @@ pub(in crate::data::executor) struct ProviderScanParams<'a> { impl CoreLoop { /// Execute a `ProviderScan` plan node. /// - /// Processing order: decode rows → filter → sort → windows + computed - /// columns → distinct → offset → project → limit → emit. + /// Processing order: decode rows → filter → windows + computed columns → + /// sort → distinct → offset → project → limit → emit. pub(in crate::data::executor) fn execute_provider_scan( &mut self, task: &ExecutionTask, @@ -100,18 +101,11 @@ impl CoreLoop { } } - // ── 3. Sort. ────────────────────────────────────────────────────────── - // Runs before offset: `ORDER BY ... OFFSET n` skips the first `n` rows - // of the SORTED set, not the decoded set. - if !sort_keys.is_empty() - && let Err(e) = sort_msgpack_rows(&mut rows, sort_keys) - { - return self.response_error(task, crate::Error::from(e)); - } - - // ── 4. Window functions + computed columns. ───────────────────────── - // Skipped entirely (zero-decode msgpack path) when both byte slices - // are empty. + // ── 3. Window functions + computed columns. ───────────────────────── + // Runs before sort so `ORDER BY` can name a window or computed alias. + // Each window spec orders its own partitions, so the row order here + // does not affect window results. Skipped entirely (zero-decode + // msgpack path) when both byte slices are empty. if !window_functions_bytes.is_empty() || !computed_columns_bytes.is_empty() { rows = match apply_windows_and_computed( rows, @@ -123,6 +117,15 @@ impl CoreLoop { }; } + // ── 4. Sort. ────────────────────────────────────────────────────────── + // Runs before offset: `ORDER BY ... OFFSET n` skips the first `n` rows + // of the SORTED set, not the decoded set. + if !sort_keys.is_empty() + && let Err(e) = sort_msgpack_rows(&mut rows, sort_keys) + { + return self.response_error(task, crate::Error::from(e)); + } + // ── 5. Distinct (on the would-be projected row). ────────────────────── // Deduplicate on the projected shape so SQL DISTINCT semantics are // honoured: two rows with the same projected columns but different diff --git a/nodedb/src/data/executor/handlers/provider_scan_compute.rs b/nodedb/src/data/executor/handlers/provider_scan_compute.rs index 64284f960..5dc27c063 100644 --- a/nodedb/src/data/executor/handlers/provider_scan_compute.rs +++ b/nodedb/src/data/executor/handlers/provider_scan_compute.rs @@ -2,7 +2,7 @@ //! Window-function and computed-column evaluation for `QueryOp::ProviderScan`. //! -//! Runs after sort and before distinct/offset/project/limit in the +//! Runs after filter and before sort/distinct/offset/project/limit in the //! `ProviderScan` pipeline. Each msgpack row decodes to a `serde_json::Value`, //! windows evaluate over the full row set, computed columns evaluate //! per-row, and the result re-encodes to msgpack. Skipped entirely when both From 091658084f4a5da8274fc3d982edf1fc5b2f3de7 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 18 Sep 2026 00:03:25 +0800 Subject: [PATCH 06/15] feat(query): aggregate over derived-table and union bodies Extend input-sourced aggregation, previously restricted to catalog ProviderScan inputs, to any body with no routing collection: a derived-table subquery, a constant result, or a UNION ALL. The planner now routes such bodies through a shared input_sourced lowering that materializes the body and builds a coordinator-local Aggregate task, factored out of the catalog and single-collection paths via build_input_sourced_aggregate_task. extract_collection_name and join_side_collection return Option instead of an empty-string sentinel to distinguish a scan-shaped input from one with no routing collection at all. The exchange resolver gains an Aggregate{input: Some} arm that materializes the child (reusing the PostProcess gather path, generalized as materialize_child_rows) and runs it once on the coordinator's owning core when the child itself reads no per-shard collection. The executor's aggregate handler now fails the statement on a child error instead of masking it as zero rows, and treats an undecodable non-empty payload as an internal error rather than an empty result. --- nodedb-physical/src/physical_plan/query.rs | 16 +-- .../aggregate/input_sourced.rs | 115 ++++++++++++++++++ .../planner/sql_plan_convert/aggregate/mod.rs | 4 +- .../sql_plan_convert/aggregate/plan.rs | 75 +++++++----- .../sql_plan_convert/aggregate/spec.rs | 93 ++++++++++++-- .../planner/sql_plan_convert/scan/join.rs | 8 +- .../resolve/exchange/aggregate_input_arm.rs | 105 ++++++++++++++++ .../exchange/resolve/exchange/dispatch.rs | 43 ++++++- .../server/exchange/resolve/exchange/mod.rs | 4 + .../resolve/exchange/post_process_arm.rs | 104 ++++++++++++---- .../data/executor/handlers/aggregate/exec.rs | 49 +++++--- nodedb/tests/wire/cases/sql_subquery_from.rs | 15 ++- 12 files changed, 529 insertions(+), 102 deletions(-) create mode 100644 nodedb/src/control/planner/sql_plan_convert/aggregate/input_sourced.rs create mode 100644 nodedb/src/control/server/exchange/resolve/exchange/aggregate_input_arm.rs diff --git a/nodedb-physical/src/physical_plan/query.rs b/nodedb-physical/src/physical_plan/query.rs index 001c5d459..110bf35b5 100644 --- a/nodedb-physical/src/physical_plan/query.rs +++ b/nodedb-physical/src/physical_plan/query.rs @@ -153,14 +153,14 @@ pub enum QueryOp { Aggregate { collection: QualifiedCollection, /// Optional sub-plan whose decoded rows are aggregated instead of - /// scanning `collection` per-shard. `Some` currently means EXACTLY a - /// catalog source (a `ProviderScan` lowered by the converter): the - /// aggregate runs over the coordinator-materialized catalog rows and is - /// therefore coordinator-local (never broadcast — see - /// `is_sharded_source`). `None` = legacy path: scan the named - /// `collection` on every shard. `collection` stays populated in both - /// cases so downstream RLS / permission / classification continue to - /// read it; the executor simply prefers `input` when present. + /// scanning `collection` per-shard. `Some` = an input-sourced + /// aggregate over a materialized relation: a catalog `ProviderScan`, + /// or any derived-table body the coordinator materializes into a + /// `ProviderScan` before dispatch. Coordinator-local, never broadcast + /// (see `is_sharded_source`). `None` = scan the named `collection` on + /// every shard. `collection` stays populated in both cases so + /// downstream RLS / permission / classification continue to read it; + /// the executor prefers `input` when present. #[serde(default)] input: Option>, group_by: Vec, diff --git a/nodedb/src/control/planner/sql_plan_convert/aggregate/input_sourced.rs b/nodedb/src/control/planner/sql_plan_convert/aggregate/input_sourced.rs new file mode 100644 index 000000000..e4e9f289f --- /dev/null +++ b/nodedb/src/control/planner/sql_plan_convert/aggregate/input_sourced.rs @@ -0,0 +1,115 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Aggregate over an input-sourced body: a derived table, constant result, or +//! union that carries no routing collection. The body lowers to one physical +//! plan, a sharded body is gathered, and one coordinator-local `Aggregate` +//! task runs over the materialized rows. + +use nodedb_sql::types::{AggregateExpr, Filter, SqlExpr, SqlPlan}; + +use crate::bridge::envelope::PhysicalPlan; +use crate::types::TenantId; +use nodedb_physical::physical_plan::*; +use nodedb_physical::physical_task::PhysicalTask; + +use super::super::convert::{ConvertContext, convert_one}; +use super::super::filter::serialize_filters; +use super::spec::{InputSourcedTaskParams, build_input_sourced_aggregate_task}; + +pub(super) struct InputSourcedAggregateParams<'a> { + pub input: &'a SqlPlan, + pub group_by: &'a [SqlExpr], + pub aggregates: &'a [AggregateExpr], + pub having: &'a [Filter], + pub limit: usize, + pub grouping_sets: Option<&'a [Vec]>, + /// Post-aggregate sort keys, already lowered to bridge specs. + pub sort_keys: Vec, + pub tenant_id: TenantId, + pub ctx: &'a ConvertContext, +} + +/// Lower an aggregate whose input is a materialized relation. +/// +/// The body converts through `convert_one` and must produce exactly one task. +/// A sharded body is wrapped in `Exchange{Gather}` so the coordinator resolves +/// it to a `ProviderScan` before the aggregate runs. The emitted task is +/// coordinator-local: an empty collection keeps it on the coordinator vshard +/// and `is_sharded_source` reports the `Some(input)` aggregate as +/// non-sharded, so it runs once and is never broadcast. +pub(super) fn convert_input_sourced_aggregate( + p: InputSourcedAggregateParams<'_>, +) -> crate::Result> { + let InputSourcedAggregateParams { + input, + group_by, + aggregates, + having, + limit, + grouping_sets, + sort_keys, + tenant_id, + ctx, + } = p; + + // The input-sourced aggregate executor does not expand ROLLUP / CUBE / + // GROUPING SETS. A typed error beats a silent base-grouping-only answer. + if grouping_sets.is_some_and(|sets| !sets.is_empty()) { + return Err(crate::Error::PlanError { + detail: "ROLLUP / CUBE / GROUPING SETS over a derived-table body is not supported" + .to_string(), + }); + } + + // The body is one relation. A body that lowers to several tasks (a set + // operation) has no single row stream to aggregate. + let mut body = convert_one(input, tenant_id, ctx)?; + if body.len() != 1 { + return Err(crate::Error::PlanError { + detail: format!( + "aggregate over a derived-table body that lowers to {} physical tasks is not \ + supported; the body must produce a single relation", + body.len() + ), + }); + } + let mut child = match body.pop() { + Some(task) => task.plan, + None => { + return Err(crate::Error::PlanError { + detail: "aggregate over a derived-table body produced no physical task".to_string(), + }); + } + }; + + // A sharded body is gathered first so the aggregate observes the FULL + // union exactly once. The aggregate task is coordinator-local, so the + // top-level `convert()` wrap loop does not gather the child. + if child.is_sharded_source() { + let as_aggregate = matches!( + &child, + PhysicalPlan::Query(QueryOp::Aggregate { .. }) + | PhysicalPlan::Query(QueryOp::PartialAggregate { .. }) + ); + child = PhysicalPlan::Query(QueryOp::Exchange(ExchangeOp { + child: Box::new(child), + mode: ExchangeMode::Gather { as_aggregate }, + })); + } + + let having_bytes = serialize_filters(having)?; + + Ok(vec![build_input_sourced_aggregate_task( + InputSourcedTaskParams { + tenant_id, + ctx, + raw_collection: String::new(), + child, + group_by, + aggregates, + having_bytes, + limit, + sort_keys, + }, + )]) +} diff --git a/nodedb/src/control/planner/sql_plan_convert/aggregate/mod.rs b/nodedb/src/control/planner/sql_plan_convert/aggregate/mod.rs index ac20801bd..bdc156c37 100644 --- a/nodedb/src/control/planner/sql_plan_convert/aggregate/mod.rs +++ b/nodedb/src/control/planner/sql_plan_convert/aggregate/mod.rs @@ -4,11 +4,13 @@ //! //! Split by concern so each file stays under the project's hard size limit: //! `plan` (the `convert_aggregate` entry point and its join / catalog / -//! timeseries lowering), `spec` (aggregate-spec + collection/alias helpers and +//! timeseries lowering), `input_sourced` (aggregate over a materialized +//! derived-table body), `spec` (aggregate-spec + collection/alias helpers and //! join-side embedding), and `projection` (projection / computed-column / //! window-function serialization). mod cost; +mod input_sourced; mod plan; mod projection; mod spec; diff --git a/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs b/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs index e9bb60684..ae9ba6e9b 100644 --- a/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs +++ b/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs @@ -1,7 +1,14 @@ // SPDX-License-Identifier: BUSL-1.1 -//! The `convert_aggregate` entry point: join-sourced, catalog (input-sourced), -//! timeseries, and standard single-collection aggregate lowering. +//! The `convert_aggregate` entry point. Four lowering shapes: +//! +//! - join: `HashJoin` with post-join grouping +//! - input-sourced body (derived table, constant result, union): the body is +//! materialized and the aggregate runs over its rows on the coordinator +//! (`input_sourced.rs`) +//! - catalog: `ProviderScan` input, coordinator-local +//! - single collection: per-shard `Aggregate` (timeseries routes through +//! `TimeseriesOp::Scan`) use nodedb_sql::types::{EngineType, Filter, SortKey, SqlExpr, SqlPlan}; @@ -14,8 +21,9 @@ use super::super::convert::{ConvertContext, db_qualified}; use super::super::expr::convert_sort_keys; use super::super::filter::serialize_filters; use super::spec::{ - agg_expr_to_pair, agg_expr_to_spec, extract_collection_name, extract_scan_alias, - group_by_to_specs, group_by_to_strings, inline_join_side, join_side_collection, + InputSourcedTaskParams, agg_expr_to_pair, agg_expr_to_spec, build_input_sourced_aggregate_task, + extract_collection_name, extract_scan_alias, group_by_to_specs, group_by_to_strings, + inline_join_side, join_side_collection, }; use nodedb_sql::types::AggregateExpr; @@ -123,8 +131,23 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_aggregate( }]); } - // Standard aggregate on a single collection. - let raw_collection = extract_collection_name(input); + // A body with no routing collection (derived table, constant result, + // union) is materialized and aggregated on the coordinator. + let Some(raw_collection) = extract_collection_name(input) else { + return super::input_sourced::convert_input_sourced_aggregate( + super::input_sourced::InputSourcedAggregateParams { + input, + group_by, + aggregates, + having, + limit, + grouping_sets, + sort_keys: bridge_sort_keys, + tenant_id, + ctx, + }, + ); + }; let (filters_ref, engine) = match input { SqlPlan::Scan { filters, engine, .. @@ -157,14 +180,12 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_aggregate( ), }); } - let group_specs = group_by_to_specs(group_by); - let agg_specs: Vec = aggregates.iter().map(agg_expr_to_spec).collect(); let provider_scan = PhysicalPlan::Query(QueryOp::ProviderScan { provider: Some(raw_collection.clone()), rows: Vec::new(), // WHERE predicates on the catalog are applied by the ProviderScan // before the rows reach the aggregate. - filters: filter_bytes.clone(), + filters: filter_bytes, projection: Vec::new(), computed_columns: Vec::new(), window_functions: Vec::new(), @@ -173,31 +194,19 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_aggregate( offset: 0, distinct: false, }); - return Ok(vec![PhysicalTask { - tenant_id, - // Coordinator-local: empty collection keeps the task on the - // coordinator vshard (catalog rows are not per-shard). - vshard_id: VShardId::from_collection_in_database(ctx.database_id, ""), - database_id: ctx.database_id, - plan: PhysicalPlan::Query(QueryOp::Aggregate { - collection: nodedb_types::QualifiedCollection::from_stored(raw_collection), - input: Some(Box::new(provider_scan)), - group_by: group_specs, - aggregates: agg_specs, - // Filters live on the ProviderScan input; the aggregate node - // applies none of its own over the already-filtered rows. - filters: Vec::new(), - having: having_bytes, + return Ok(vec![build_input_sourced_aggregate_task( + InputSourcedTaskParams { + tenant_id, + ctx, + raw_collection, + child: provider_scan, + group_by, + aggregates, + having_bytes, limit, - sub_group_by: Vec::new(), - sub_aggregates: Vec::new(), - // Guarded above: catalog aggregates never carry grouping sets. - grouping_sets: Vec::new(), sort_keys: bridge_sort_keys, - }), - post_set_op: PostSetOp::None, - txn_id: None, - }]); + }, + )]); } let collection = db_qualified(ctx.database_id, &raw_collection); @@ -219,7 +228,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_aggregate( collection: qualified_collection, // Derived in the Data Plane against the declared TIME_KEY. time_range: UNBOUNDED_TIME_RANGE, - sort_keys: bridge_sort_keys.clone(), + sort_keys: bridge_sort_keys, projection: Vec::new(), limit, filters: filter_bytes, diff --git a/nodedb/src/control/planner/sql_plan_convert/aggregate/spec.rs b/nodedb/src/control/planner/sql_plan_convert/aggregate/spec.rs index e51e3a898..65f4497f4 100644 --- a/nodedb/src/control/planner/sql_plan_convert/aggregate/spec.rs +++ b/nodedb/src/control/planner/sql_plan_convert/aggregate/spec.rs @@ -6,8 +6,9 @@ use nodedb_sql::types::{AggregateExpr, SqlExpr, SqlPlan}; use crate::bridge::envelope::PhysicalPlan; -use crate::types::TenantId; +use crate::types::{TenantId, VShardId}; use nodedb_physical::physical_plan::*; +use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; use super::super::convert::{ConvertContext, convert_one, db_qualified}; use super::super::expr::sql_expr_to_bridge_expr; @@ -86,26 +87,31 @@ pub(in crate::control::planner::sql_plan_convert) fn join_side_collection( plan: &SqlPlan, database_id: crate::types::DatabaseId, ) -> String { - let raw = extract_collection_name(plan); - if scan_is_catalog(&raw) { - String::new() - } else { - db_qualified(database_id, &raw) + // An input-sourced join side carries no routing collection; its rows come + // from `left_input` / `right_input`. + match extract_collection_name(plan) { + Some(raw) if !scan_is_catalog(&raw) => db_qualified(database_id, &raw), + Some(_) | None => String::new(), } } +/// Routing collection for a scan-shaped input. `None` for an input-sourced +/// (materialized) body. +/// +/// A `Join` yields its left side's collection as the routing hint for a join +/// over scans. An `Aggregate` input is a materialized relation, not a scan. pub(in crate::control::planner::sql_plan_convert) fn extract_collection_name( plan: &SqlPlan, -) -> String { +) -> Option { match plan { - SqlPlan::Scan { collection, .. } => collection.clone(), - SqlPlan::PointGet { collection, .. } => collection.clone(), + SqlPlan::Scan { collection, .. } => Some(collection.clone()), + SqlPlan::PointGet { collection, .. } => Some(collection.clone()), SqlPlan::Join { left, .. } => extract_collection_name(left), - SqlPlan::Aggregate { input, .. } => extract_collection_name(input), - _ => String::new(), + _ => None, } } +/// Scan alias for a scan-shaped input. `None` for an input-sourced body. pub(in crate::control::planner::sql_plan_convert) fn extract_scan_alias( plan: &SqlPlan, ) -> Option { @@ -113,7 +119,6 @@ pub(in crate::control::planner::sql_plan_convert) fn extract_scan_alias( SqlPlan::Scan { alias, .. } => alias.clone(), SqlPlan::PointGet { alias, .. } => alias.clone(), SqlPlan::Join { left, .. } => extract_scan_alias(left), - SqlPlan::Aggregate { input, .. } => extract_scan_alias(input), _ => None, } } @@ -197,6 +202,70 @@ pub(super) fn group_by_to_strings(exprs: &[SqlExpr]) -> Vec { .collect() } +/// Build the coordinator-local `Aggregate{input: Some(child)}` task shared by +/// the catalog and derived-table-body lowerings: an aggregate that runs once +/// over an already-materialized `child` plan instead of scanning a per-shard +/// collection. `raw_collection` is the RAW catalog source name for a catalog +/// body, or `String::new()` for a body with no routing collection (derived +/// table, constant result, union) — either way the task's vshard is the +/// coordinator's empty-collection vshard, so it is never broadcast. +/// Inputs to [`build_input_sourced_aggregate_task`]. +pub(in crate::control::planner::sql_plan_convert) struct InputSourcedTaskParams<'a> { + pub tenant_id: TenantId, + pub ctx: &'a ConvertContext, + pub raw_collection: String, + pub child: PhysicalPlan, + pub group_by: &'a [SqlExpr], + pub aggregates: &'a [AggregateExpr], + pub having_bytes: Vec, + pub limit: usize, + pub sort_keys: Vec, +} + +pub(in crate::control::planner::sql_plan_convert) fn build_input_sourced_aggregate_task( + p: InputSourcedTaskParams<'_>, +) -> PhysicalTask { + let InputSourcedTaskParams { + tenant_id, + ctx, + raw_collection, + child, + group_by, + aggregates, + having_bytes, + limit, + sort_keys, + } = p; + let group_specs = group_by_to_specs(group_by); + let agg_specs: Vec = aggregates.iter().map(agg_expr_to_spec).collect(); + PhysicalTask { + tenant_id, + // Coordinator-local: empty collection keeps the task on the + // coordinator vshard (the child's rows are not per-shard). + vshard_id: VShardId::from_collection_in_database(ctx.database_id, ""), + database_id: ctx.database_id, + plan: PhysicalPlan::Query(QueryOp::Aggregate { + collection: nodedb_types::QualifiedCollection::from_stored(raw_collection), + input: Some(Box::new(child)), + group_by: group_specs, + aggregates: agg_specs, + // The child carries its own WHERE / ProviderScan filters; the + // aggregate node applies none of its own. + filters: Vec::new(), + having: having_bytes, + limit, + sub_group_by: Vec::new(), + sub_aggregates: Vec::new(), + // Guarded by the caller: input-sourced aggregates never carry + // grouping sets. + grouping_sets: Vec::new(), + sort_keys, + }), + post_set_op: PostSetOp::None, + txn_id: None, + } +} + /// Lower GROUP BY expressions to Data-Plane group-key specs. /// /// A bare `Column` key extracts from, and is emitted under, its own column diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/join.rs b/nodedb/src/control/planner/sql_plan_convert/scan/join.rs index f201e513f..2523ad394 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/join.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/join.rs @@ -108,6 +108,8 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_join( // keys its persisted column stats by the bare collection name, so the // shuffle cost model must look them up by the same raw name (not the // db-qualified token used for storage routing). + // An input-sourced join side carries no routing collection; its rows come + // from `left_input` / `right_input`. let mut left_raw = super::super::aggregate::extract_collection_name(left); let mut right_raw = super::super::aggregate::extract_collection_name(right); let mut left_alias = extract_scan_alias(left); @@ -210,7 +212,11 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_join( ); let shuffle_eligible = structurally_shufflable && (p.ctx.force_shuffle_join - || super::join_cost::cost_model_picks_shuffle(p.ctx, &left_raw, &right_raw)); + || super::join_cost::cost_model_picks_shuffle( + p.ctx, + left_raw.as_deref().unwrap_or(""), + right_raw.as_deref().unwrap_or(""), + )); // Shuffle hash keys mirror the resolver's per-side split: the LEFT column of // each `on` pair partitions the probe side, the RIGHT column the build side diff --git a/nodedb/src/control/server/exchange/resolve/exchange/aggregate_input_arm.rs b/nodedb/src/control/server/exchange/resolve/exchange/aggregate_input_arm.rs new file mode 100644 index 000000000..426b4bd25 --- /dev/null +++ b/nodedb/src/control/server/exchange/resolve/exchange/aggregate_input_arm.rs @@ -0,0 +1,105 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Input-sourced `Aggregate` resolution: materialize the aggregate's child on +//! the coordinator, then hand the Data Plane an aggregate over a +//! `ProviderScan` of those rows. + +use nodedb_physical::physical_plan::{ + AggregateSpec, GroupKeySpec, PhysicalPlan, QueryOp, SortKeySpec, +}; +use nodedb_types::QualifiedCollection; + +use crate::control::server::exchange::resolve::capture::DistributedReadCapture; +use crate::control::state::SharedState; + +use super::dispatch::ResolveCtx; +use super::entry::Resolved; +use super::post_process_arm::{ChildRows, materialize_child_rows}; + +/// Fields of a `QueryOp::Aggregate { input: Some(_) }` plan node, carried +/// through resolution as one value. +pub(super) struct AggregateFields { + pub collection: QualifiedCollection, + pub input: Box, + pub group_by: Vec, + pub aggregates: Vec, + pub filters: Vec, + pub having: Vec, + pub limit: usize, + pub sub_group_by: Vec, + pub sub_aggregates: Vec, + pub grouping_sets: Vec>, + pub sort_keys: Vec, +} + +/// Resolve an input-sourced `QueryOp::Aggregate`. +/// +/// A child that is already a materialized `ProviderScan{provider: None}` +/// (a catalog source filled by pass 1) passes through unchanged. Any other +/// child — an `Exchange{Gather}` over a sharded body, a `PostProcess`, a +/// constant result — is materialized on the coordinator and replaced by a +/// `ProviderScan` over its rows, so the aggregate runs exactly once over the +/// full relation and no Exchange reaches a Data-Plane core. +pub(super) async fn resolve_aggregate_input( + state: &SharedState, + ctx: ResolveCtx, + captures: &mut Vec, + fields: AggregateFields, +) -> crate::Result { + let AggregateFields { + collection, + input, + group_by, + aggregates, + filters, + having, + limit, + sub_group_by, + sub_aggregates, + grouping_sets, + sort_keys, + } = fields; + + let rebuild = |input: Box| { + Resolved::Plan(Box::new(PhysicalPlan::Query(QueryOp::Aggregate { + collection, + input: Some(input), + group_by, + aggregates, + filters, + having, + limit, + sub_group_by, + sub_aggregates, + grouping_sets, + sort_keys, + }))) + }; + + // Fast path: the child is already materialized rows. + if matches!( + *input, + PhysicalPlan::Query(QueryOp::ProviderScan { provider: None, .. }) + ) { + return Ok(rebuild(input)); + } + + let rows = match materialize_child_rows(state, ctx, captures, *input).await? { + ChildRows::Rows(rows) => rows, + ChildRows::Passthrough(resolved) => return Ok(resolved), + }; + Ok(rebuild(Box::new(PhysicalPlan::Query( + QueryOp::ProviderScan { + provider: None, + rows, + filters: Vec::new(), + projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), + sort_keys: Vec::new(), + limit: None, + offset: 0, + distinct: false, + }, + )))) +} diff --git a/nodedb/src/control/server/exchange/resolve/exchange/dispatch.rs b/nodedb/src/control/server/exchange/resolve/exchange/dispatch.rs index fcdccdade..125df5210 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/dispatch.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/dispatch.rs @@ -9,10 +9,11 @@ use crate::control::server::exchange::resolve::capture::DistributedReadCapture; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId, TraceId, TxnId}; +use super::aggregate_input_arm::AggregateFields; use super::entry::Resolved; use super::hash_join_arm::HashJoinFields; use super::post_process_arm::PostProcessFields; -use super::{gather_arm, hash_join_arm, post_process_arm, shuffle_arm}; +use super::{aggregate_input_arm, gather_arm, hash_join_arm, post_process_arm, shuffle_arm}; /// Request-scoped identifiers threaded through every arm resolver, bundled /// to keep each resolver's argument list within the clippy default arity. @@ -32,6 +33,8 @@ pub(super) struct ResolveCtx { /// - Root-level `Shuffle` wrapping a `HashJoin` → orchestrate a cross-node /// grace hash join, return `Resolved::Gathered`. `Shuffle` as a join input is /// a typed error. +/// - `Aggregate{input: Some}` → materialize the child on the coordinator and +/// embed it as `ProviderScan{None, rows}`, return `Resolved::Plan`. /// - Anything else → `Resolved::Plan` unchanged. /// /// `captures` accumulates one [`DistributedReadCapture`] per base collection an @@ -183,6 +186,44 @@ pub(super) async fn resolve_exchange( .await } + // Input-sourced Aggregate: materialize the child on the coordinator + // (unless it is already a `ProviderScan` of rows) and aggregate over + // those rows once. The aggregate is coordinator-local, so the root + // Gather arm never sees it; the child is gathered here instead. + PhysicalPlan::Query(QueryOp::Aggregate { + collection, + input: Some(input), + group_by, + aggregates, + filters, + having, + limit, + sub_group_by, + sub_aggregates, + grouping_sets, + sort_keys, + }) => { + aggregate_input_arm::resolve_aggregate_input( + state, + ctx, + captures, + AggregateFields { + collection, + input, + group_by, + aggregates, + filters, + having, + limit, + sub_group_by, + sub_aggregates, + grouping_sets, + sort_keys, + }, + ) + .await + } + // All other plan variants: pass through unchanged. other => Ok(Resolved::Plan(Box::new(other))), } diff --git a/nodedb/src/control/server/exchange/resolve/exchange/mod.rs b/nodedb/src/control/server/exchange/resolve/exchange/mod.rs index 2bf036327..dff2214a4 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/mod.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/mod.rs @@ -12,8 +12,12 @@ //! cross-node grace hash join (`super::shuffle`) and return the merged rows //! as `Resolved::Gathered`. `Shuffle` as a join INPUT is a typed error (it //! only ever wraps a complete join). +//! - `Aggregate{input: Some}` whose child is not yet materialized rows → +//! materialize the child on the coordinator and embed it as +//! `ProviderScan{provider: None, rows}`; return `Resolved::Plan`. //! - No Exchange / no empty ProviderScan → `Resolved::Plan` unchanged. +mod aggregate_input_arm; mod dispatch; mod entry; mod gather_arm; diff --git a/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs b/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs index 0bc66c706..c5e5de523 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs @@ -5,12 +5,14 @@ use nodedb_physical::physical_plan::{ ExchangeMode, ExchangeOp, PhysicalPlan, QueryOp, SortKeySpec, TextOp, VectorOp, + plan_contains_cluster_partitioned_leaf, }; use crate::control::server::exchange::full_scan::{ScanSide, full_scan_plan_for_collection}; use crate::control::server::exchange::gather::{ GatherOutcome, finalize_aggregate, gather_all_vshards, }; +use crate::control::server::exchange::owning_core::gather_single_owning_core; use crate::control::server::exchange::resolve::capture::DistributedReadCapture; use crate::control::server::response_translate::hit_key::parse_surrogate_hex; use crate::control::server::response_translate::vector::resolve_surrogate_pk; @@ -19,6 +21,7 @@ use crate::data::executor::response_codec::{ flatten_hybrid_hits_to_relational_rows, flatten_to_relational_rows, flatten_vector_hits_to_relational_rows, }; +use crate::types::VShardId; use super::dispatch::{ResolveCtx, resolve_exchange}; use super::entry::Resolved; @@ -88,39 +91,41 @@ fn hit_collection_name(plan: &PhysicalPlan) -> Option { } } -/// Resolve a `QueryOp::PostProcess` node: materialize the child's rows on the -/// coordinator, then lower to a `ProviderScan` that applies filter → offset → -/// sort → distinct → project → limit on a single core (its existing tail). -/// This keeps "run exactly once over the full union" correct: the child is -/// gathered here, so the relational tail never runs per-shard. -pub(super) async fn resolve_post_process( +/// Rows of a materialized child, or a resolution the caller returns as-is. +pub(super) enum ChildRows { + /// The child's rows, flattened to the bare relational row shape a + /// `ProviderScan{provider: None}` consumes. + Rows(Vec), + /// The child resolved to a root `Gathered` / `Stream` result. The caller + /// returns it unchanged. + Passthrough(Resolved), +} + +/// Materialize a coordinator-side child plan into relational rows. +/// +/// Unwraps the converter's `Exchange{Gather}` wrapper, resolves any Exchange +/// nested inside the body (a `HashJoin` build-side `Broadcast`), gathers the +/// body across all vShards, finalizes a partial-aggregate payload, and +/// flattens hit-shaped payloads (vector / hybrid) to columned rows with the +/// surrogate resolved to the user PK. An in-transaction read records the +/// child's base collection in `captures` at its observed read-version. +pub(super) async fn materialize_child_rows( state: &SharedState, ctx: ResolveCtx, captures: &mut Vec, - fields: PostProcessFields, -) -> crate::Result { + input: PhysicalPlan, +) -> crate::Result { let ResolveCtx { database_id, tenant_id, trace_id, txn_id, } = ctx; - let PostProcessFields { - input, - filters, - projection, - computed_columns, - window_functions, - sort_keys, - limit, - offset, - distinct, - } = fields; // The converter wraps a sharded body in `Exchange{Gather}`; unwrap // it so the child is the real body plan (a plain body has no // wrapper and routes to its owning vShard directly). - let (child, as_aggregate) = match *input { + let (child, as_aggregate) = match input { PhysicalPlan::Query(QueryOp::Exchange(ExchangeOp { child, mode: ExchangeMode::Gather { as_aggregate }, @@ -144,11 +149,11 @@ pub(super) async fn resolve_post_process( { Resolved::Plan(p) => *p, // The unwrapped body is not itself a root Gather / stream; - // surface these defensively without dropping post-processing. + // surface these without dropping the caller's tail. Resolved::Gathered(resp, wms, caps) => { - return Ok(Resolved::Gathered(resp, wms, caps)); + return Ok(ChildRows::Passthrough(Resolved::Gathered(resp, wms, caps))); } - Resolved::Stream(s) => return Ok(Resolved::Stream(s)), + Resolved::Stream(s) => return Ok(ChildRows::Passthrough(Resolved::Stream(s))), }; // Classify the body's row shape so the gathered payload is @@ -178,8 +183,27 @@ pub(super) async fn resolve_post_process( None }; - let outcome: GatherOutcome = - gather_all_vshards(state, tenant_id, database_id, child, trace_id, txn_id).await?; + // A coordinator-local body (a `ProviderScan` carrying embedded rows, a + // nested `PostProcess`) reads no per-shard collection. It runs exactly + // once on the coordinator vshard: fanning it to every core returns its + // rows once per core. + let coordinator_local = child.collection().is_none() + && !child.is_sharded_source() + && !plan_contains_cluster_partitioned_leaf(&child); + let outcome: GatherOutcome = if coordinator_local { + gather_single_owning_core( + state, + tenant_id, + database_id, + child, + VShardId::from_collection_in_database(database_id, ""), + trace_id, + txn_id, + ) + .await? + } else { + gather_all_vshards(state, tenant_id, database_id, child, trace_id, txn_id).await? + }; if let Some(coll) = probe_collection && let Some(scan_plan) = full_scan_plan_for_collection( @@ -221,6 +245,36 @@ pub(super) async fn resolve_post_process( }), HitShape::None => flatten_to_relational_rows(&merged), }; + Ok(ChildRows::Rows(rows)) +} + +/// Resolve a `QueryOp::PostProcess` node: materialize the child's rows on the +/// coordinator, then lower to a `ProviderScan` that applies filter → offset → +/// sort → distinct → project → limit on a single core (its existing tail). +/// This keeps "run exactly once over the full union" correct: the child is +/// gathered here, so the relational tail never runs per-shard. +pub(super) async fn resolve_post_process( + state: &SharedState, + ctx: ResolveCtx, + captures: &mut Vec, + fields: PostProcessFields, +) -> crate::Result { + let PostProcessFields { + input, + filters, + projection, + computed_columns, + window_functions, + sort_keys, + limit, + offset, + distinct, + } = fields; + + let rows = match materialize_child_rows(state, ctx, captures, *input).await? { + ChildRows::Rows(rows) => rows, + ChildRows::Passthrough(resolved) => return Ok(resolved), + }; Ok(Resolved::Plan(Box::new(PhysicalPlan::Query( QueryOp::ProviderScan { provider: None, diff --git a/nodedb/src/data/executor/handlers/aggregate/exec.rs b/nodedb/src/data/executor/handlers/aggregate/exec.rs index 6629f5c7a..9ee91777f 100644 --- a/nodedb/src/data/executor/handlers/aggregate/exec.rs +++ b/nodedb/src/data/executor/handlers/aggregate/exec.rs @@ -8,7 +8,7 @@ use tracing::debug; use super::cache_key::{AggregateCacheKeyInputs, aggregate_cache_key, legacy_aggregate_pairs}; use super::rows::{apply_user_aliases_to_rows, sort_aggregated_rows}; -use crate::bridge::envelope::{ErrorCode, Response}; +use crate::bridge::envelope::{ErrorCode, Response, Status}; use crate::bridge::scan_filter::ScanFilter; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; @@ -58,22 +58,41 @@ impl CoreLoop { debug!(core = self.core_id, %collection, has_input = input.is_some(), group_fields = group_by.len(), aggs = aggregates.len(), "aggregate"); - // Input-sourced aggregate (catalog): the rows come from executing the - // sub-plan (a coordinator-materialized `ProviderScan`), not from a - // per-shard collection scan. Decode the sub-plan rows and aggregate - // over them using the same streaming logic, then short-circuit before - // the per-shard fast paths (cache / index-backed / columnar memtable), - // none of which apply to coordinator-local catalog data. + // Input-sourced aggregate: the rows come from executing the sub-plan + // (a coordinator-materialized `ProviderScan` over catalog rows or a + // derived-table body), not from a per-shard collection scan. Decode + // the sub-plan rows and aggregate over them using the same streaming + // logic, then short-circuit before the per-shard fast paths (cache / + // index-backed / columnar memtable), none of which apply to + // coordinator-local rows. if let Some(sub_plan) = input { let sub_response = self.execute_plan(task, sub_plan); - // Empty / undecodable payload → aggregate over zero rows, the same as - // a per-shard scan that matched nothing. Feeding an empty doc set - // through the shared path keeps behavior identical to the scan path - // rather than surfacing the sub-plan Response (which may be a - // non-row payload). - let docs = - crate::data::executor::response_codec::decode_response_to_docs(&sub_response) - .unwrap_or_default(); + // A child error (22012 from a computed column, a resolver refusal) + // fails the statement. + if sub_response.status == Status::Error { + return sub_response; + } + // An empty payload is zero rows. A non-empty payload that is not + // a MessagePack row array is an internal error, never an empty + // aggregate. + let docs = if sub_response.payload.is_empty() { + Vec::new() + } else { + match crate::data::executor::response_codec::decode_response_to_docs(&sub_response) + { + Some(docs) => docs, + None => { + return self.response_error( + task, + ErrorCode::Internal { + detail: "aggregate input rows failed to decode: payload is not \ + a MessagePack row array" + .to_string(), + }, + ); + } + } + }; return self.aggregate_over_docs( super::streaming::over_docs::AggregateOverDocsParams { task, diff --git a/nodedb/tests/wire/cases/sql_subquery_from.rs b/nodedb/tests/wire/cases/sql_subquery_from.rs index 016d888a8..ebd420987 100644 --- a/nodedb/tests/wire/cases/sql_subquery_from.rs +++ b/nodedb/tests/wire/cases/sql_subquery_from.rs @@ -328,11 +328,11 @@ async fn aggregate_over_grouped_derived_table_evaluates() { .await .expect("aggregate over a grouped derived table must plan"); - assert_eq!( - rows, - vec![vec!["15".to_string(), "3".to_string()]], - "got {rows:?}" - ); + // SUM renders as a float text today; compare numerically. + assert_eq!(rows.len(), 1, "got {rows:?}"); + let grand: f64 = rows[0][0].parse().expect("grand total must be numeric"); + assert_eq!(grand, 15.0, "got {rows:?}"); + assert_eq!(rows[0][1], "3", "got {rows:?}"); } /// An aggregate over a UNION ALL derived table must run over the union rows. @@ -345,7 +345,10 @@ async fn aggregate_over_union_derived_table_evaluates() { .await .expect("aggregate over a UNION ALL derived table must plan"); - assert_eq!(rows, vec![vec!["3".to_string()]], "got {rows:?}"); + // SUM renders as a float text today; compare numerically. + assert_eq!(rows.len(), 1, "got {rows:?}"); + let total: f64 = rows[0][0].parse().expect("total must be numeric"); + assert_eq!(total, 3.0, "got {rows:?}"); } /// A window function over a grouped derived table must rank the inner group From a1a31997ee51062e57d0ecf2059b47ced0cc427d Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 18 Sep 2026 01:27:31 +0800 Subject: [PATCH 07/15] feat(query): evaluate query tails over set-operation bodies MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add a coordinator-resolved SetOp physical plan node so UNION/UNION ALL/INTERSECT/EXCEPT bodies can sit under a subquery tail (ORDER BY, DISTINCT, OFFSET/LIMIT, aggregate) instead of only being reachable at the top level. Body-to-single-plan lowering is factored out of convert_subquery into a shared convert_body_to_single_plan, which now also recognizes a set-operation body and lowers it to the new SetOp node instead of rejecting multi-task bodies outright. The exchange resolver gains a SetOp arm that materializes each branch task, extracted alongside a shared materialize_child_rows helper, and dispatches to shared merge logic for UNION/UNION ALL/INTERSECT/EXCEPT. That merge logic — previously private to the pgwire response path — moves into a new set_op_merge module (union dedup, INTERSECT/EXCEPT row matching, and the row-key normalization they share) so both the pgwire routing path and the exchange resolver call the same implementation instead of duplicating it. Constant-result rows and cell values are now built as typed nodedb_types::Value/HashMap instead of a serde_json::Value tree before encoding to msgpack, matching how other response paths carry shaped rows. --- .../src/physical_plan/collection.rs | 2 + nodedb-physical/src/physical_plan/mod.rs | 2 + nodedb-physical/src/physical_plan/query.rs | 15 + nodedb-physical/src/physical_plan/routing.rs | 8 + nodedb-physical/src/physical_plan/set_op.rs | 33 ++ nodedb/src/control/clone/resolver/rewrite.rs | 50 ++- nodedb/src/control/exec_receiver/support.rs | 3 + nodedb/src/control/gateway/version_set.rs | 7 + .../control/planner/redaction_refusal/plan.rs | 2 + .../rls_injection/permission_tree/query.rs | 4 + .../control/planner/rls_injection/query.rs | 4 + .../aggregate/input_sourced.rs | 49 +-- .../control/planner/sql_plan_convert/body.rs | 241 +++++++++++++ .../control/planner/sql_plan_convert/mod.rs | 1 + .../sql_plan_convert/output_schema/build.rs | 51 ++- .../planner/sql_plan_convert/set_ops.rs | 139 +++++--- .../security/identity/plan_permission.rs | 35 ++ .../resolve/exchange/aggregate_input_arm.rs | 17 +- .../exchange/resolve/exchange/dispatch.rs | 13 +- .../server/exchange/resolve/exchange/mod.rs | 2 + .../resolve/exchange/post_process_arm.rs | 18 + .../exchange/resolve/exchange/set_op_arm.rs | 120 +++++++ .../server/exchange/resolve/join_input.rs | 43 +-- .../server/exchange/resolve/materialize.rs | 10 + nodedb/src/control/server/mod.rs | 1 + .../server/pgwire/handler/routing/set_ops.rs | 328 +----------------- .../server/response_shape/redaction/query.rs | 8 + .../server/response_shape/types/plan_kind.rs | 4 + .../server/set_op_merge/intersect_except.rs | 102 ++++++ nodedb/src/control/server/set_op_merge/mod.rs | 13 + .../control/server/set_op_merge/row_key.rs | 129 +++++++ .../src/control/server/set_op_merge/union.rs | 55 +++ .../authorization/requirements/query.rs | 5 + .../server/shared/clone_read/temporal.rs | 3 + nodedb/src/control/server/shared/plan_util.rs | 6 + .../predicate/txn_buffering/classify.rs | 3 +- nodedb/src/data/executor/dispatch/query.rs | 9 + .../tests/wire/cases/pgwire_extended_query.rs | 10 +- .../cases/pgwire_extended_query_engines2.rs | 4 +- 39 files changed, 1060 insertions(+), 489 deletions(-) create mode 100644 nodedb-physical/src/physical_plan/set_op.rs create mode 100644 nodedb/src/control/planner/sql_plan_convert/body.rs create mode 100644 nodedb/src/control/server/exchange/resolve/exchange/set_op_arm.rs create mode 100644 nodedb/src/control/server/set_op_merge/intersect_except.rs create mode 100644 nodedb/src/control/server/set_op_merge/mod.rs create mode 100644 nodedb/src/control/server/set_op_merge/row_key.rs create mode 100644 nodedb/src/control/server/set_op_merge/union.rs diff --git a/nodedb-physical/src/physical_plan/collection.rs b/nodedb-physical/src/physical_plan/collection.rs index 50a6ef243..90001efca 100644 --- a/nodedb-physical/src/physical_plan/collection.rs +++ b/nodedb-physical/src/physical_plan/collection.rs @@ -113,6 +113,8 @@ impl PhysicalPlan { PhysicalPlan::Query(QueryOp::Exchange(op)) => op.child.collection(), // PostProcess: recurse into the materialized input plan. PhysicalPlan::Query(QueryOp::PostProcess { input, .. }) => input.collection(), + // SetOp merges N branches; no single collection names the node. + PhysicalPlan::Query(QueryOp::SetOp { .. }) => None, // ProviderScan is a catalog/constant source — no user collection. PhysicalPlan::Query(QueryOp::ProviderScan { .. }) => None, // KV ops carry their own collection (sorted-index-only ops → None). diff --git a/nodedb-physical/src/physical_plan/mod.rs b/nodedb-physical/src/physical_plan/mod.rs index 98967dfb2..0df5f1b35 100644 --- a/nodedb-physical/src/physical_plan/mod.rs +++ b/nodedb-physical/src/physical_plan/mod.rs @@ -22,6 +22,7 @@ pub mod plan; pub mod query; pub mod rls_write_check_accessor; pub mod routing; +pub mod set_op; pub mod sort_key; pub mod spatial; pub mod streaming; @@ -51,6 +52,7 @@ pub use meta::MetaOp; pub use plan::PhysicalPlan; pub use query::{AggregateSpec, GroupKeySpec, JoinProjection, QueryOp}; pub use routing::plan_contains_cluster_partitioned_leaf; +pub use set_op::SetOpKind; pub use sort_key::SortKeySpec; pub use spatial::{SpatialOp, SpatialPredicate}; pub use text::TextOp; diff --git a/nodedb-physical/src/physical_plan/query.rs b/nodedb-physical/src/physical_plan/query.rs index 110bf35b5..67e023498 100644 --- a/nodedb-physical/src/physical_plan/query.rs +++ b/nodedb-physical/src/physical_plan/query.rs @@ -149,6 +149,21 @@ pub enum QueryOp { distinct: bool, }, + /// Set operation over N materialized children. Coordinator-only: the + /// resolver materializes every child and merges the rows into one + /// `ProviderScan` before dispatch. A Data-Plane core never sees this node. + /// + /// Lowered from a derived-table body that is `UNION [ALL]`, + /// `INTERSECT [ALL]`, or `EXCEPT [ALL]`, so the body is one relation for + /// an outer [`QueryOp::PostProcess`] or input-sourced [`QueryOp::Aggregate`]. + SetOp { + /// Child relations in SQL order. Each sharded child is wrapped in + /// `Exchange{Gather}` by the converter so its gather runs once. + inputs: Vec, + /// Which set operation merges the inputs. + op: crate::physical_plan::SetOpKind, + }, + /// Aggregate: GROUP BY + aggregate functions. Aggregate { collection: QualifiedCollection, diff --git a/nodedb-physical/src/physical_plan/routing.rs b/nodedb-physical/src/physical_plan/routing.rs index 0fda15961..ba224ae09 100644 --- a/nodedb-physical/src/physical_plan/routing.rs +++ b/nodedb-physical/src/physical_plan/routing.rs @@ -68,6 +68,11 @@ pub fn plan_contains_cluster_partitioned_leaf(plan: &PhysicalPlan) -> bool { plan_contains_cluster_partitioned_leaf(input) } + // Recurse through every SetOp branch for the same reason. + PhysicalPlan::Query(QueryOp::SetOp { inputs, .. }) => { + inputs.iter().any(plan_contains_cluster_partitioned_leaf) + } + // Recurse through lateral outer plans. PhysicalPlan::Query(QueryOp::LateralTopK { outer_plan, .. }) | PhysicalPlan::Query(QueryOp::LateralLoop { outer_plan, .. }) => { @@ -159,6 +164,9 @@ impl PhysicalPlan { right_bitmap, ) } + // Coordinator-local: the resolver materializes every branch and + // merges on the coordinator, so the node itself is never fanned out. + PhysicalPlan::Query(QueryOp::SetOp { .. }) => false, _ => self.is_sharded_source_leaf(), } } diff --git a/nodedb-physical/src/physical_plan/set_op.rs b/nodedb-physical/src/physical_plan/set_op.rs new file mode 100644 index 000000000..d4f3c1ae1 --- /dev/null +++ b/nodedb-physical/src/physical_plan/set_op.rs @@ -0,0 +1,33 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Set-operation kinds for [`crate::physical_plan::QueryOp::SetOp`]. + +/// Which SQL set operation a [`crate::physical_plan::QueryOp::SetOp`] node +/// applies over its materialized inputs. Coordinator-resolved, never reaches +/// a Data-Plane core. +#[derive( + Debug, + Clone, + Copy, + PartialEq, + Eq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +#[msgpack(c_enum)] +pub enum SetOpKind { + /// `UNION ALL`: concatenate every input in order. + UnionAll, + /// `UNION`: concatenate, then drop duplicate rows. + UnionDistinct, + /// `INTERSECT`: rows present in every input, deduplicated. + Intersect, + /// `INTERSECT ALL`: rows present in every input, bag semantics. + IntersectAll, + /// `EXCEPT`: rows of the first input absent from the rest, deduplicated. + Except, + /// `EXCEPT ALL`: rows of the first input absent from the rest, bag semantics. + ExceptAll, +} diff --git a/nodedb/src/control/clone/resolver/rewrite.rs b/nodedb/src/control/clone/resolver/rewrite.rs index c2f7df732..f1189ebbd 100644 --- a/nodedb/src/control/clone/resolver/rewrite.rs +++ b/nodedb/src/control/clone/resolver/rewrite.rs @@ -8,7 +8,7 @@ use nodedb_types::TenantId; use crate::control::state::SharedState; use nodedb_physical::physical_plan::{ - ColumnarOp, DocumentOp, ExchangeOp, KvOp, PhysicalPlan, QueryOp, TimeseriesOp, + ColumnarOp, DocumentOp, ExchangeOp, KvOp, PhysicalPlan, QueryOp, SetOpKind, TimeseriesOp, }; use nodedb_types::SystemTimeScope; @@ -153,6 +153,54 @@ pub fn rewrite_plan_for_source(params: RewriteForSourceParams<'_>) -> crate::Res }) } + // SetOp: rewrite every branch. Branches that do not read the cloned + // collection yield no source task and are dropped, so the source-side + // node carries only the rows the target side is missing. That is + // sound for `UNION ALL` (the merge appends). Any other kind dedups or + // subtracts by exact row match against the target rows, which is + // unsound across an unmaterialized clone; refuse it the same way the + // task-level `post_set_op` refusal does. + PhysicalPlan::Query(QueryOp::SetOp { inputs, op }) => { + let mut rewritten_inputs = Vec::with_capacity(inputs.len()); + for input in inputs { + let rewritten = rewrite_plan_for_source(RewriteForSourceParams { + plan: input, + target_db_id, + source_db_id, + tenant_id, + target_coll, + source_coll, + effective_source_ms, + kv_surrogate_ceiling, + state, + })?; + if let SourceRewrite::Task(child) = rewritten { + rewritten_inputs.push(*child); + } + } + if rewritten_inputs.is_empty() { + return Ok(SourceRewrite::NoSourceTask); + } + match op { + SetOpKind::UnionAll => { + Ok(SourceRewrite::task(PhysicalPlan::Query(QueryOp::SetOp { + inputs: rewritten_inputs, + op: SetOpKind::UnionAll, + }))) + } + SetOpKind::UnionDistinct + | SetOpKind::Intersect + | SetOpKind::IntersectAll + | SetOpKind::Except + | SetOpKind::ExceptAll => Err(crate::Error::PlanError { + detail: format!( + "a set operation over '{target_coll}' cannot be read through an \ + unmaterialized clone; run ALTER DATABASE MATERIALIZE first" + ), + }), + } + } + PhysicalPlan::Document(DocumentOp::Scan { collection, limit, diff --git a/nodedb/src/control/exec_receiver/support.rs b/nodedb/src/control/exec_receiver/support.rs index b6c802352..c0a8b72cf 100644 --- a/nodedb/src/control/exec_receiver/support.rs +++ b/nodedb/src/control/exec_receiver/support.rs @@ -55,6 +55,9 @@ pub(super) fn plan_contains_exchange(plan: &PhysicalPlan) -> bool { // resolution, still carries an `Exchange{Gather}` — recurse so an // unresolved PostProcess is correctly flagged as Exchange-bearing. QueryOp::PostProcess { input, .. } => plan_contains_exchange(input), + // SetOp inputs are unresolved bodies; any of them can carry a + // Gather. + QueryOp::SetOp { inputs, .. } => inputs.iter().any(plan_contains_exchange), // Aggregate may carry a sub-plan input (catalog `ProviderScan`), // which could in principle nest an Exchange — recurse when present. QueryOp::Aggregate { input, .. } => { diff --git a/nodedb/src/control/gateway/version_set.rs b/nodedb/src/control/gateway/version_set.rs index cd5639ba5..fffb0cdb9 100644 --- a/nodedb/src/control/gateway/version_set.rs +++ b/nodedb/src/control/gateway/version_set.rs @@ -479,6 +479,13 @@ pub fn touched_collections(plan: &PhysicalPlan) -> Vec { out.extend(touched_collections(input)); } + // SetOp: every branch is a body that reads its own collections. + SetOp { inputs, .. } => { + for input in inputs { + out.extend(touched_collections(input)); + } + } + // ProviderScan is a catalog/constant source — no user collection. ProviderScan { .. } => {} diff --git a/nodedb/src/control/planner/redaction_refusal/plan.rs b/nodedb/src/control/planner/redaction_refusal/plan.rs index 4e1f08fe0..aab8e64ab 100644 --- a/nodedb/src/control/planner/redaction_refusal/plan.rs +++ b/nodedb/src/control/planner/redaction_refusal/plan.rs @@ -198,6 +198,8 @@ fn walk_query(op: &QueryOp, ctx: &RefusalCtx<'_>) -> crate::Result<()> { QueryOp::PostProcess { input, .. } => walk(input, ctx), + QueryOp::SetOp { inputs, .. } => inputs.iter().try_for_each(|input| walk(input, ctx)), + QueryOp::Aggregate { collection, input, diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/query.rs b/nodedb/src/control/planner/rls_injection/permission_tree/query.rs index 3667a3723..854a2e0e6 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/query.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/query.rs @@ -21,6 +21,10 @@ pub(super) fn apply_query(ctx: &PermCtx<'_>, op: &mut QueryOp) -> crate::Result< // tree restricts. QueryOp::PostProcess { input, .. } => walk(ctx, input), + // Recurse: every set-operation branch is its own body whose rows the + // policy restricts. + QueryOp::SetOp { inputs, .. } => inputs.iter_mut().try_for_each(|input| walk(ctx, input)), + // Filter and recurse: the aggregate handler evaluates `filters` // against both row sources — the per-shard collection scan and the // rows decoded from an embedded sub-plan — so the subtree filter goes diff --git a/nodedb/src/control/planner/rls_injection/query.rs b/nodedb/src/control/planner/rls_injection/query.rs index 2c0bd05dd..5ba24f397 100644 --- a/nodedb/src/control/planner/rls_injection/query.rs +++ b/nodedb/src/control/planner/rls_injection/query.rs @@ -21,6 +21,10 @@ pub(super) fn inject_query(ctx: &RlsCtx<'_>, op: &mut QueryOp) -> crate::Result< // policy restricts. QueryOp::PostProcess { input, .. } => walk(ctx, input), + // Recurse: every set-operation branch is its own body whose rows the + // policy restricts. + QueryOp::SetOp { inputs, .. } => inputs.iter_mut().try_for_each(|input| walk(ctx, input)), + // Inject or recurse: a catalog aggregate (`input: Some`) sources rows // from the embedded sub-plan, so the policy belongs in that input // rather than in the aggregate's own (empty) filters. A legacy diff --git a/nodedb/src/control/planner/sql_plan_convert/aggregate/input_sourced.rs b/nodedb/src/control/planner/sql_plan_convert/aggregate/input_sourced.rs index e4e9f289f..bdafd78c3 100644 --- a/nodedb/src/control/planner/sql_plan_convert/aggregate/input_sourced.rs +++ b/nodedb/src/control/planner/sql_plan_convert/aggregate/input_sourced.rs @@ -7,12 +7,12 @@ use nodedb_sql::types::{AggregateExpr, Filter, SqlExpr, SqlPlan}; -use crate::bridge::envelope::PhysicalPlan; use crate::types::TenantId; use nodedb_physical::physical_plan::*; use nodedb_physical::physical_task::PhysicalTask; -use super::super::convert::{ConvertContext, convert_one}; +use super::super::body::convert_body_to_single_plan; +use super::super::convert::ConvertContext; use super::super::filter::serialize_filters; use super::spec::{InputSourcedTaskParams, build_input_sourced_aggregate_task}; @@ -31,9 +31,10 @@ pub(super) struct InputSourcedAggregateParams<'a> { /// Lower an aggregate whose input is a materialized relation. /// -/// The body converts through `convert_one` and must produce exactly one task. -/// A sharded body is wrapped in `Exchange{Gather}` so the coordinator resolves -/// it to a `ProviderScan` before the aggregate runs. The emitted task is +/// The body lowers to ONE relation through `convert_body_to_single_plan`: a +/// set-operation body becomes a coordinator-resolved `SetOp`, and a sharded +/// body is wrapped in `Exchange{Gather}` so the coordinator resolves it to a +/// `ProviderScan` before the aggregate runs. The emitted task is /// coordinator-local: an empty collection keeps it on the coordinator vshard /// and `is_sharded_source` reports the `Some(input)` aggregate as /// non-sharded, so it runs once and is never broadcast. @@ -61,41 +62,9 @@ pub(super) fn convert_input_sourced_aggregate( }); } - // The body is one relation. A body that lowers to several tasks (a set - // operation) has no single row stream to aggregate. - let mut body = convert_one(input, tenant_id, ctx)?; - if body.len() != 1 { - return Err(crate::Error::PlanError { - detail: format!( - "aggregate over a derived-table body that lowers to {} physical tasks is not \ - supported; the body must produce a single relation", - body.len() - ), - }); - } - let mut child = match body.pop() { - Some(task) => task.plan, - None => { - return Err(crate::Error::PlanError { - detail: "aggregate over a derived-table body produced no physical task".to_string(), - }); - } - }; - - // A sharded body is gathered first so the aggregate observes the FULL - // union exactly once. The aggregate task is coordinator-local, so the - // top-level `convert()` wrap loop does not gather the child. - if child.is_sharded_source() { - let as_aggregate = matches!( - &child, - PhysicalPlan::Query(QueryOp::Aggregate { .. }) - | PhysicalPlan::Query(QueryOp::PartialAggregate { .. }) - ); - child = PhysicalPlan::Query(QueryOp::Exchange(ExchangeOp { - child: Box::new(child), - mode: ExchangeMode::Gather { as_aggregate }, - })); - } + // The body is ONE relation, already gathered when sharded, so the + // aggregate observes the full union exactly once. + let child = convert_body_to_single_plan(input, tenant_id, ctx)?; let having_bytes = serialize_filters(having)?; diff --git a/nodedb/src/control/planner/sql_plan_convert/body.rs b/nodedb/src/control/planner/sql_plan_convert/body.rs new file mode 100644 index 000000000..38720030c --- /dev/null +++ b/nodedb/src/control/planner/sql_plan_convert/body.rs @@ -0,0 +1,241 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Derived-table body lowering: one `SqlPlan` body to ONE physical relation, +//! for a post-processor or an input-sourced aggregate. + +use nodedb_sql::types::SqlPlan; + +use crate::bridge::envelope::PhysicalPlan; +use crate::types::TenantId; +use nodedb_physical::physical_plan::{ExchangeMode, ExchangeOp, QueryOp, SetOpKind}; + +use super::convert::{ConvertContext, convert_one}; + +/// Lower a derived-table body to ONE physical relation for a post-processor +/// or an input-sourced aggregate. +/// +/// A set-operation body lowers to a coordinator-resolved `QueryOp::SetOp` +/// whose branches recurse through this function, so nested set operations +/// (`(a UNION b) INTERSECT c`) stay one relation. Every other body lowers +/// through `convert_one` and must yield exactly one task. +/// +/// A sharded relation (the body itself, or any set-operation branch) is +/// wrapped in `Exchange{Gather}` so its gather runs exactly once over the +/// full union before the enclosing tail or merge observes it. `SetOp` is +/// coordinator-local, so it is never wrapped itself. +pub(super) fn convert_body_to_single_plan( + input: &SqlPlan, + tenant_id: TenantId, + ctx: &ConvertContext, +) -> crate::Result { + match input { + SqlPlan::Union { inputs, distinct } => { + let op = if *distinct { + SetOpKind::UnionDistinct + } else { + SetOpKind::UnionAll + }; + let inputs = inputs + .iter() + .map(|branch| convert_body_to_single_plan(branch, tenant_id, ctx)) + .collect::>>()?; + Ok(PhysicalPlan::Query(QueryOp::SetOp { inputs, op })) + } + SqlPlan::Intersect { left, right, all } => { + let op = if *all { + SetOpKind::IntersectAll + } else { + SetOpKind::Intersect + }; + convert_binary_set_op(left, right, op, tenant_id, ctx) + } + SqlPlan::Except { left, right, all } => { + let op = if *all { + SetOpKind::ExceptAll + } else { + SetOpKind::Except + }; + convert_binary_set_op(left, right, op, tenant_id, ctx) + } + // Every other body lowers through the ordinary converter. Listed in + // full so a new `SqlPlan` variant forces a decision here. + SqlPlan::ConstantResult { .. } + | SqlPlan::Scan { .. } + | SqlPlan::PointGet { .. } + | SqlPlan::DocumentIndexLookup { .. } + | SqlPlan::RangeScan { .. } + | SqlPlan::Insert { .. } + | SqlPlan::KvInsert { .. } + | SqlPlan::Upsert { .. } + | SqlPlan::InsertSelect { .. } + | SqlPlan::Update { .. } + | SqlPlan::UpdateFrom { .. } + | SqlPlan::Delete { .. } + | SqlPlan::Truncate { .. } + | SqlPlan::Join { .. } + | SqlPlan::Aggregate { .. } + | SqlPlan::TimeseriesScan { .. } + | SqlPlan::TimeseriesIngest { .. } + | SqlPlan::VectorSearch { .. } + | SqlPlan::MultiVectorSearch { .. } + | SqlPlan::SparseSearch { .. } + | SqlPlan::TextSearch { .. } + | SqlPlan::HybridSearch { .. } + | SqlPlan::HybridSearchTriple { .. } + | SqlPlan::SpatialScan { .. } + | SqlPlan::RecursiveScan { .. } + | SqlPlan::RecursiveValue { .. } + | SqlPlan::Cte { .. } + | SqlPlan::Subquery { .. } + | SqlPlan::CreateArray { .. } + | SqlPlan::DropArray { .. } + | SqlPlan::AlterArray { .. } + | SqlPlan::InsertArray { .. } + | SqlPlan::DeleteArray { .. } + | SqlPlan::ArraySlice { .. } + | SqlPlan::ArrayProject { .. } + | SqlPlan::ArrayAgg { .. } + | SqlPlan::ArrayElementwise { .. } + | SqlPlan::ArrayFlush { .. } + | SqlPlan::ArrayCompact { .. } + | SqlPlan::Merge { .. } + | SqlPlan::LateralTopK { .. } + | SqlPlan::LateralLoop { .. } + | SqlPlan::VectorPrimaryInsert { .. } + | SqlPlan::CreateIndex { .. } + | SqlPlan::DropIndex { .. } => { + let mut tasks = convert_one(input, tenant_id, ctx)?; + let plan = match (tasks.len(), tasks.pop()) { + (1, Some(task)) => task.plan, + (n, _) => { + return Err(crate::Error::PlanError { + detail: format!( + "derived-table body lowers to {n} physical tasks; the body must \ + produce a single relation" + ), + }); + } + }; + Ok(gather_if_sharded(plan)) + } + } +} + +/// Lower the two sides of `INTERSECT` / `EXCEPT` to a two-branch `SetOp`. +fn convert_binary_set_op( + left: &SqlPlan, + right: &SqlPlan, + op: SetOpKind, + tenant_id: TenantId, + ctx: &ConvertContext, +) -> crate::Result { + let inputs = vec![ + convert_body_to_single_plan(left, tenant_id, ctx)?, + convert_body_to_single_plan(right, tenant_id, ctx)?, + ]; + Ok(PhysicalPlan::Query(QueryOp::SetOp { inputs, op })) +} + +/// Wrap a sharded relation in `Exchange{Gather}` so its gather runs exactly +/// once over the full union. The enclosing node is coordinator-local, so the +/// top-level `convert()` wrap loop does not gather it. +fn gather_if_sharded(plan: PhysicalPlan) -> PhysicalPlan { + if !plan.is_sharded_source() { + return plan; + } + let as_aggregate = matches!( + &plan, + PhysicalPlan::Query(QueryOp::Aggregate { .. }) + | PhysicalPlan::Query(QueryOp::PartialAggregate { .. }) + ); + PhysicalPlan::Query(QueryOp::Exchange(ExchangeOp { + child: Box::new(plan), + mode: ExchangeMode::Gather { as_aggregate }, + })) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::planner::sql_plan_convert::PlanningPurpose; + use nodedb_sql::types::SqlValue; + + fn ctx() -> ConvertContext { + ConvertContext { + purpose: PlanningPurpose::Execute, + retention_registry: None, + array_catalog: None, + credentials: None, + wal: None, + surrogate_assigner: None, + cluster_enabled: false, + bitemporal_retention_registry: None, + max_vector_dim: 0, + force_shuffle_join: false, + shuffle_num_parts: 0, + force_shuffle_agg: false, + shuffle_agg_num_parts: 0, + broadcast_threshold_bytes: 8 * 1024 * 1024, + shuffle_agg_threshold: 10_000, + database_id: crate::types::DatabaseId::DEFAULT, + tenant_id: crate::types::TenantId::new(0), + } + } + + fn constant(x: i64) -> SqlPlan { + SqlPlan::ConstantResult { + columns: vec!["x".into()], + values: vec![SqlValue::Int(x)], + volatile: false, + } + } + + #[test] + fn union_all_body_lowers_to_one_set_op() { + let body = SqlPlan::Union { + inputs: vec![constant(1), constant(2)], + distinct: false, + }; + let plan = convert_body_to_single_plan(&body, TenantId::new(1), &ctx()) + .expect("union body lowers"); + match plan { + PhysicalPlan::Query(QueryOp::SetOp { inputs, op }) => { + assert_eq!(op, SetOpKind::UnionAll); + assert_eq!(inputs.len(), 2); + for input in &inputs { + assert!(matches!( + input, + PhysicalPlan::Query(QueryOp::ProviderScan { provider: None, .. }) + )); + } + } + other => panic!("expected SetOp, got {other:?}"), + } + } + + #[test] + fn nested_set_ops_stay_one_relation() { + let body = SqlPlan::Intersect { + left: Box::new(SqlPlan::Union { + inputs: vec![constant(1), constant(2)], + distinct: true, + }), + right: Box::new(constant(2)), + all: false, + }; + let plan = convert_body_to_single_plan(&body, TenantId::new(1), &ctx()) + .expect("nested set-op body lowers"); + let PhysicalPlan::Query(QueryOp::SetOp { inputs, op }) = plan else { + panic!("expected SetOp"); + }; + assert_eq!(op, SetOpKind::Intersect); + assert!(matches!( + &inputs[0], + PhysicalPlan::Query(QueryOp::SetOp { + op: SetOpKind::UnionDistinct, + .. + }) + )); + assert!(!PhysicalPlan::Query(QueryOp::SetOp { inputs, op }).is_sharded_source()); + } +} diff --git a/nodedb/src/control/planner/sql_plan_convert/mod.rs b/nodedb/src/control/planner/sql_plan_convert/mod.rs index ba341a353..c08ece90b 100644 --- a/nodedb/src/control/planner/sql_plan_convert/mod.rs +++ b/nodedb/src/control/planner/sql_plan_convert/mod.rs @@ -4,6 +4,7 @@ pub mod aggregate; pub mod array_alter_convert; pub mod array_convert; pub mod array_fn_convert; +pub mod body; pub mod cache_verdict; pub mod convert; pub mod dml; diff --git a/nodedb/src/control/planner/sql_plan_convert/output_schema/build.rs b/nodedb/src/control/planner/sql_plan_convert/output_schema/build.rs index 06733ce39..1d9a5dd14 100644 --- a/nodedb/src/control/planner/sql_plan_convert/output_schema/build.rs +++ b/nodedb/src/control/planner/sql_plan_convert/output_schema/build.rs @@ -173,21 +173,29 @@ pub fn build_output_schema( let types = super::join_types::join_column_types(left, right, catalog, database_id); schema_from_projection(projection, &types, &[]) } - SqlPlan::ConstantResult { columns, .. } => { + SqlPlan::ConstantResult { + columns, values, .. + } => { // The row payload keys each cell by the unique per-column key // (`cell_keys`), not the raw display name: two constant columns may // share a name (`SELECT nextval('s'), nextval('s')`), and a single - // JSON object would collapse them. `display_name` keeps the + // object would collapse them. `display_name` keeps the // client-facing name; `lookup_key` is the cell key. + // + // The type mirrors the cell `convert_constant_result` encodes: + // `Int`/`Float`/`Bool` keep their typed cell, every other variant + // (`String`/`Null`/`Decimal`/`Bytes`/`Array`/`Timestamp`/ + // `Timestamptz`) is encoded as text. let lookup_keys = crate::control::server::response_shape::project::cell_keys(columns); OutputSchema { columns: columns .iter() .zip(lookup_keys) - .map(|(c, lookup_key)| OutputColumn { + .enumerate() + .map(|(index, (c, lookup_key))| OutputColumn { display_name: c.clone(), lookup_key, - ty: DdlColType::Text, + ty: constant_cell_type(values.get(index)), }) .collect(), is_star: false, @@ -379,6 +387,27 @@ pub fn build_output_schema( } } +/// Wire type of one constant cell. A column with no value (a plan built +/// without values) is `Text`. +fn constant_cell_type(value: Option<&nodedb_sql::types_expr::SqlValue>) -> DdlColType { + use nodedb_sql::types_expr::SqlValue; + match value { + Some(SqlValue::Int(_)) => DdlColType::Int8, + Some(SqlValue::Float(_)) => DdlColType::Float8, + Some(SqlValue::Bool(_)) => DdlColType::Bool, + Some( + SqlValue::String(_) + | SqlValue::Null + | SqlValue::Decimal(_) + | SqlValue::Bytes(_) + | SqlValue::Array(_) + | SqlValue::Timestamp(_) + | SqlValue::Timestamptz(_), + ) + | None => DdlColType::Text, + } +} + #[cfg(test)] mod tests { use super::*; @@ -402,18 +431,24 @@ mod tests { } #[test] - fn constant_result_columns_map_to_text_output_columns() { + fn constant_result_columns_are_typed_from_their_values() { + use nodedb_sql::types_expr::SqlValue; let plans = vec![SqlPlan::ConstantResult { - columns: vec!["a".to_string(), "b".to_string()], - values: vec![], + columns: vec!["a".to_string(), "b".to_string(), "c".to_string()], + values: vec![SqlValue::Int(1), SqlValue::String("x".into())], volatile: false, }]; let schema = build_output_schema(&plans, &NoCatalog, nodedb_types::DatabaseId::DEFAULT, None); - assert_eq!(schema.columns.len(), 2); + assert_eq!(schema.columns.len(), 3); assert_eq!(schema.columns[0].display_name, "a"); assert_eq!(schema.columns[0].lookup_key, "a"); + assert_eq!(schema.columns[0].ty, DdlColType::Int8); assert_eq!(schema.columns[1].display_name, "b"); + assert_eq!(schema.columns[1].ty, DdlColType::Text); + // A column without a value keeps its slot and types as text. + assert_eq!(schema.columns[2].display_name, "c"); + assert_eq!(schema.columns[2].ty, DdlColType::Text); assert!(!schema.is_star); } diff --git a/nodedb/src/control/planner/sql_plan_convert/set_ops.rs b/nodedb/src/control/planner/sql_plan_convert/set_ops.rs index 383c9b78f..d8beeea04 100644 --- a/nodedb/src/control/planner/sql_plan_convert/set_ops.rs +++ b/nodedb/src/control/planner/sql_plan_convert/set_ops.rs @@ -8,10 +8,12 @@ use crate::bridge::envelope::PhysicalPlan; use crate::types::{TenantId, VShardId}; use nodedb_physical::physical_plan::*; +use super::body::convert_body_to_single_plan; use super::convert::{ConvertContext, convert_one}; use super::expr::inline_cte; -use super::value::sql_value_to_string; +use super::value::{sql_value_to_nodedb_value, sql_value_to_string}; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; +use nodedb_types::Value; pub(super) fn convert_constant_result( columns: &[String], @@ -19,25 +21,38 @@ pub(super) fn convert_constant_result( tenant_id: TenantId, ctx: &ConvertContext, ) -> crate::Result> { - // A constant row is one JSON object, which cannot hold two cells under one + // A constant row is one object, which cannot hold two cells under one // key. `SELECT nextval('s'), nextval('s')` legally repeats an output name; // keying both cells by the name would collapse them to the last value. Use // the same unique per-column keys every response encoder derives, so each // column keeps its own cell. let cell_keys = crate::control::server::response_shape::project::cell_keys(columns); - let mut obj = serde_json::Map::new(); - for ((_col, val), key) in columns.iter().zip(values.iter()).zip(cell_keys.iter()) { - let json_val = match val { - SqlValue::Null => serde_json::Value::Null, - other => serde_json::Value::String(sql_value_to_string(other)), + let mut obj = std::collections::HashMap::with_capacity(columns.len()); + for ((_col, val), key) in columns.iter().zip(values.iter()).zip(cell_keys) { + let cell = match val { + SqlValue::Int(_) + | SqlValue::Float(_) + | SqlValue::Bool(_) + | SqlValue::Null + | SqlValue::String(_) => sql_value_to_nodedb_value(val), + // The shaper has no typed renderer that reproduces PostgreSQL's + // text form for these — `\x..` for bytes, `{1,2}` for arrays, the + // ISO string for timestamps — from a typed value, so they keep + // that text form under a `Text` column instead. + SqlValue::Decimal(_) + | SqlValue::Bytes(_) + | SqlValue::Array(_) + | SqlValue::Timestamp(_) + | SqlValue::Timestamptz(_) => Value::String(sql_value_to_string(val)), }; - obj.insert(key.clone(), json_val); + obj.insert(key, cell); } - let arr = serde_json::Value::Array(vec![serde_json::Value::Object(obj)]); - let payload = nodedb_types::json_to_msgpack(&arr).map_err(|e| crate::Error::Serialization { - format: "msgpack".into(), - detail: format!("constant result: {e}"), - })?; + let arr = Value::Array(vec![Value::Object(obj)]); + let payload = + nodedb_types::value_to_msgpack(&arr).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("constant result: {e}"), + })?; Ok(vec![PhysicalTask { tenant_id, vshard_id: VShardId::from_collection_in_database(ctx.database_id, ""), @@ -243,9 +258,11 @@ pub(super) fn convert_cte( /// whose leaf could not absorb the outer constraints — into a coordinator- /// resolved `QueryOp::PostProcess`. /// -/// The body is converted to a single physical plan and, when it is a sharded -/// source, wrapped in `Exchange{Gather}` so the sort/distinct/offset/limit tail -/// runs exactly once over the full union at resolve time. +/// The body lowers to ONE physical relation through +/// `convert_body_to_single_plan`: a set-operation body becomes a +/// coordinator-resolved `SetOp`, and a sharded body is wrapped in +/// `Exchange{Gather}` so the sort/distinct/offset/limit tail runs exactly +/// once over the full union at resolve time. pub(super) fn convert_subquery( args: nodedb_sql::SubqueryVisitArgs<'_>, tenant_id: TenantId, @@ -262,20 +279,8 @@ pub(super) fn convert_subquery( limit, } = args; - // Materialize the body as a single physical plan. A subquery/derived-table - // body is one relation; a body that lowers to multiple tasks (e.g. a set - // operation) has no single row stream to post-process here. - let mut body_tasks = convert_one(input, tenant_id, ctx)?; - if body_tasks.len() != 1 { - return Err(crate::Error::PlanError { - detail: format!( - "ORDER BY / OFFSET / DISTINCT over a subquery whose body lowers to {} physical \ - tasks is not supported; the body must produce a single relation", - body_tasks.len() - ), - }); - } - let mut child = body_tasks.pop().expect("checked len == 1").plan; + // The body is ONE relation, already gathered when sharded. + let child = convert_body_to_single_plan(input, tenant_id, ctx)?; // A join / lateral body emits ONE merged document per output row whose // columns keep their table prefix (`a.attnum`), which is why the response @@ -283,33 +288,9 @@ pub(super) fn convert_subquery( // computed columns, and window specs must address the same shape — an // unqualified key resolves to NULL on every merged row, and a sort where // every key is NULL is a no-op that silently answers an ordered query in - // the body's own order. - let merged_doc_body = matches!( - child, - PhysicalPlan::Query( - QueryOp::HashJoin { .. } - | QueryOp::NestedLoopJoin { .. } - | QueryOp::SortMergeJoin { .. } - | QueryOp::LateralTopK { .. } - | QueryOp::LateralLoop { .. } - ) - ); - - // A sharded body must be gathered before the relational tail runs, so the - // sort/distinct/offset/limit observe the FULL union exactly once. - // PostProcess is itself coordinator-local (`is_sharded_source() == false`), - // so the top-level `convert()` wrap loop will not gather the child for us. - if child.is_sharded_source() { - let as_aggregate = matches!( - &child, - PhysicalPlan::Query(QueryOp::Aggregate { .. }) - | PhysicalPlan::Query(QueryOp::PartialAggregate { .. }) - ); - child = PhysicalPlan::Query(QueryOp::Exchange(ExchangeOp { - child: Box::new(child), - mode: ExchangeMode::Gather { as_aggregate }, - })); - } + // the body's own order. The body may sit under the `Exchange{Gather}` + // wrapper, so the detection looks through it. + let merged_doc_body = is_merged_doc_body(&child); Ok(vec![PhysicalTask { tenant_id, @@ -340,6 +321,50 @@ pub(super) fn convert_subquery( }]) } +/// Whether a body plan is a join / lateral whose rows keep their table +/// prefix on every column, looking through the converter's +/// `Exchange{Gather}` wrapper. +fn is_merged_doc_body(plan: &PhysicalPlan) -> bool { + match plan { + PhysicalPlan::Query(QueryOp::Exchange(ExchangeOp { child, .. })) => { + is_merged_doc_body(child) + } + PhysicalPlan::Query( + QueryOp::HashJoin { .. } + | QueryOp::NestedLoopJoin { .. } + | QueryOp::SortMergeJoin { .. } + | QueryOp::LateralTopK { .. } + | QueryOp::LateralLoop { .. }, + ) => true, + PhysicalPlan::Query( + QueryOp::ProviderScan { .. } + | QueryOp::PostProcess { .. } + | QueryOp::SetOp { .. } + | QueryOp::Aggregate { .. } + | QueryOp::PartialAggregate { .. } + | QueryOp::PartialAggregateState { .. } + | QueryOp::ShuffleJoinConsume { .. } + | QueryOp::ShuffleAggregateConsume { .. } + | QueryOp::FacetCounts { .. } + | QueryOp::RecursiveScan { .. } + | QueryOp::RecursiveValue { .. }, + ) + | PhysicalPlan::Document(_) + | PhysicalPlan::Vector(_) + | PhysicalPlan::Graph(_) + | PhysicalPlan::Text(_) + | PhysicalPlan::Columnar(_) + | PhysicalPlan::Timeseries(_) + | PhysicalPlan::Spatial(_) + | PhysicalPlan::Kv(_) + | PhysicalPlan::Crdt(_) + | PhysicalPlan::Meta(_) + | PhysicalPlan::Array(_) + | PhysicalPlan::ClusterArray(_) + | PhysicalPlan::ClusterEvent(_) => false, + } +} + /// Lower outer projection items to the row keys the relational tail matches. /// /// A bare column keeps its unqualified name (the flattened row's column key); a diff --git a/nodedb/src/control/security/identity/plan_permission.rs b/nodedb/src/control/security/identity/plan_permission.rs index ec68a472c..7461d7132 100644 --- a/nodedb/src/control/security/identity/plan_permission.rs +++ b/nodedb/src/control/security/identity/plan_permission.rs @@ -92,6 +92,14 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm // PostProcess only reshapes child rows; permission is the child's — recurse, don't assume Read. PhysicalPlan::Query(QueryOp::PostProcess { input, .. }) => required_permission(input), + // SetOp merges N child relations; it requires the strictest permission + // any branch requires. An empty input list is unreachable from the + // converter and maps to Read, the weakest tier. + PhysicalPlan::Query(QueryOp::SetOp { inputs, .. }) => inputs + .iter() + .map(required_permission) + .fold(Permission::Read, strictest), + PhysicalPlan::Text( TextOp::Search { .. } | TextOp::BM25ScoreScan { .. } @@ -363,3 +371,30 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm } } } + +/// The stricter of two permissions under the tier order used to fold a +/// multi-input node. Exhaustive so a new `Permission` variant forces a +/// placement here. +fn strictest(a: Permission, b: Permission) -> Permission { + if strictness_rank(b) > strictness_rank(a) { + b + } else { + a + } +} + +/// Tier order from weakest to strictest. Read-class tiers come first, then +/// write, then schema, then cluster-wide control. +fn strictness_rank(permission: Permission) -> u8 { + match permission { + Permission::Read => 0, + Permission::Monitor => 1, + Permission::Execute => 2, + Permission::Write => 3, + Permission::Create => 4, + Permission::Drop => 5, + Permission::Alter => 6, + Permission::Backup => 7, + Permission::Admin => 8, + } +} diff --git a/nodedb/src/control/server/exchange/resolve/exchange/aggregate_input_arm.rs b/nodedb/src/control/server/exchange/resolve/exchange/aggregate_input_arm.rs index 426b4bd25..63667c288 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/aggregate_input_arm.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/aggregate_input_arm.rs @@ -14,7 +14,7 @@ use crate::control::state::SharedState; use super::dispatch::ResolveCtx; use super::entry::Resolved; -use super::post_process_arm::{ChildRows, materialize_child_rows}; +use super::post_process_arm::{ChildRows, materialize_child_rows, provider_scan_of_rows}; /// Fields of a `QueryOp::Aggregate { input: Some(_) }` plan node, carried /// through resolution as one value. @@ -88,18 +88,5 @@ pub(super) async fn resolve_aggregate_input( ChildRows::Rows(rows) => rows, ChildRows::Passthrough(resolved) => return Ok(resolved), }; - Ok(rebuild(Box::new(PhysicalPlan::Query( - QueryOp::ProviderScan { - provider: None, - rows, - filters: Vec::new(), - projection: Vec::new(), - computed_columns: Vec::new(), - window_functions: Vec::new(), - sort_keys: Vec::new(), - limit: None, - offset: 0, - distinct: false, - }, - )))) + Ok(rebuild(Box::new(provider_scan_of_rows(rows)))) } diff --git a/nodedb/src/control/server/exchange/resolve/exchange/dispatch.rs b/nodedb/src/control/server/exchange/resolve/exchange/dispatch.rs index 125df5210..29212c4a4 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/dispatch.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/dispatch.rs @@ -13,7 +13,9 @@ use super::aggregate_input_arm::AggregateFields; use super::entry::Resolved; use super::hash_join_arm::HashJoinFields; use super::post_process_arm::PostProcessFields; -use super::{aggregate_input_arm, gather_arm, hash_join_arm, post_process_arm, shuffle_arm}; +use super::{ + aggregate_input_arm, gather_arm, hash_join_arm, post_process_arm, set_op_arm, shuffle_arm, +}; /// Request-scoped identifiers threaded through every arm resolver, bundled /// to keep each resolver's argument list within the clippy default arity. @@ -35,6 +37,8 @@ pub(super) struct ResolveCtx { /// a typed error. /// - `Aggregate{input: Some}` → materialize the child on the coordinator and /// embed it as `ProviderScan{None, rows}`, return `Resolved::Plan`. +/// - `SetOp{inputs, op}` → materialize every branch, merge with `op`, and +/// embed as `ProviderScan{None, rows}`, return `Resolved::Plan`. /// - Anything else → `Resolved::Plan` unchanged. /// /// `captures` accumulates one [`DistributedReadCapture`] per base collection an @@ -224,6 +228,13 @@ pub(super) async fn resolve_exchange( .await } + // SetOp: materialize every branch on the coordinator, merge with the + // set operation, and lower to a `ProviderScan` of the merged rows. + // The node is coordinator-local, so the root Gather arm never sees it. + PhysicalPlan::Query(QueryOp::SetOp { inputs, op }) => { + set_op_arm::resolve_set_op(state, ctx, captures, inputs, op).await + } + // All other plan variants: pass through unchanged. other => Ok(Resolved::Plan(Box::new(other))), } diff --git a/nodedb/src/control/server/exchange/resolve/exchange/mod.rs b/nodedb/src/control/server/exchange/resolve/exchange/mod.rs index dff2214a4..7356e5fb5 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/mod.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/mod.rs @@ -23,6 +23,8 @@ mod entry; mod gather_arm; mod hash_join_arm; mod post_process_arm; +mod set_op_arm; mod shuffle_arm; pub use entry::{Resolved, resolve_and_materialize, resolve_exchange_in_plan}; +pub(crate) use post_process_arm::provider_scan_of_rows; diff --git a/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs b/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs index c5e5de523..8230cffec 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs @@ -91,6 +91,24 @@ fn hit_collection_name(plan: &PhysicalPlan) -> Option { } } +/// A `ProviderScan` carrying final `rows` and an empty relational tail: +/// no filter, projection, computed column, window, sort, limit, offset, or +/// distinct. The shape every coordinator-materialized child is embedded as. +pub(crate) fn provider_scan_of_rows(rows: Vec) -> PhysicalPlan { + PhysicalPlan::Query(QueryOp::ProviderScan { + provider: None, + rows, + filters: Vec::new(), + projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), + sort_keys: Vec::new(), + limit: None, + offset: 0, + distinct: false, + }) +} + /// Rows of a materialized child, or a resolution the caller returns as-is. pub(super) enum ChildRows { /// The child's rows, flattened to the bare relational row shape a diff --git a/nodedb/src/control/server/exchange/resolve/exchange/set_op_arm.rs b/nodedb/src/control/server/exchange/resolve/exchange/set_op_arm.rs new file mode 100644 index 000000000..814038084 --- /dev/null +++ b/nodedb/src/control/server/exchange/resolve/exchange/set_op_arm.rs @@ -0,0 +1,120 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `SetOp` exchange resolution: materialize every branch on the coordinator, +//! merge the rows with the set operation, and lower to a `ProviderScan`. + +use nodedb_physical::physical_plan::{PhysicalPlan, SetOpKind}; + +use crate::control::server::exchange::resolve::capture::DistributedReadCapture; +use crate::control::server::payload_merge::merge_msgpack_arrays; +use crate::control::server::set_op_merge::{ + SetMergeMode, dedup_union_payloads, merge_set_op_payloads, +}; +use crate::control::state::SharedState; + +use super::dispatch::ResolveCtx; +use super::entry::Resolved; +use super::post_process_arm::{ChildRows, materialize_child_rows, provider_scan_of_rows}; + +/// Resolve a `QueryOp::SetOp` node. +/// +/// Every branch is an independent body, so all of them materialize +/// concurrently, each into its own read-capture list. The captures are then +/// appended to `captures` in branch order, so the in-transaction read-set +/// sees every branch's base collection exactly once. A branch that resolves +/// to a root `Gathered` / `Stream` result is returned as-is, the way the +/// post-processor returns it. +/// +/// The merged rows are embedded as a `ProviderScan{provider: None}` with an +/// empty relational tail; the enclosing post-processor or input-sourced +/// aggregate supplies its own tail over these rows. +pub(super) async fn resolve_set_op( + state: &SharedState, + ctx: ResolveCtx, + captures: &mut Vec, + inputs: Vec, + op: SetOpKind, +) -> crate::Result { + let branches = inputs.into_iter().map(|input| async move { + let mut branch_captures = Vec::new(); + let rows = materialize_child_rows(state, ctx, &mut branch_captures, input).await?; + Ok::<_, crate::Error>((rows, branch_captures)) + }); + let materialized = futures::future::try_join_all(branches).await?; + + let mut payloads = Vec::with_capacity(materialized.len()); + for (rows, branch_captures) in materialized { + captures.extend(branch_captures); + match rows { + ChildRows::Rows(rows) => payloads.push(rows), + ChildRows::Passthrough(resolved) => return Ok(resolved), + } + } + + let merged = merge_set_op_rows(&payloads, op); + Ok(Resolved::Plan(Box::new(provider_scan_of_rows(merged)))) +} + +/// Merge materialized branch payloads with `op`. Each payload is one msgpack +/// array of flat row maps. The mapping from kind to merge mirrors the pgwire +/// per-task set-op path: `INTERSECT [ALL]` and `EXCEPT [ALL]` share one +/// value-keyed merge each, `UNION` dedups on raw bytes, and `UNION ALL` +/// concatenates. +fn merge_set_op_rows(payloads: &[Vec], op: SetOpKind) -> Vec { + match op { + SetOpKind::UnionAll => merge_msgpack_arrays(payloads), + SetOpKind::UnionDistinct => dedup_union_payloads(payloads), + SetOpKind::Intersect | SetOpKind::IntersectAll => { + merge_set_op_payloads(payloads, SetMergeMode::Intersect) + } + SetOpKind::Except | SetOpKind::ExceptAll => { + merge_set_op_payloads(payloads, SetMergeMode::Except) + } + } +} + +#[cfg(test)] +mod tests { + use super::merge_set_op_rows; + use nodedb_physical::physical_plan::SetOpKind; + + fn encode_array(rows: &[serde_json::Value]) -> Vec { + nodedb_types::json_to_msgpack(&serde_json::Value::Array(rows.to_vec())).unwrap() + } + + fn decode(payload: &[u8]) -> String { + crate::data::executor::response_codec::decode_payload_to_json(payload) + } + + #[test] + fn union_all_keeps_every_row_in_branch_order() { + let left = encode_array(&[serde_json::json!({"x": 1}), serde_json::json!({"x": 2})]); + let right = encode_array(&[serde_json::json!({"x": 2})]); + let merged = merge_set_op_rows(&[left, right], SetOpKind::UnionAll); + assert_eq!(decode(&merged), r#"[{"x":1},{"x":2},{"x":2}]"#); + } + + #[test] + fn union_distinct_drops_duplicates() { + let left = encode_array(&[serde_json::json!({"x": 1}), serde_json::json!({"x": 2})]); + let right = encode_array(&[serde_json::json!({"x": 2})]); + let merged = merge_set_op_rows(&[left, right], SetOpKind::UnionDistinct); + assert_eq!(decode(&merged), r#"[{"x":1},{"x":2}]"#); + } + + #[test] + fn intersect_keeps_rows_present_in_every_branch() { + let left = encode_array(&[serde_json::json!({"x": 1}), serde_json::json!({"x": 2})]); + let right = encode_array(&[serde_json::json!({"x": 2}), serde_json::json!({"x": 3})]); + let merged = merge_set_op_rows(&[left, right], SetOpKind::Intersect); + assert_eq!(decode(&merged), r#"[{"x":2}]"#); + } + + #[test] + fn except_drops_rows_present_in_later_branches() { + let left = encode_array(&[serde_json::json!({"x": 1}), serde_json::json!({"x": 2})]); + let right = encode_array(&[serde_json::json!({"x": 2})]); + let merged = merge_set_op_rows(&[left, right], SetOpKind::Except); + assert_eq!(decode(&merged), r#"[{"x":1}]"#); + } +} diff --git a/nodedb/src/control/server/exchange/resolve/join_input.rs b/nodedb/src/control/server/exchange/resolve/join_input.rs index 2d3e1ab32..a0de9f7e9 100644 --- a/nodedb/src/control/server/exchange/resolve/join_input.rs +++ b/nodedb/src/control/server/exchange/resolve/join_input.rs @@ -15,6 +15,7 @@ use crate::control::server::exchange::gather::{ }; use super::capture::DistributedReadCapture; +use super::exchange::provider_scan_of_rows; /// Resolve a `HashJoin` input slot. /// @@ -55,18 +56,8 @@ pub(super) async fn resolve_join_input( // Response as a msgpack array — so the two shapes match. let outcome = gather_all_cores(state, tenant_id, database_id, *child, trace_id, txn_id).await?; - let provider_scan = PhysicalPlan::Query(QueryOp::ProviderScan { - provider: None, - rows: flatten_to_relational_rows(&outcome.merged_array), - filters: Vec::new(), - projection: Vec::new(), - computed_columns: Vec::new(), - window_functions: Vec::new(), - sort_keys: Vec::new(), - limit: None, - offset: 0, - distinct: false, - }); + let provider_scan = + provider_scan_of_rows(flatten_to_relational_rows(&outcome.merged_array)); Ok(Some(Box::new(provider_scan))) } @@ -131,18 +122,7 @@ pub(super) async fn resolve_join_input( } else { outcome.merged_array }; - let provider_scan = PhysicalPlan::Query(QueryOp::ProviderScan { - provider: None, - rows: flatten_to_relational_rows(&merged), - filters: Vec::new(), - projection: Vec::new(), - computed_columns: Vec::new(), - window_functions: Vec::new(), - sort_keys: Vec::new(), - limit: None, - offset: 0, - distinct: false, - }); + let provider_scan = provider_scan_of_rows(flatten_to_relational_rows(&merged)); Ok(Some(Box::new(provider_scan))) } @@ -235,16 +215,7 @@ pub(super) async fn gather_join_build_side( }); } - Ok(Some(Box::new(PhysicalPlan::Query(QueryOp::ProviderScan { - provider: None, - rows: flatten_to_relational_rows(&outcome.merged_array), - filters: Vec::new(), - projection: Vec::new(), - computed_columns: Vec::new(), - window_functions: Vec::new(), - sort_keys: Vec::new(), - limit: None, - offset: 0, - distinct: false, - })))) + Ok(Some(Box::new(provider_scan_of_rows( + flatten_to_relational_rows(&outcome.merged_array), + )))) } diff --git a/nodedb/src/control/server/exchange/resolve/materialize.rs b/nodedb/src/control/server/exchange/resolve/materialize.rs index a4b53518d..009d8636f 100644 --- a/nodedb/src/control/server/exchange/resolve/materialize.rs +++ b/nodedb/src/control/server/exchange/resolve/materialize.rs @@ -248,6 +248,16 @@ pub(super) async fn materialize_providers( })) } + // SetOp: recurse into every branch so nested catalog providers are + // filled before the set-op resolver materializes the branches. + PhysicalPlan::Query(QueryOp::SetOp { inputs, op }) => { + let mut filled = Vec::with_capacity(inputs.len()); + for input in inputs { + filled.push(Box::pin(materialize_providers(state, identity, input)).await?); + } + Ok(PhysicalPlan::Query(QueryOp::SetOp { inputs: filled, op })) + } + // All other variants: no catalog providers can be nested here — // pass through unchanged. other => Ok(other), diff --git a/nodedb/src/control/server/mod.rs b/nodedb/src/control/server/mod.rs index 06a16523d..b2653b9db 100644 --- a/nodedb/src/control/server/mod.rs +++ b/nodedb/src/control/server/mod.rs @@ -21,6 +21,7 @@ pub mod response_shape; pub mod response_translate; pub mod result_stream; pub mod session_auth; +pub mod set_op_merge; pub mod shared; pub mod shuffle; pub mod surrogate_exchange; diff --git a/nodedb/src/control/server/pgwire/handler/routing/set_ops.rs b/nodedb/src/control/server/pgwire/handler/routing/set_ops.rs index 2baaaa9a3..43d15f976 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/set_ops.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/set_ops.rs @@ -1,8 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Set operation payload merging: UNION DISTINCT, INTERSECT, EXCEPT. -//! -//! Operates on raw msgpack payloads — no decode/re-encode round-trip. +//! Set operation payload merging for pgwire: UNION DISTINCT, INTERSECT, +//! EXCEPT over collected per-task payloads, shaped into a pgwire response. use pgwire::api::results::{FieldFormat, Response}; use pgwire::error::PgWireResult; @@ -12,6 +11,9 @@ use nodedb_physical::physical_task::PostSetOp; use crate::control::server::response_shape::compose::{self, ShapeOutcome}; use crate::control::server::response_shape::redaction::RedactionCtx; use crate::control::server::response_shape::schema::OutputSchema; +use crate::control::server::set_op_merge::{ + SetMergeMode, dedup_union_payloads, merge_set_op_payloads, +}; use super::super::super::types::sqlstate_error; use super::super::plan::{PlanKind, multirow_payload_to_response}; @@ -49,323 +51,3 @@ pub(super) fn apply_set_ops( }, ) } - -/// Merge multiple Data Plane response payloads and deduplicate rows (UNION DISTINCT). -/// -/// Each payload is a msgpack-encoded array of rows. Deduplication is performed -/// at the binary level: each row's raw msgpack bytes serve as the canonical key, -/// eliminating the decode → JSON string → re-encode round-trip. -/// -/// Output: a single msgpack array containing all unique rows in encounter order. -fn dedup_union_payloads(payloads: &[Vec]) -> Vec { - use nodedb_query::msgpack_scan; - - let mut seen: std::collections::HashSet> = std::collections::HashSet::new(); - let mut unique_row_bytes: Vec> = Vec::new(); - - for payload in payloads { - if payload.is_empty() { - continue; - } - - let bytes = payload.as_slice(); - let first = bytes[0]; - - let (count, hdr_len) = if (0x90..=0x9f).contains(&first) { - ((first & 0x0f) as usize, 1) - } else if first == 0xdc && bytes.len() >= 3 { - (u16::from_be_bytes([bytes[1], bytes[2]]) as usize, 3) - } else if first == 0xdd && bytes.len() >= 5 { - ( - u32::from_be_bytes([bytes[1], bytes[2], bytes[3], bytes[4]]) as usize, - 5, - ) - } else { - tracing::warn!( - payload_len = bytes.len(), - "dedup_union_payloads: payload is not a msgpack array; treating as single row" - ); - let key = bytes.to_vec(); - if seen.insert(key.clone()) { - unique_row_bytes.push(key); - } - continue; - }; - - let mut pos = hdr_len; - for _ in 0..count { - if pos >= bytes.len() { - break; - } - let elem_start = pos; - match msgpack_scan::skip_value(bytes, pos) { - Some(next_pos) => { - let row_bytes = bytes[elem_start..next_pos].to_vec(); - if seen.insert(row_bytes.clone()) { - unique_row_bytes.push(row_bytes); - } - pos = next_pos; - } - None => { - tracing::warn!( - pos, - payload_len = bytes.len(), - "dedup_union_payloads: could not skip msgpack element; stopping early" - ); - break; - } - } - } - } - - let row_count = unique_row_bytes.len(); - let total_data: usize = unique_row_bytes.iter().map(|r| r.len()).sum(); - let mut out = Vec::with_capacity(total_data + 5); - write_array_header(&mut out, row_count); - for row in unique_row_bytes { - out.extend_from_slice(&row); - } - out -} - -enum SetMergeMode { - Intersect, - Except, -} - -/// Merge payloads for INTERSECT or EXCEPT set operations. -/// -/// For INTERSECT: keep rows that appear in ALL payloads. -/// For EXCEPT: keep rows from first payload that don't appear in any subsequent payload. -fn merge_set_op_payloads(payloads: &[Vec], mode: SetMergeMode) -> Vec { - use nodedb_query::msgpack_scan; - - if payloads.is_empty() { - return vec![0x90]; - } - - fn extract_rows(payload: &[u8]) -> Vec> { - if payload.is_empty() { - return Vec::new(); - } - let first = payload[0]; - let (count, hdr_len) = if (0x90..=0x9f).contains(&first) { - ((first & 0x0f) as usize, 1) - } else if first == 0xdc && payload.len() >= 3 { - (u16::from_be_bytes([payload[1], payload[2]]) as usize, 3) - } else if first == 0xdd && payload.len() >= 5 { - ( - u32::from_be_bytes([payload[1], payload[2], payload[3], payload[4]]) as usize, - 5, - ) - } else { - return vec![payload.to_vec()]; - }; - - let mut rows = Vec::with_capacity(count); - let mut pos = hdr_len; - for _ in 0..count { - if pos >= payload.len() { - break; - } - let start = pos; - match msgpack_scan::skip_value(payload, pos) { - Some(next) => { - rows.push(payload[start..next].to_vec()); - pos = next; - } - None => break, - } - } - rows - } - - fn logical_row_bytes(row: &[u8]) -> &[u8] { - msgpack_scan::extract_field(row, 0, "data") - .map(|(start, end)| &row[start..end]) - .unwrap_or(row) - } - - fn write_values_only_key(value: &[u8], out: &mut Vec) -> Option<()> { - if let Some((count, mut pos)) = msgpack_scan::map_header(value, 0) { - write_array_header(out, count); - for _ in 0..count { - pos = msgpack_scan::skip_value(value, pos)?; - let val_start = pos; - pos = msgpack_scan::skip_value(value, pos)?; - write_values_only_key(&value[val_start..pos], out)?; - } - return Some(()); - } - - if let Some((count, mut pos)) = msgpack_scan::array_header(value, 0) { - write_array_header(out, count); - for _ in 0..count { - let elem_start = pos; - pos = msgpack_scan::skip_value(value, pos)?; - write_values_only_key(&value[elem_start..pos], out)?; - } - return Some(()); - } - - out.extend_from_slice(value); - Some(()) - } - - fn extract_value_parts(row: &[u8]) -> Vec> { - let logical = logical_row_bytes(row); - - if let Some((count, mut pos)) = msgpack_scan::map_header(logical, 0) { - let mut parts = Vec::with_capacity(count); - for _ in 0..count { - pos = match msgpack_scan::skip_value(logical, pos) { - Some(next) => next, - None => return vec![logical.to_vec()], - }; - let val_start = pos; - pos = match msgpack_scan::skip_value(logical, pos) { - Some(next) => next, - None => return vec![logical.to_vec()], - }; - let mut normalized = Vec::with_capacity(pos - val_start); - if write_values_only_key(&logical[val_start..pos], &mut normalized).is_none() { - return vec![logical.to_vec()]; - } - parts.push(normalized); - } - return parts; - } - - if let Some((count, mut pos)) = msgpack_scan::array_header(logical, 0) { - let mut parts = Vec::with_capacity(count); - for _ in 0..count { - let elem_start = pos; - pos = match msgpack_scan::skip_value(logical, pos) { - Some(next) => next, - None => return vec![logical.to_vec()], - }; - let mut normalized = Vec::with_capacity(pos - elem_start); - if write_values_only_key(&logical[elem_start..pos], &mut normalized).is_none() { - return vec![logical.to_vec()]; - } - parts.push(normalized); - } - return parts; - } - - vec![logical.to_vec()] - } - - fn extract_values_key(row: &[u8]) -> Vec { - let parts = extract_value_parts(row); - let mut vals = Vec::new(); - write_array_header(&mut vals, parts.len()); - for part in parts { - vals.extend_from_slice(&part); - } - vals - } - - fn rows_match(left: &[u8], right: &[u8]) -> bool { - let left_parts = extract_value_parts(left); - let right_parts = extract_value_parts(right); - let shared_len = left_parts.len().min(right_parts.len()); - - if shared_len == 0 { - return left_parts.is_empty() && right_parts.is_empty(); - } - - left_parts[..shared_len] == right_parts[..shared_len] - && (left_parts.len() == shared_len || right_parts.len() == shared_len) - } - - let first_rows = extract_rows(&payloads[0]); - let mut result_rows: Vec> = match mode { - SetMergeMode::Intersect => { - let other_rows: Vec>> = - payloads[1..].iter().map(|p| extract_rows(p)).collect(); - first_rows - .into_iter() - .filter(|row| { - other_rows - .iter() - .all(|rows| rows.iter().any(|other| rows_match(row, other))) - }) - .map(|row| logical_row_bytes(&row).to_vec()) - .collect() - } - SetMergeMode::Except => { - let other_rows: Vec> = - payloads[1..].iter().flat_map(|p| extract_rows(p)).collect(); - first_rows - .into_iter() - .filter(|row| !other_rows.iter().any(|other| rows_match(row, other))) - .map(|row| logical_row_bytes(&row).to_vec()) - .collect() - } - }; - - let mut seen = std::collections::HashSet::new(); - result_rows.retain(|r| seen.insert(extract_values_key(r))); - - let row_count = result_rows.len(); - let total: usize = result_rows.iter().map(|r| r.len()).sum(); - let mut out = Vec::with_capacity(total + 5); - write_array_header(&mut out, row_count); - for row in result_rows { - out.extend_from_slice(&row); - } - out -} - -fn write_array_header(out: &mut Vec, count: usize) { - if count < 16 { - out.push(0x90 | count as u8); - } else if count <= u16::MAX as usize { - out.push(0xdc); - out.extend_from_slice(&(count as u16).to_be_bytes()); - } else { - out.push(0xdd); - out.extend_from_slice(&(count as u32).to_be_bytes()); - } -} - -#[cfg(test)] -mod tests { - use super::{SetMergeMode, merge_set_op_payloads}; - - fn encode_array(rows: &[serde_json::Value]) -> Vec { - nodedb_types::json_to_msgpack(&serde_json::Value::Array(rows.to_vec())).unwrap() - } - - #[test] - fn intersect_compares_wrapped_rows_by_logical_data_values() { - let left = encode_array(&[ - serde_json::json!({"id":"u1","data":{"id":"u1","name":"Alice"}}), - serde_json::json!({"id":"u2","data":{"id":"u2","name":"Bob"}}), - ]); - let right = encode_array(&[ - serde_json::json!({"id":"doc-1","data":{"user_id":"u1"}}), - serde_json::json!({"id":"doc-2","data":{"user_id":"u3"}}), - ]); - - let merged = merge_set_op_payloads(&[left, right], SetMergeMode::Intersect); - let json = crate::data::executor::response_codec::decode_payload_to_json(&merged); - - assert_eq!(json, r#"[{"id":"u1","name":"Alice"}]"#); - } - - #[test] - fn except_returns_unwrapped_logical_rows() { - let left = encode_array(&[ - serde_json::json!({"id":"u1","data":{"id":"u1"}}), - serde_json::json!({"id":"u2","data":{"id":"u2"}}), - ]); - let right = encode_array(&[serde_json::json!({"id":"doc-1","data":{"user_id":"u1"}})]); - - let merged = merge_set_op_payloads(&[left, right], SetMergeMode::Except); - let json = crate::data::executor::response_codec::decode_payload_to_json(&merged); - - assert_eq!(json, r#"[{"id":"u2"}]"#); - } -} diff --git a/nodedb/src/control/server/response_shape/redaction/query.rs b/nodedb/src/control/server/response_shape/redaction/query.rs index 0b54ecd80..fa5642f4b 100644 --- a/nodedb/src/control/server/response_shape/redaction/query.rs +++ b/nodedb/src/control/server/response_shape/redaction/query.rs @@ -238,6 +238,14 @@ fn collect_sources(plan: &PhysicalPlan, qualifier: &str, out: &mut Vec<(String, collect_sources(input, qualifier, out); return; } + // Every set-operation branch contributes rows under the same + // derived-table qualifier. + QueryOp::SetOp { inputs, .. } => { + for input in inputs { + collect_sources(input, qualifier, out); + } + return; + } QueryOp::Aggregate { collection, input, .. } diff --git a/nodedb/src/control/server/response_shape/types/plan_kind.rs b/nodedb/src/control/server/response_shape/types/plan_kind.rs index 0f68cd00d..1eb811cfb 100644 --- a/nodedb/src/control/server/response_shape/types/plan_kind.rs +++ b/nodedb/src/control/server/response_shape/types/plan_kind.rs @@ -104,6 +104,10 @@ pub fn describe_plan(plan: &PhysicalPlan) -> PlanKind { // PostProcess reshapes a multi-row subquery; its kind is the child's. PhysicalPlan::Query(QueryOp::PostProcess { input, .. }) => describe_plan(input), + // SetOp resolves to a ProviderScan of merged rows; route MultiRow so + // each row streams as its own pgwire row. + PhysicalPlan::Query(QueryOp::SetOp { .. }) => PlanKind::MultiRow, + // An insert with a projection returns real stored rows and must be decoded // and redacted, else it silently leaks unredacted rows like `Merge` did. PhysicalPlan::Kv( diff --git a/nodedb/src/control/server/set_op_merge/intersect_except.rs b/nodedb/src/control/server/set_op_merge/intersect_except.rs new file mode 100644 index 000000000..0dc9c11b0 --- /dev/null +++ b/nodedb/src/control/server/set_op_merge/intersect_except.rs @@ -0,0 +1,102 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `INTERSECT` / `EXCEPT` merge over msgpack row payloads, comparing rows +//! by logical column values. + +use crate::control::server::payload_merge::{encode_msgpack_array, extract_msgpack_elements}; + +use super::row_key::{extract_values_key, logical_row_bytes, rows_match}; + +/// Which filter [`merge_set_op_payloads`] applies to the first payload. +pub(crate) enum SetMergeMode { + /// Keep rows present in every payload. + Intersect, + /// Keep rows of the first payload absent from every later payload. + Except, +} + +/// Merge payloads for INTERSECT or EXCEPT. +/// +/// Rows compare by value (see `row_key::rows_match`). The output holds the +/// logical row bytes (the `{id, data}` wrapper is stripped), deduplicated by +/// values key, as one msgpack array. No payloads yields an empty array. +pub(crate) fn merge_set_op_payloads(payloads: &[Vec], mode: SetMergeMode) -> Vec { + if payloads.is_empty() { + return vec![0x90]; + } + + let first_rows = extract_msgpack_elements(&payloads[0]); + let mut result_rows: Vec> = match mode { + SetMergeMode::Intersect => { + let other_rows: Vec>> = payloads[1..] + .iter() + .map(|p| extract_msgpack_elements(p)) + .collect(); + first_rows + .into_iter() + .filter(|row| { + other_rows + .iter() + .all(|rows| rows.iter().any(|other| rows_match(row, other))) + }) + .map(|row| logical_row_bytes(&row).to_vec()) + .collect() + } + SetMergeMode::Except => { + let other_rows: Vec> = payloads[1..] + .iter() + .flat_map(|p| extract_msgpack_elements(p)) + .collect(); + first_rows + .into_iter() + .filter(|row| !other_rows.iter().any(|other| rows_match(row, other))) + .map(|row| logical_row_bytes(&row).to_vec()) + .collect() + } + }; + + let mut seen = std::collections::HashSet::with_capacity(result_rows.len()); + result_rows.retain(|r| seen.insert(extract_values_key(r))); + + encode_msgpack_array(&result_rows) +} + +#[cfg(test)] +mod tests { + use super::{SetMergeMode, merge_set_op_payloads}; + + fn encode_array(rows: &[serde_json::Value]) -> Vec { + nodedb_types::json_to_msgpack(&serde_json::Value::Array(rows.to_vec())).unwrap() + } + + #[test] + fn intersect_compares_wrapped_rows_by_logical_data_values() { + let left = encode_array(&[ + serde_json::json!({"id":"u1","data":{"id":"u1","name":"Alice"}}), + serde_json::json!({"id":"u2","data":{"id":"u2","name":"Bob"}}), + ]); + let right = encode_array(&[ + serde_json::json!({"id":"doc-1","data":{"user_id":"u1"}}), + serde_json::json!({"id":"doc-2","data":{"user_id":"u3"}}), + ]); + + let merged = merge_set_op_payloads(&[left, right], SetMergeMode::Intersect); + let json = crate::data::executor::response_codec::decode_payload_to_json(&merged); + + assert_eq!(json, r#"[{"id":"u1","name":"Alice"}]"#); + } + + #[test] + fn except_returns_unwrapped_logical_rows() { + let left = encode_array(&[ + serde_json::json!({"id":"u1","data":{"id":"u1"}}), + serde_json::json!({"id":"u2","data":{"id":"u2"}}), + ]); + let right = encode_array(&[serde_json::json!({"id":"doc-1","data":{"user_id":"u1"}})]); + + let merged = merge_set_op_payloads(&[left, right], SetMergeMode::Except); + let json = crate::data::executor::response_codec::decode_payload_to_json(&merged); + + assert_eq!(json, r#"[{"id":"u2"}]"#); + } +} diff --git a/nodedb/src/control/server/set_op_merge/mod.rs b/nodedb/src/control/server/set_op_merge/mod.rs new file mode 100644 index 000000000..bb2e7716b --- /dev/null +++ b/nodedb/src/control/server/set_op_merge/mod.rs @@ -0,0 +1,13 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Protocol-neutral set-operation merging over msgpack row payloads: +//! `UNION DISTINCT`, `INTERSECT`, `EXCEPT`. Operates on raw msgpack bytes +//! with no decode/re-encode round-trip. Used by the pgwire per-task set-op +//! path and by the coordinator's `QueryOp::SetOp` resolver. + +mod intersect_except; +mod row_key; +mod union; + +pub(crate) use intersect_except::{SetMergeMode, merge_set_op_payloads}; +pub(crate) use union::dedup_union_payloads; diff --git a/nodedb/src/control/server/set_op_merge/row_key.rs b/nodedb/src/control/server/set_op_merge/row_key.rs new file mode 100644 index 000000000..06c06d37e --- /dev/null +++ b/nodedb/src/control/server/set_op_merge/row_key.rs @@ -0,0 +1,129 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Row-identity helpers for set-operation merging: the logical row inside a +//! `{id, data}` wrapper, and the values-only key that compares two rows by +//! column values regardless of column names. + +use nodedb_query::msgpack_scan; + +/// The logical row: the `data` field of an `{id, data}` storage wrapper, or +/// the row itself when it carries no wrapper. +pub(super) fn logical_row_bytes(row: &[u8]) -> &[u8] { + msgpack_scan::extract_field(row, 0, "data") + .map(|(start, end)| &row[start..end]) + .unwrap_or(row) +} + +/// Write a msgpack array header for `count` elements. +pub(super) fn write_array_header(out: &mut Vec, count: usize) { + if count < 16 { + out.push(0x90 | count as u8); + } else if count <= u16::MAX as usize { + out.push(0xdc); + out.extend_from_slice(&(count as u16).to_be_bytes()); + } else { + out.push(0xdd); + out.extend_from_slice(&(count as u32).to_be_bytes()); + } +} + +/// Append `value` to `out` with every map rewritten as an array of its +/// values, recursively, so two rows with different column names but equal +/// values compare equal. `None` when `value` is malformed msgpack. +fn write_values_only_key(value: &[u8], out: &mut Vec) -> Option<()> { + if let Some((count, mut pos)) = msgpack_scan::map_header(value, 0) { + write_array_header(out, count); + for _ in 0..count { + pos = msgpack_scan::skip_value(value, pos)?; + let val_start = pos; + pos = msgpack_scan::skip_value(value, pos)?; + write_values_only_key(&value[val_start..pos], out)?; + } + return Some(()); + } + + if let Some((count, mut pos)) = msgpack_scan::array_header(value, 0) { + write_array_header(out, count); + for _ in 0..count { + let elem_start = pos; + pos = msgpack_scan::skip_value(value, pos)?; + write_values_only_key(&value[elem_start..pos], out)?; + } + return Some(()); + } + + out.extend_from_slice(value); + Some(()) +} + +/// The logical row's column values, each normalized to a values-only key, +/// in column order. A scalar or malformed row yields itself as one part. +pub(super) fn extract_value_parts(row: &[u8]) -> Vec> { + let logical = logical_row_bytes(row); + + if let Some((count, mut pos)) = msgpack_scan::map_header(logical, 0) { + let mut parts = Vec::with_capacity(count); + for _ in 0..count { + pos = match msgpack_scan::skip_value(logical, pos) { + Some(next) => next, + None => return vec![logical.to_vec()], + }; + let val_start = pos; + pos = match msgpack_scan::skip_value(logical, pos) { + Some(next) => next, + None => return vec![logical.to_vec()], + }; + let mut normalized = Vec::with_capacity(pos - val_start); + if write_values_only_key(&logical[val_start..pos], &mut normalized).is_none() { + return vec![logical.to_vec()]; + } + parts.push(normalized); + } + return parts; + } + + if let Some((count, mut pos)) = msgpack_scan::array_header(logical, 0) { + let mut parts = Vec::with_capacity(count); + for _ in 0..count { + let elem_start = pos; + pos = match msgpack_scan::skip_value(logical, pos) { + Some(next) => next, + None => return vec![logical.to_vec()], + }; + let mut normalized = Vec::with_capacity(pos - elem_start); + if write_values_only_key(&logical[elem_start..pos], &mut normalized).is_none() { + return vec![logical.to_vec()]; + } + parts.push(normalized); + } + return parts; + } + + vec![logical.to_vec()] +} + +/// One key per row: its value parts as a msgpack array. +pub(super) fn extract_values_key(row: &[u8]) -> Vec { + let parts = extract_value_parts(row); + let mut vals = Vec::new(); + write_array_header(&mut vals, parts.len()); + for part in parts { + vals.extend_from_slice(&part); + } + vals +} + +/// Whether two rows match by value: their shared column prefix is equal and +/// one row has no columns beyond that prefix. +pub(super) fn rows_match(left: &[u8], right: &[u8]) -> bool { + let left_parts = extract_value_parts(left); + let right_parts = extract_value_parts(right); + let shared_len = left_parts.len().min(right_parts.len()); + + if shared_len == 0 { + return left_parts.is_empty() && right_parts.is_empty(); + } + + left_parts[..shared_len] == right_parts[..shared_len] + && (left_parts.len() == shared_len || right_parts.len() == shared_len) +} diff --git a/nodedb/src/control/server/set_op_merge/union.rs b/nodedb/src/control/server/set_op_merge/union.rs new file mode 100644 index 000000000..4bcb9da77 --- /dev/null +++ b/nodedb/src/control/server/set_op_merge/union.rs @@ -0,0 +1,55 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `UNION DISTINCT` merge: every row from every payload, deduplicated on +//! its raw msgpack bytes, in encounter order. + +use crate::control::server::payload_merge::{encode_msgpack_array, extract_msgpack_elements}; + +/// Merge multiple row payloads and drop duplicate rows. +/// +/// Each payload is a msgpack array of rows. A row's raw msgpack bytes are +/// its dedup key, so no decode round-trip runs. A payload that is not a +/// msgpack array is treated as one row. The output is one msgpack array of +/// the unique rows in encounter order. +pub(crate) fn dedup_union_payloads(payloads: &[Vec]) -> Vec { + let rows: Vec> = payloads + .iter() + .flat_map(|payload| extract_msgpack_elements(payload)) + .collect(); + let mut seen: std::collections::HashSet> = + std::collections::HashSet::with_capacity(rows.len()); + let mut unique_rows: Vec> = Vec::with_capacity(rows.len()); + + for row in rows { + if seen.insert(row.clone()) { + unique_rows.push(row); + } + } + + encode_msgpack_array(&unique_rows) +} + +#[cfg(test)] +mod tests { + use super::dedup_union_payloads; + + fn encode_array(rows: &[serde_json::Value]) -> Vec { + nodedb_types::json_to_msgpack(&serde_json::Value::Array(rows.to_vec())).unwrap() + } + + #[test] + fn union_distinct_drops_byte_identical_rows_across_payloads() { + let left = encode_array(&[serde_json::json!({"x": 1}), serde_json::json!({"x": 2})]); + let right = encode_array(&[serde_json::json!({"x": 2}), serde_json::json!({"x": 3})]); + + let merged = dedup_union_payloads(&[left, right]); + let json = crate::data::executor::response_codec::decode_payload_to_json(&merged); + + assert_eq!(json, r#"[{"x":1},{"x":2},{"x":3}]"#); + } + + #[test] + fn union_distinct_of_no_payloads_is_empty_array() { + assert_eq!(dedup_union_payloads(&[]), vec![0x90]); + } +} diff --git a/nodedb/src/control/server/shared/authorization/requirements/query.rs b/nodedb/src/control/server/shared/authorization/requirements/query.rs index 54ab53f8b..58f191e8a 100644 --- a/nodedb/src/control/server/shared/authorization/requirements/query.rs +++ b/nodedb/src/control/server/shared/authorization/requirements/query.rs @@ -77,6 +77,11 @@ pub(super) fn collect_query_requirements<'a>( pending.push(input); true } + // Every set-operation branch is a body whose collections are authorized. + PhysicalPlan::Query(QueryOp::SetOp { inputs, .. }) => { + pending.extend(inputs.iter()); + true + } PhysicalPlan::Query(QueryOp::ProviderScan { provider: Some(provider), .. diff --git a/nodedb/src/control/server/shared/clone_read/temporal.rs b/nodedb/src/control/server/shared/clone_read/temporal.rs index 037d55205..40c8990ca 100644 --- a/nodedb/src/control/server/shared/clone_read/temporal.rs +++ b/nodedb/src/control/server/shared/clone_read/temporal.rs @@ -19,6 +19,9 @@ pub(super) fn extract_system_as_of_ms( PhysicalPlan::Query(QueryOp::PostProcess { input, .. }) => { extract_system_as_of_ms(Some(&**input)) } + PhysicalPlan::Query(QueryOp::SetOp { inputs, .. }) => inputs + .iter() + .find_map(|input| extract_system_as_of_ms(Some(input))), // Index-only/overlay engines carry no qualifier; compose with a data-bearing collection. PhysicalPlan::Vector(_) | PhysicalPlan::Graph(_) diff --git a/nodedb/src/control/server/shared/plan_util.rs b/nodedb/src/control/server/shared/plan_util.rs index c663ea901..044bdffbf 100644 --- a/nodedb/src/control/server/shared/plan_util.rs +++ b/nodedb/src/control/server/shared/plan_util.rs @@ -111,6 +111,12 @@ pub(crate) fn extract_collection(plan: &PhysicalPlan) -> Option<&str> { PhysicalPlan::Query(QueryOp::Exchange(op)) => extract_collection(&op.child), // PostProcess: recurse into the materialized child. PhysicalPlan::Query(QueryOp::PostProcess { input, .. }) => extract_collection(input), + // SetOp: the first branch that names a collection, the same way a + // join reports its left side. Callers that need every branch walk + // the inputs themselves. + PhysicalPlan::Query(QueryOp::SetOp { inputs, .. }) => { + inputs.iter().find_map(extract_collection) + } // ProviderScan is a catalog/constant source — no user collection. PhysicalPlan::Query(QueryOp::ProviderScan { .. }) => None, // KV ops carry their own collection (sorted-index-only ops return None). diff --git a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs index fe9387d2c..8c08d289d 100644 --- a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs +++ b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs @@ -284,7 +284,8 @@ pub fn plan_requires_txn_buffering(plan: &PhysicalPlan) -> bool { | QueryOp::RecursiveValue { .. } | QueryOp::LateralTopK { .. } | QueryOp::LateralLoop { .. } - | QueryOp::PostProcess { .. }, + | QueryOp::PostProcess { .. } + | QueryOp::SetOp { .. }, ) => false, // ---- Meta: control / maintenance ops — internal orchestration, never a client `task.plan`. diff --git a/nodedb/src/data/executor/dispatch/query.rs b/nodedb/src/data/executor/dispatch/query.rs index ac9d9115d..46bcb76ae 100644 --- a/nodedb/src/data/executor/dispatch/query.rs +++ b/nodedb/src/data/executor/dispatch/query.rs @@ -67,6 +67,15 @@ impl CoreLoop { }, ), + QueryOp::SetOp { .. } => self.response_error( + task, + crate::bridge::envelope::ErrorCode::Internal { + detail: "SetOp must be resolved by the coordinator (materialized and \ + merged into a ProviderScan) before dispatch" + .to_string(), + }, + ), + QueryOp::ProviderScan { rows, filters, diff --git a/nodedb/tests/wire/cases/pgwire_extended_query.rs b/nodedb/tests/wire/cases/pgwire_extended_query.rs index 612ee388c..1c309641e 100644 --- a/nodedb/tests/wire/cases/pgwire_extended_query.rs +++ b/nodedb/tests/wire/cases/pgwire_extended_query.rs @@ -105,10 +105,10 @@ async fn extended_query_constant_and_param_projection() { rows[0].len() ); - // x may decode as any integer-compatible type; compare via text. - let x_text: String = rows[0].get::<_, String>("x"); + // A constant integer is typed as int8 in the row description. + let x: i64 = rows[0].get("x"); let y: &str = rows[0].get("y"); - assert_eq!(x_text, "1"); + assert_eq!(x, 1); assert_eq!(y, "hi"); } @@ -136,9 +136,9 @@ async fn extended_query_pure_constant_projection() { rows[0].len() ); - let x_text: String = rows[0].get::<_, String>("x"); + let x: i64 = rows[0].get("x"); let y: &str = rows[0].get("y"); - assert_eq!(x_text, "1"); + assert_eq!(x, 1); assert_eq!(y, "hi"); } diff --git a/nodedb/tests/wire/cases/pgwire_extended_query_engines2.rs b/nodedb/tests/wire/cases/pgwire_extended_query_engines2.rs index 38537449e..3f03ae3c7 100644 --- a/nodedb/tests/wire/cases/pgwire_extended_query_engines2.rs +++ b/nodedb/tests/wire/cases/pgwire_extended_query_engines2.rs @@ -228,8 +228,8 @@ async fn extended_query_array_engine_smoke_and_const_stmt() { .await .expect("constant execute after Array DDL"); assert_eq!(const_rows.len(), 1, "constant projection must return 1 row"); - let x_text: String = const_rows[0].get::<_, String>(0); - assert_eq!(x_text, "1"); + let x: i64 = const_rows[0].get(0); + assert_eq!(x, 1); } // ── Cross-engine: parameter error cases ────────────────────────────────────── From c22ed895dbd0f955d1c1f990ce9b104319c7dc59 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 18 Sep 2026 07:36:23 +0800 Subject: [PATCH 08/15] feat(query): evaluate a SELECT-list sequence accessor per output row A sequence accessor (nextval/currval) in a SELECT list over a FROM relation becomes Projection::CpComputed instead of being refused. The Control Plane evaluates it once per output row after the Data Plane returns the fetched columns, then drops the base columns it read and writes the alias. Every other row-scope clause (WHERE, ORDER BY, GROUP BY, HAVING, JOIN ON, SET, aggregate/window arguments, subqueries) still refuses the call, since the row evaluator has no sequence state there. TableScope tracks whether it backs the statement's output SELECT and whether it currently allows a Control-Plane function; neither flag is inherited by a nested scope, so a correlated subquery or a non-output SELECT keeps refusing. Aggregate projections that carry a Control-Plane-computed item are wrapped so ORDER BY/LIMIT keep operating on the underlying aggregate output. Plan cache eligibility now treats a Control-Plane-computed projection as data-dependent, since it allocates on every execution. Wire tests replace the old blanket-refusal cases with coverage for the allowed SELECT-list position and the row-scope clauses that still refuse. --- nodedb-sql/src/error.rs | 20 +- nodedb-sql/src/lib.rs | 3 +- nodedb-sql/src/optimizer/point_get.rs | 9 +- nodedb-sql/src/planner/agg_naming.rs | 30 + nodedb-sql/src/planner/aggregate_cp_wrap.rs | 532 ++++++++++++++++++ nodedb-sql/src/planner/aggregate_order.rs | 35 +- nodedb-sql/src/planner/catalog_plan_shapes.rs | 18 +- nodedb-sql/src/planner/cp_projection.rs | 177 ++++++ nodedb-sql/src/planner/mod.rs | 2 + nodedb-sql/src/planner/select/derived_from.rs | 12 +- nodedb-sql/src/planner/select/entry.rs | 66 ++- nodedb-sql/src/planner/select/helpers.rs | 22 +- nodedb-sql/src/planner/select/mod.rs | 2 +- nodedb-sql/src/planner/select/post_process.rs | 4 +- nodedb-sql/src/planner/select/select_stmt.rs | 19 +- nodedb-sql/src/resolver/columns.rs | 43 +- nodedb-sql/src/resolver/expr/functions.rs | 39 +- nodedb-sql/src/resolver/scope.rs | 12 + nodedb-sql/src/types/plan/cacheability.rs | 74 +++ nodedb-sql/src/types/plan/expr_scan.rs | 259 +++++++++ nodedb-sql/src/types/plan/mod.rs | 5 + nodedb-sql/src/types/query.rs | 6 + .../sql_plan_convert/aggregate/projection.rs | 109 +++- .../sql_plan_convert/expr/inline_cte.rs | 13 +- .../sql_plan_convert/group_key_name.rs | 11 +- .../planner/sql_plan_convert/lateral.rs | 39 +- .../sql_plan_convert/output_schema/columns.rs | 63 +++ .../planner/sql_plan_convert/set_ops.rs | 12 + nodedb/tests/wire/cases/mod.rs | 2 + .../wire/cases/sql_sequence_row_scope.rs | 401 +++++++++++++ .../cases/sql_sequence_row_scope_refusals.rs | 142 +++++ nodedb/tests/wire/cases/sql_sequences.rs | 82 --- 32 files changed, 2086 insertions(+), 177 deletions(-) create mode 100644 nodedb-sql/src/planner/aggregate_cp_wrap.rs create mode 100644 nodedb-sql/src/planner/cp_projection.rs create mode 100644 nodedb-sql/src/types/plan/expr_scan.rs create mode 100644 nodedb/tests/wire/cases/sql_sequence_row_scope.rs create mode 100644 nodedb/tests/wire/cases/sql_sequence_row_scope_refusals.rs diff --git a/nodedb-sql/src/error.rs b/nodedb-sql/src/error.rs index ae2763b89..eaa5733aa 100644 --- a/nodedb-sql/src/error.rs +++ b/nodedb-sql/src/error.rs @@ -19,19 +19,21 @@ pub enum SqlError { #[error("function {name}(...) does not exist")] UndefinedFunction { name: String }, - /// A sequence accessor appeared where every output row needs its own - /// allocation, such as a SELECT list over a FROM clause. + /// A sequence accessor appeared in a row-scope clause the row evaluator + /// runs: WHERE, ORDER BY, GROUP BY, HAVING, JOIN ON, SET, an aggregate or + /// window argument, or a subquery. /// /// Rendered as SQLSTATE `0A000` (feature_not_supported). Constant /// contexts evaluate the call for real: a FROM-less `SELECT`, a column - /// `DEFAULT`, a `VALUES` list. The refusal keeps a per-row call from - /// reaching the row evaluator, which has no sequence state and would - /// return `NULL` for every row. + /// `DEFAULT`, a `VALUES` list. A SELECT-list item over a FROM relation + /// is evaluated by the Control Plane once per output row. The refusal + /// keeps every other per-row call from reaching the row evaluator, which + /// has no sequence state and would return `NULL` for every row. #[error( - "{name}(...) is not supported in a per-row context; \ - a SELECT list, WHERE clause, or SET clause over a FROM relation \ - evaluates once per row. Call it in a FROM-less SELECT or a column \ - DEFAULT instead" + "{name}(...) is not supported in this clause; over a FROM relation \ + only a SELECT-list item that holds no aggregate evaluates it, once \ + per output row. Call it there, in a FROM-less SELECT, or in a \ + column DEFAULT instead" )] SequencePerRowUnsupported { name: String }, diff --git a/nodedb-sql/src/lib.rs b/nodedb-sql/src/lib.rs index 7952b0505..377e74832 100644 --- a/nodedb-sql/src/lib.rs +++ b/nodedb-sql/src/lib.rs @@ -138,7 +138,8 @@ fn plan_statements( for stmt in statements { match classify(stmt) { StatementKind::Select(query) => { - let plan = planner::select::plan_query(query, catalog, &functions, temporal)?; + let plan = + planner::select::plan_statement_query(query, catalog, &functions, temporal)?; let plan = optimizer::optimize(plan, catalog); plans.push(plan); } diff --git a/nodedb-sql/src/optimizer/point_get.rs b/nodedb-sql/src/optimizer/point_get.rs index 843892c98..fbc446dc7 100644 --- a/nodedb-sql/src/optimizer/point_get.rs +++ b/nodedb-sql/src/optimizer/point_get.rs @@ -29,9 +29,12 @@ pub fn optimize(plan: SqlPlan, catalog: &dyn SqlCatalog) -> SqlPlan { .. } if filters.len() == 1 && !temporal.is_temporal() - && !projection - .iter() - .any(|p| matches!(p, Projection::Computed { .. })) => + && !projection.iter().any(|p| { + matches!( + p, + Projection::Computed { .. } | Projection::CpComputed { .. } + ) + }) => { let pk = catalog .get_collection(DatabaseId::DEFAULT, collection) diff --git a/nodedb-sql/src/planner/agg_naming.rs b/nodedb-sql/src/planner/agg_naming.rs index 4977ab3ac..be9c7252b 100644 --- a/nodedb-sql/src/planner/agg_naming.rs +++ b/nodedb-sql/src/planner/agg_naming.rs @@ -52,3 +52,33 @@ pub fn aggregate_field_name(a: &AggregateExpr) -> String { pub fn aggregate_output_key(a: &AggregateExpr) -> String { nodedb_query::agg_key::canonical_agg_key(&aggregate_function_name(a), &aggregate_field_name(a)) } + +/// The internal name of a computed (non-column) GROUP BY key: a stable +/// `group_{index}` placeholder. The executor emits the evaluated key under +/// it, and the response shaper reads the value back by it. The SELECT alias +/// never reaches this name; it is a display name only. +pub fn computed_group_key_name(index: usize) -> String { + format!("group_{index}") +} + +/// The key a finalized group row carries one GROUP BY key under. A column +/// key keeps its bare column name; a computed key uses +/// [`computed_group_key_name`]. +pub fn group_key_row_name(key: &SqlExpr, index: usize) -> String { + match key { + SqlExpr::Column { name, .. } => name.clone(), + SqlExpr::Function { .. } + | SqlExpr::BinaryOp { .. } + | SqlExpr::UnaryOp { .. } + | SqlExpr::Cast { .. } + | SqlExpr::IsNull { .. } + | SqlExpr::Case { .. } + | SqlExpr::InList { .. } + | SqlExpr::Between { .. } + | SqlExpr::Like { .. } + | SqlExpr::ArrayLiteral(_) + | SqlExpr::Literal(_) + | SqlExpr::Subquery(_) + | SqlExpr::Wildcard => computed_group_key_name(index), + } +} diff --git a/nodedb-sql/src/planner/aggregate_cp_wrap.rs b/nodedb-sql/src/planner/aggregate_cp_wrap.rs new file mode 100644 index 000000000..51cab4b56 --- /dev/null +++ b/nodedb-sql/src/planner/aggregate_cp_wrap.rs @@ -0,0 +1,532 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Wrapping a grouped plan so a Control-Plane-computed SELECT-list item keeps +//! its position. +//! +//! An `Aggregate` plan emits group keys and aggregate values only. A SELECT +//! item that calls a sequence accessor is neither, so the aggregate cannot +//! carry it. The planner wraps the finished aggregate in a `Subquery` whose +//! projection restates every output column in SELECT-list order and places a +//! [`Projection::CpComputed`] entry at the item's position. The wrap runs +//! after ORDER BY and LIMIT are attached, so those stay on the aggregate. + +use sqlparser::ast; + +use crate::aggregate_walk::contains_aggregate; +use crate::error::{Result, SqlError}; +use crate::functions::registry::FunctionRegistry; +use crate::parser::normalize::normalize_ident; +use crate::planner::agg_naming::group_key_row_name; +use crate::planner::aggregate_order::compute_output_order_by_item; +use crate::planner::cp_projection::{ast_calls_sequence_accessor, ast_sequence_accessor}; +use crate::resolver::ColumnScope; +use crate::resolver::columns::TableScope; +use crate::resolver::expr::convert_expr; +use crate::types::plan::{first_sequence_accessor, referenced_columns}; +use crate::types::{AggOutputSlot, AggregateExpr, Projection, SqlExpr, SqlPlan}; + +/// Wrap `plan` when `items` hold a sequence accessor the aggregate cannot +/// carry. A plan with no such item, or one that is not grouped, returns +/// unchanged: `convert_projection` marks the item on a row plan itself. +/// +/// `grouped` says whether the SELECT aggregates (GROUP BY, or an aggregate +/// call in its list). `scope` is the relation set the SELECT resolved +/// against. A grouped SELECT that does not produce the statement's output +/// rows (a subquery, CTE body, UNION branch, derived table) refuses the +/// item: the aggregate planner never converts a bare non-key item, so +/// nothing else reports it. +pub fn wrap_aggregate_cp_items( + plan: SqlPlan, + items: &[ast::SelectItem], + grouped: bool, + functions: &FunctionRegistry, + scope: &TableScope, +) -> Result { + if !grouped || !scope.is_row_scope() { + return Ok(plan); + } + let Some(accessor) = items.iter().find_map(item_accessor) else { + return Ok(plan); + }; + if !scope.is_statement_output() { + return Err(SqlError::SequencePerRowUnsupported { name: accessor }); + } + match plan { + SqlPlan::Cte { definitions, outer } => Ok(SqlPlan::Cte { + definitions, + outer: Box::new(wrap_aggregate_cp_items( + *outer, items, grouped, functions, scope, + )?), + }), + SqlPlan::Aggregate { .. } => { + let projection = aggregate_cp_projection(&plan, items, functions, scope)?; + Ok(SqlPlan::Subquery { + input: Box::new(plan), + filters: Vec::new(), + projection, + window_functions: Vec::new(), + sort_keys: Vec::new(), + offset: 0, + distinct: false, + limit: None, + }) + } + // An OFFSET the aggregate cannot hold already sits in a + // post-processing tail with no projection of its own. The tail + // projects after it offsets and sorts, so the restated columns land + // on it instead of a second wrapper. + SqlPlan::Subquery { + input, + filters, + projection, + window_functions, + sort_keys, + offset, + distinct, + limit, + } if matches!(*input, SqlPlan::Aggregate { .. }) && projection.is_empty() => { + let projection = aggregate_cp_projection(&input, items, functions, scope)?; + Ok(SqlPlan::Subquery { + input, + filters, + projection, + window_functions, + sort_keys, + offset, + distinct, + limit, + }) + } + // A grouped query the engine rules planned as another shape (a + // timeseries bucket scan, a grouped query joined to a scalar + // subquery) names its output columns by rules this wrap does not + // restate. Refusing beats dropping the item from the result. + other => Err(SqlError::Unsupported { + detail: format!( + "a sequence accessor in the SELECT list of a grouped query planned as {}; \ + call it in a FROM-less SELECT instead", + other.variant_name() + ), + }), + } +} + +/// The sequence accessor a SELECT item calls, when it calls one. +fn item_accessor(item: &ast::SelectItem) -> Option { + match item { + ast::SelectItem::UnnamedExpr(expr) | ast::SelectItem::ExprWithAlias { expr, .. } => { + ast_sequence_accessor(expr) + } + ast::SelectItem::ExprWithAliases { .. } + | ast::SelectItem::Wildcard(_) + | ast::SelectItem::QualifiedWildcard(..) => None, + } +} + +/// The projection restating `plan`'s output columns in SELECT-list order, +/// with a [`Projection::CpComputed`] entry at each accessor item's position. +/// `plan` is the `Aggregate` the items were planned into. +fn aggregate_cp_projection( + plan: &SqlPlan, + items: &[ast::SelectItem], + functions: &FunctionRegistry, + scope: &TableScope, +) -> Result> { + let (group_by, aggregates) = match plan { + SqlPlan::Aggregate { + group_by, + aggregates, + .. + } => (group_by, aggregates), + other => { + return Err(SqlError::Unsupported { + detail: format!("aggregate wrap over a {} plan", other.variant_name()), + }); + } + }; + let key_names: Vec = group_by + .iter() + .enumerate() + .map(|(index, key)| group_key_row_name(key, index)) + .collect(); + let by_item = compute_output_order_by_item(items, group_by, functions, scope)?; + // The item resolves against the input relations, so an unknown column is + // the usual resolve error, and against the group-key row names, so a + // computed key (`group_0`) is addressable. The reference check below + // then narrows to the keys: the finalized group row carries nothing + // else the Control Plane can read. + let cp_scope = scope + .with_output_names(key_names.iter().cloned()) + .allowing_cp_functions(); + + let mut projection = Vec::with_capacity(items.len()); + for (item, slots) in items.iter().zip(by_item) { + let (expr, alias) = match item { + ast::SelectItem::UnnamedExpr(expr) => (expr, format!("{expr}").to_lowercase()), + ast::SelectItem::ExprWithAlias { expr, alias } => (expr, normalize_ident(alias)), + ast::SelectItem::ExprWithAliases { .. } + | ast::SelectItem::Wildcard(_) + | ast::SelectItem::QualifiedWildcard(..) => continue, + }; + if ast_calls_sequence_accessor(expr) { + let converted = convert_expr(expr, &ColumnScope::Relations(&cp_scope))?; + if let Some(name) = first_sequence_accessor(&converted) { + let name = name.to_string(); + projection.push(cp_item( + converted, name, alias, expr, &key_names, functions, + )?); + continue; + } + } + for slot in slots { + projection.push(Projection::Column(slot_row_name( + slot, &key_names, aggregates, + )?)); + } + } + Ok(projection) +} + +/// The Control-Plane entry for one accessor item over a grouped result. +/// +/// The item holds no aggregate: the Control Plane evaluates it over the +/// finalized group row, which carries aggregate values under their output +/// names but no per-row aggregate state. Every column it references is a +/// group key, unqualified so the reference matches the row's bare key. +fn cp_item( + converted: SqlExpr, + accessor: String, + alias: String, + raw: &ast::Expr, + key_names: &[String], + functions: &FunctionRegistry, +) -> Result { + if contains_aggregate(raw, functions) { + return Err(SqlError::SequencePerRowUnsupported { name: accessor }); + } + for column in referenced_columns(&converted) { + let bare = column.rsplit('.').next().unwrap_or(&column); + if !key_names.iter().any(|key| key.eq_ignore_ascii_case(bare)) { + return Err(SqlError::Unsupported { + detail: format!( + "column '{column}' beside a sequence accessor in a grouped SELECT list \ + must be a GROUP BY key" + ), + }); + } + } + Ok(Projection::CpComputed { + expr: unqualify_columns(converted), + alias, + }) +} + +/// The key a finalized group row carries one output slot under. +fn slot_row_name( + slot: AggOutputSlot, + key_names: &[String], + aggregates: &[AggregateExpr], +) -> Result { + match slot { + AggOutputSlot::GroupKey(index) => key_names.get(index).cloned(), + AggOutputSlot::Aggregate(index) => aggregates.get(index).map(|a| a.alias.clone()), + } + .ok_or_else(|| SqlError::Unsupported { + detail: format!("aggregate output slot {slot:?} names no output column"), + }) +} + +/// `expr` with every column reference stripped of its table qualifier. A +/// finalized group row keys its columns by bare name. +fn unqualify_columns(expr: SqlExpr) -> SqlExpr { + match expr { + SqlExpr::Column { name, .. } => SqlExpr::Column { table: None, name }, + SqlExpr::Function { + name, + args, + distinct, + } => SqlExpr::Function { + name, + args: args.into_iter().map(unqualify_columns).collect(), + distinct, + }, + SqlExpr::BinaryOp { left, op, right } => SqlExpr::BinaryOp { + left: Box::new(unqualify_columns(*left)), + op, + right: Box::new(unqualify_columns(*right)), + }, + SqlExpr::UnaryOp { op, expr } => SqlExpr::UnaryOp { + op, + expr: Box::new(unqualify_columns(*expr)), + }, + SqlExpr::Cast { expr, to_type } => SqlExpr::Cast { + expr: Box::new(unqualify_columns(*expr)), + to_type, + }, + SqlExpr::IsNull { expr, negated } => SqlExpr::IsNull { + expr: Box::new(unqualify_columns(*expr)), + negated, + }, + SqlExpr::Case { + operand, + when_then, + else_expr, + } => SqlExpr::Case { + operand: operand.map(|e| Box::new(unqualify_columns(*e))), + when_then: when_then + .into_iter() + .map(|(when, then)| (unqualify_columns(when), unqualify_columns(then))) + .collect(), + else_expr: else_expr.map(|e| Box::new(unqualify_columns(*e))), + }, + SqlExpr::InList { + expr, + list, + negated, + } => SqlExpr::InList { + expr: Box::new(unqualify_columns(*expr)), + list: list.into_iter().map(unqualify_columns).collect(), + negated, + }, + SqlExpr::Between { + expr, + low, + high, + negated, + } => SqlExpr::Between { + expr: Box::new(unqualify_columns(*expr)), + low: Box::new(unqualify_columns(*low)), + high: Box::new(unqualify_columns(*high)), + negated, + }, + SqlExpr::Like { + expr, + pattern, + negated, + case_insensitive, + } => SqlExpr::Like { + expr: Box::new(unqualify_columns(*expr)), + pattern: Box::new(unqualify_columns(*pattern)), + negated, + case_insensitive, + }, + SqlExpr::ArrayLiteral(items) => { + SqlExpr::ArrayLiteral(items.into_iter().map(unqualify_columns).collect()) + } + SqlExpr::Literal(_) | SqlExpr::Subquery(_) | SqlExpr::Wildcard => expr, + } +} + +#[cfg(test)] +mod tests { + use crate::types::{ + CollectionInfo, EngineType, PlanCacheEligibility, Projection, SqlCatalog, SqlCatalogError, + SqlExpr, SqlPlan, + }; + use crate::{SqlError, plan_sql}; + + struct Catalog; + + impl SqlCatalog for Catalog { + fn get_collection( + &self, + _: nodedb_types::DatabaseId, + name: &str, + ) -> std::result::Result, SqlCatalogError> { + if name != "t" { + return Ok(None); + } + Ok(Some(CollectionInfo { + name: "t".into(), + engine: EngineType::DocumentSchemaless, + columns: Vec::new(), + primary_key: Some("id".into()), + has_auto_tier: false, + indexes: Vec::new(), + bitemporal: false, + primary: nodedb_types::PrimaryEngine::Document, + vector_primary: None, + partition_strategy: nodedb_types::PartitionStrategy::CollectionHomed, + open_schema: CollectionInfo::open_schema_for(EngineType::DocumentSchemaless), + })) + } + } + + fn plan(sql: &str) -> SqlPlan { + plan_sql(sql, &Catalog) + .unwrap_or_else(|e| panic!("{sql}: {e}")) + .remove(0) + } + + fn plan_err(sql: &str) -> SqlError { + match plan_sql(sql, &Catalog) { + Ok(plans) => panic!("{sql} must not plan, got {plans:?}"), + Err(e) => e, + } + } + + fn is_nextval(expr: &SqlExpr) -> bool { + matches!(expr, SqlExpr::Function { name, .. } if name == "nextval") + } + + #[test] + fn a_scan_marks_the_accessor_item_as_control_plane_computed() { + let plan = plan("SELECT id, nextval('s') FROM t"); + let SqlPlan::Scan { projection, .. } = &plan else { + panic!("expected Scan, got {plan:?}"); + }; + assert_eq!(projection.len(), 2); + assert!(matches!(&projection[0], Projection::Column(name) if name == "id")); + match &projection[1] { + Projection::CpComputed { expr, alias } => { + assert_eq!(alias, "nextval('s')"); + assert!(is_nextval(expr)); + } + other => panic!("expected CpComputed, got {other:?}"), + } + assert_eq!( + plan.cache_eligibility(), + PlanCacheEligibility::DataDependent + ); + } + + #[test] + fn a_wider_aliased_expression_keeps_its_alias() { + let plan = plan("SELECT nextval('s') * 2 AS n FROM t"); + let SqlPlan::Scan { projection, .. } = &plan else { + panic!("expected Scan, got {plan:?}"); + }; + match &projection[0] { + Projection::CpComputed { expr, alias } => { + assert_eq!(alias, "n"); + assert!(matches!(expr, SqlExpr::BinaryOp { .. })); + } + other => panic!("expected CpComputed, got {other:?}"), + } + } + + #[test] + fn every_other_row_clause_refuses_the_accessor() { + for sql in [ + "SELECT id FROM t WHERE nextval('s') > 0", + "SELECT id FROM t ORDER BY nextval('s')", + "SELECT SUM(nextval('s')) FROM t", + "SELECT grp, COUNT(*) + nextval('s') FROM t GROUP BY grp", + "SELECT grp, COUNT(*) FROM t GROUP BY grp HAVING COUNT(*) > nextval('s')", + "SELECT n FROM (SELECT id, nextval('s') AS n FROM t) d", + "SELECT n FROM (SELECT grp, COUNT(*), nextval('s') AS n FROM t GROUP BY grp) d", + "SELECT id FROM t UNION ALL SELECT nextval('s') FROM t", + "WITH c AS (SELECT nextval('s') AS n FROM t) SELECT n FROM c", + "INSERT INTO t (id, n) SELECT id, nextval('s') FROM t", + ] { + let err = plan_err(sql); + assert!( + matches!(err, SqlError::SequencePerRowUnsupported { .. }), + "{sql}: expected SequencePerRowUnsupported, got {err:?}" + ); + } + } + + #[test] + fn a_grouped_query_wraps_the_aggregate_in_select_list_order() { + let plan = plan("SELECT grp, COUNT(*), nextval('s') FROM t GROUP BY grp"); + let SqlPlan::Subquery { + input, + projection, + filters, + sort_keys, + limit, + .. + } = &plan + else { + panic!("expected Subquery, got {plan:?}"); + }; + assert!(matches!(**input, SqlPlan::Aggregate { .. })); + assert!(filters.is_empty()); + assert!(sort_keys.is_empty()); + assert_eq!(*limit, None); + assert_eq!(projection.len(), 3); + assert!(matches!(&projection[0], Projection::Column(name) if name == "grp")); + assert!(matches!(&projection[1], Projection::Column(name) if name == "count(*)")); + assert!(matches!(&projection[2], Projection::CpComputed { expr, .. } if is_nextval(expr))); + assert_eq!( + plan.cache_eligibility(), + PlanCacheEligibility::DataDependent + ); + } + + #[test] + fn order_by_and_limit_stay_on_the_wrapped_aggregate() { + let plan = plan( + "SELECT t.grp, nextval('s') + t.grp AS n, COUNT(*) AS c FROM t \ + GROUP BY t.grp ORDER BY c DESC LIMIT 5", + ); + let SqlPlan::Subquery { + input, projection, .. + } = &plan + else { + panic!("expected Subquery, got {plan:?}"); + }; + let SqlPlan::Aggregate { + sort_keys, limit, .. + } = &**input + else { + panic!("expected Aggregate, got {input:?}"); + }; + assert_eq!(sort_keys.len(), 1); + assert_eq!(*limit, 5); + assert!(matches!(&projection[0], Projection::Column(name) if name == "grp")); + match &projection[1] { + Projection::CpComputed { expr, alias } => { + assert_eq!(alias, "n"); + let SqlExpr::BinaryOp { right, .. } = expr else { + panic!("expected a binary expression, got {expr:?}"); + }; + assert!( + matches!(&**right, SqlExpr::Column { table: None, name } if name == "grp"), + "the key reference must lose its qualifier, got {right:?}" + ); + } + other => panic!("expected CpComputed, got {other:?}"), + } + assert!(matches!(&projection[2], Projection::Column(name) if name == "c")); + } + + #[test] + fn an_offset_tail_carries_the_restated_projection_itself() { + let plan = plan("SELECT grp, nextval('s'), COUNT(*) FROM t GROUP BY grp OFFSET 2"); + let SqlPlan::Subquery { + input, + projection, + offset, + .. + } = &plan + else { + panic!("expected Subquery, got {plan:?}"); + }; + assert!(matches!(**input, SqlPlan::Aggregate { .. })); + assert_eq!(*offset, 2); + assert_eq!(projection.len(), 3); + assert!(matches!(&projection[1], Projection::CpComputed { .. })); + } + + #[test] + fn a_grouped_accessor_item_may_reference_group_keys_only() { + let err = plan_err("SELECT grp, COUNT(*), nextval('s') + other FROM t GROUP BY grp"); + assert!( + matches!(err, SqlError::Unsupported { ref detail } if detail.contains("'other'")), + "got {err:?}" + ); + } + + #[test] + fn a_from_less_select_still_folds_the_accessor_at_plan_time() { + let plan = plan("SELECT 1 AS one"); + assert!(matches!(plan, SqlPlan::ConstantResult { .. })); + let err = plan_err("SELECT nextval('s')"); + assert!( + !matches!(err, SqlError::SequencePerRowUnsupported { .. }), + "a FROM-less accessor reaches the catalog, not the per-row gate: {err:?}" + ); + } +} diff --git a/nodedb-sql/src/planner/aggregate_order.rs b/nodedb-sql/src/planner/aggregate_order.rs index 6b89fe2d2..9a43fabc6 100644 --- a/nodedb-sql/src/planner/aggregate_order.rs +++ b/nodedb-sql/src/planner/aggregate_order.rs @@ -35,15 +35,40 @@ pub fn compute_output_order( functions: &FunctionRegistry, scope: &TableScope, ) -> Result> { + Ok( + compute_output_order_by_item(projection, group_by, functions, scope)? + .into_iter() + .flatten() + .collect(), + ) +} + +/// The output slots each projection item produces, one entry per item in +/// SELECT-list order. An item that is neither a group key nor an aggregate +/// (a star, or a bare non-key expression) produces an empty entry. +/// +/// The flattened entries are exactly [`compute_output_order`]; the per-item +/// grouping lets a caller interleave other output columns at their +/// SELECT-list positions. +pub fn compute_output_order_by_item( + projection: &[ast::SelectItem], + group_by: &[SqlExpr], + functions: &FunctionRegistry, + scope: &TableScope, +) -> Result>> { let real_agg_count = extract_aggregates_from_projection(projection, functions, scope)?.len(); - let mut order = Vec::new(); + let mut by_item = Vec::with_capacity(projection.len()); let mut agg_cursor = 0usize; let mut grouping_cursor = 0usize; for item in projection { + let mut order = Vec::new(); let expr = match item { ast::SelectItem::UnnamedExpr(expr) => expr, ast::SelectItem::ExprWithAlias { expr, .. } => expr, - _ => continue, + _ => { + by_item.push(order); + continue; + } }; // Bare column that names one of the GROUP BY keys. if let Some(name) = expr_column_name(expr) @@ -52,6 +77,7 @@ pub fn compute_output_order( .position(|key| key_column_name(key) == Some(name.as_str())) { order.push(AggOutputSlot::GroupKey(index)); + by_item.push(order); continue; } // Computed expression that structurally matches a GROUP BY key @@ -69,6 +95,7 @@ pub fn compute_output_order( .position(|key| format!("{key:?}") == rendered) { order.push(AggOutputSlot::GroupKey(index)); + by_item.push(order); continue; } } @@ -81,6 +108,7 @@ pub fn compute_output_order( order.push(AggOutputSlot::Aggregate(real_agg_count + grouping_cursor)); grouping_cursor += 1; } + by_item.push(order); continue; } // Ordinary aggregate expression(s). @@ -96,8 +124,9 @@ pub fn compute_output_order( agg_cursor += 1; } } + by_item.push(order); } - Ok(order) + Ok(by_item) } /// Count the `GROUPING(col)` calls reachable in `expr`, mirroring the planner's diff --git a/nodedb-sql/src/planner/catalog_plan_shapes.rs b/nodedb-sql/src/planner/catalog_plan_shapes.rs index 20e393f3e..a281daaf0 100644 --- a/nodedb-sql/src/planner/catalog_plan_shapes.rs +++ b/nodedb-sql/src/planner/catalog_plan_shapes.rs @@ -16,8 +16,11 @@ pub(super) fn validate_projection( tenant_id: u64, ) -> crate::Result<()> { for item in projection { - if let Projection::Computed { expr, .. } = item { - validate_expr(expr, catalog, database_id, tenant_id)?; + match item { + Projection::Computed { expr, .. } | Projection::CpComputed { expr, .. } => { + validate_expr(expr, catalog, database_id, tenant_id)?; + } + Projection::Column(_) | Projection::Star | Projection::QualifiedStar(_) => {} } } Ok(()) @@ -70,10 +73,15 @@ pub(super) fn fold_projection( database_id: DatabaseId, tenant_id: u64, ) { + // The fold resolves catalog casts (`regclass`, `regtype`) only; a + // sequence accessor in a Control-Plane-computed item stays a call. for item in projection { - if let Projection::Computed { expr, .. } = item { - let owned = std::mem::replace(expr, SqlExpr::Wildcard); - *expr = fold_expr(owned, catalog, database_id, tenant_id); + match item { + Projection::Computed { expr, .. } | Projection::CpComputed { expr, .. } => { + let owned = std::mem::replace(expr, SqlExpr::Wildcard); + *expr = fold_expr(owned, catalog, database_id, tenant_id); + } + Projection::Column(_) | Projection::Star | Projection::QualifiedStar(_) => {} } } } diff --git a/nodedb-sql/src/planner/cp_projection.rs b/nodedb-sql/src/planner/cp_projection.rs new file mode 100644 index 000000000..efd9c0a82 --- /dev/null +++ b/nodedb-sql/src/planner/cp_projection.rs @@ -0,0 +1,177 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Control-Plane-computed SELECT-list items. +//! +//! A SELECT-list item over a relation that calls a sequence accessor cannot +//! run on the Data Plane: the row evaluator holds no sequence state. The +//! planner marks such an item [`Projection::CpComputed`], and the Control +//! Plane evaluates it once per output row after the rows return. Detection +//! runs on the AST, so an item with no accessor converts exactly as before, +//! against the ordinary scope. + +use core::ops::ControlFlow; + +use sqlparser::ast::{self, Expr, Visit, Visitor}; + +use crate::error::Result; +use crate::functions::sequence_accessor::is_sequence_accessor; +use crate::resolver::ColumnScope; +use crate::resolver::columns::TableScope; +use crate::resolver::expr::convert_expr; +use crate::types::Projection; +use crate::types::plan::contains_sequence_accessor; + +/// A copy of `scope` that resolves sequence accessors, when `scope` iterates +/// rows and backs the statement's output SELECT. A FROM-less SELECT gets +/// `None`: the planner evaluates the accessor itself there, and the item +/// stays an ordinary `Computed` entry. A nested SELECT (subquery, CTE +/// body, UNION branch, derived table, INSERT or MERGE source) gets `None` +/// too: the Control Plane evaluates the statement's output rows only, so +/// the ordinary scope refuses the accessor there. +pub fn cp_projection_scope(scope: &TableScope) -> Option { + (scope.is_row_scope() && scope.is_statement_output()) + .then(|| scope.clone().allowing_cp_functions()) +} + +/// The lowercase name of the first sequence accessor `expr` calls, at any +/// nesting depth. +pub fn ast_sequence_accessor(expr: &Expr) -> Option { + let mut detector = AccessorDetector { found: None }; + let _ = expr.visit(&mut detector); + detector.found +} + +/// Whether `expr` calls a sequence accessor anywhere, at any nesting depth. +pub fn ast_calls_sequence_accessor(expr: &Expr) -> bool { + ast_sequence_accessor(expr).is_some() +} + +struct AccessorDetector { + found: Option, +} + +impl Visitor for AccessorDetector { + type Break = (); + + fn pre_visit_expr(&mut self, expr: &Expr) -> ControlFlow<()> { + if let Expr::Function(f) = expr + && f.name.0.len() == 1 + && let Some(ast::ObjectNamePart::Identifier(ident)) = f.name.0.first() + && is_sequence_accessor(&ident.value) + { + self.found = Some(ident.value.to_ascii_lowercase()); + return ControlFlow::Break(()); + } + ControlFlow::Continue(()) + } +} + +/// Convert one SELECT item into a [`Projection::CpComputed`] when it calls a +/// sequence accessor over a relation. `None` when the item holds no accessor +/// or no relation is in scope; the caller converts it the ordinary way. +/// +/// An accessor inside a subquery is not the item's own: the subquery body +/// resolves in a nested scope, which refuses it. +pub fn convert_cp_item( + expr: &Expr, + alias: String, + cp_scope: Option<&TableScope>, +) -> Result> { + let Some(cp_scope) = cp_scope else { + return Ok(None); + }; + if !ast_calls_sequence_accessor(expr) { + return Ok(None); + } + let converted = convert_expr(expr, &ColumnScope::Relations(cp_scope))?; + if !contains_sequence_accessor(&converted) { + return Ok(None); + } + Ok(Some(Projection::CpComputed { + expr: converted, + alias, + })) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::resolver::columns::test_support::open_scope; + use crate::types::SqlExpr; + use sqlparser::dialect::GenericDialect; + use sqlparser::parser::Parser; + + fn select_expr(sql: &str) -> Expr { + let stmts = Parser::parse_sql(&GenericDialect {}, sql).unwrap(); + match stmts.into_iter().next().unwrap() { + ast::Statement::Query(q) => match *q.body { + ast::SetExpr::Select(s) => match s.projection.into_iter().next().unwrap() { + ast::SelectItem::UnnamedExpr(e) + | ast::SelectItem::ExprWithAlias { expr: e, .. } => e, + other => panic!("unexpected projection item {other:?}"), + }, + _ => panic!("expected SELECT"), + }, + _ => panic!("expected query"), + } + } + + #[test] + fn ast_detection_sees_nested_accessors_only() { + assert!(ast_calls_sequence_accessor(&select_expr( + "SELECT CASE WHEN a > 0 THEN nextval('s') * 2 ELSE 0 END" + ))); + assert!(ast_calls_sequence_accessor(&select_expr( + "SELECT coalesce(currval('s'), 0)" + ))); + assert!(!ast_calls_sequence_accessor(&select_expr( + "SELECT upper(a) || 'nextval'" + ))); + } + + #[test] + fn a_from_less_scope_never_marks_an_item() { + let scope = TableScope::new().as_statement_output(); + assert!(cp_projection_scope(&scope).is_none()); + let item = convert_cp_item(&select_expr("SELECT nextval('s')"), "n".into(), None).unwrap(); + assert!(item.is_none()); + } + + #[test] + fn a_nested_select_scope_never_marks_an_item() { + assert!(cp_projection_scope(&open_scope("t")).is_none()); + let nested = open_scope("t") + .as_statement_output() + .nested_in(TableScope::new()); + assert!(cp_projection_scope(&nested).is_none()); + } + + #[test] + fn an_accessor_item_over_a_relation_is_cp_computed() { + let cp_scope = cp_projection_scope(&open_scope("t").as_statement_output()) + .expect("statement-output row scope"); + let item = convert_cp_item( + &select_expr("SELECT nextval('s') + id"), + "n".into(), + Some(&cp_scope), + ) + .unwrap() + .expect("accessor item is Control-Plane computed"); + match item { + Projection::CpComputed { expr, alias } => { + assert_eq!(alias, "n"); + assert!(matches!(expr, SqlExpr::BinaryOp { .. })); + } + other => panic!("expected CpComputed, got {other:?}"), + } + } + + #[test] + fn a_plain_item_over_a_relation_is_left_to_the_ordinary_path() { + let cp_scope = cp_projection_scope(&open_scope("t").as_statement_output()) + .expect("statement-output row scope"); + let item = + convert_cp_item(&select_expr("SELECT id + 1"), "n".into(), Some(&cp_scope)).unwrap(); + assert!(item.is_none()); + } +} diff --git a/nodedb-sql/src/planner/mod.rs b/nodedb-sql/src/planner/mod.rs index e222cbc53..7d9588428 100644 --- a/nodedb-sql/src/planner/mod.rs +++ b/nodedb-sql/src/planner/mod.rs @@ -3,6 +3,7 @@ pub mod agg_bind; pub mod agg_naming; pub mod aggregate; +pub mod aggregate_cp_wrap; pub mod aggregate_order; pub mod array_ddl; pub mod array_dml; @@ -14,6 +15,7 @@ pub mod catalog_fold; pub mod catalog_plan_shapes; pub mod catalog_plan_validate; pub mod const_fold; +pub mod cp_projection; pub mod cte; pub mod declared_type_coerce; pub mod defaults; diff --git a/nodedb-sql/src/planner/select/derived_from.rs b/nodedb-sql/src/planner/select/derived_from.rs index a75398041..5be66f508 100644 --- a/nodedb-sql/src/planner/select/derived_from.rs +++ b/nodedb-sql/src/planner/select/derived_from.rs @@ -29,6 +29,7 @@ pub(in crate::planner::select) fn try_plan_derived_from( functions: &FunctionRegistry, temporal: TemporalScope, tail: &QueryTail<'_>, + statement_output: bool, ) -> Result> { if select.from.len() != 1 { return Ok(None); @@ -82,7 +83,16 @@ pub(in crate::planner::select) fn try_plan_derived_from( sample: None, index_hints: Vec::new(), }; - let outer = plan_select(&outer_select, &derived_catalog, functions, temporal, tail)?; + // The outer SELECT keeps the caller's output standing; the derived body + // planned above is nested. + let outer = plan_select( + &outer_select, + &derived_catalog, + functions, + temporal, + tail, + statement_output, + )?; Ok(Some(PlannedSelect { plan: SqlPlan::Cte { diff --git a/nodedb-sql/src/planner/select/entry.rs b/nodedb-sql/src/planner/select/entry.rs index 70a2dc426..d3d6f61d2 100644 --- a/nodedb-sql/src/planner/select/entry.rs +++ b/nodedb-sql/src/planner/select/entry.rs @@ -11,7 +11,7 @@ use super::cte_catalog::CteCatalog; use super::limit::apply_limit; use super::order_by::{apply_order_by, try_hybrid_from_projection}; use super::query_tail::QueryTail; -use super::select_stmt::plan_select; +use super::select_stmt::{has_aggregation, plan_select}; use crate::error::{Result, SqlError}; use crate::functions::registry::FunctionRegistry; use crate::reserved::check_ast_identifier; @@ -48,18 +48,50 @@ fn is_pure_vector_projection(projection: &[Projection]) -> bool { return false; } } - Projection::Star | Projection::QualifiedStar(_) => return false, + // A Control-Plane-computed item is evaluated over the fetched + // payload, so the payload must be fetched. + Projection::CpComputed { .. } | Projection::Star | Projection::QualifiedStar(_) => { + return false; + } } } true } -/// Plan a SELECT query. +/// Plan a SELECT query that produces the statement's result rows. +/// +/// Only this SELECT list may hold a Control-Plane-computed item (a sequence +/// accessor over a relation): the Control Plane evaluates the statement's +/// output rows and nothing deeper. +pub fn plan_statement_query( + query: &Query, + catalog: &dyn SqlCatalog, + functions: &FunctionRegistry, + temporal: TemporalScope, +) -> Result { + plan_query_at(query, catalog, functions, temporal, true) +} + +/// Plan a nested SELECT query: a subquery, CTE body, UNION branch, derived +/// table, INSERT source, or MERGE source. Its SELECT list refuses a +/// sequence accessor over a relation. pub fn plan_query( query: &Query, catalog: &dyn SqlCatalog, functions: &FunctionRegistry, temporal: TemporalScope, +) -> Result { + plan_query_at(query, catalog, functions, temporal, false) +} + +/// Plan a SELECT query. `statement_output` says whether its rows are the +/// statement's result; a WITH clause passes it to the outer query only. +fn plan_query_at( + query: &Query, + catalog: &dyn SqlCatalog, + functions: &FunctionRegistry, + temporal: TemporalScope, + statement_output: bool, ) -> Result { // Handle CTEs (WITH clause). if let Some(with) = &query.with @@ -106,7 +138,13 @@ pub fn plan_query( inner: catalog, relations, }; - let outer = plan_query(&inner_query, &cte_catalog, functions, temporal)?; + let outer = plan_query_at( + &inner_query, + &cte_catalog, + functions, + temporal, + statement_output, + )?; return Ok(SqlPlan::Cte { definitions, @@ -125,7 +163,14 @@ pub fn plan_query( limit_clause: &query.limit_clause, fetch: query.fetch.as_ref(), }; - let planned = plan_select(select, catalog, functions, temporal, &tail)?; + let planned = plan_select( + select, + catalog, + functions, + temporal, + &tail, + statement_output, + )?; let scope = planned.scope; let mut plan = planned.plan; // Snapshot the projection before ORDER BY transforms the plan, @@ -359,7 +404,16 @@ pub fn plan_query( } } } - apply_limit(plan, &tail) + let plan = apply_limit(plan, &tail)?; + // ORDER BY and LIMIT sit on the aggregate by now, so the wrap + // only restates the output columns around it. + crate::planner::aggregate_cp_wrap::wrap_aggregate_cp_items( + plan, + &select.projection, + has_aggregation(select, functions), + functions, + &scope, + ) } SetExpr::SetOperation { op, diff --git a/nodedb-sql/src/planner/select/helpers.rs b/nodedb-sql/src/planner/select/helpers.rs index c46421920..54a0df2d4 100644 --- a/nodedb-sql/src/planner/select/helpers.rs +++ b/nodedb-sql/src/planner/select/helpers.rs @@ -7,6 +7,7 @@ use sqlparser::ast; use crate::error::{Result, SqlError}; use crate::parser::normalize::{SCHEMA_QUALIFIED_MSG, normalize_ident}; +use crate::planner::cp_projection::{convert_cp_item, cp_projection_scope}; use crate::planner::predicate_coerce::coerce_predicate_literals; use crate::resolver::ColumnScope; use crate::resolver::columns::TableScope; @@ -27,15 +28,27 @@ pub(super) fn source_projection(plan: &SqlPlan) -> Vec { } /// Convert SELECT projection items. +/// +/// An item over a relation that calls a sequence accessor becomes +/// [`Projection::CpComputed`]: the Control Plane evaluates it once per +/// output row. Every other item resolves against `scope` unchanged, so an +/// accessor nested where it must not be (an aggregate argument, a subquery) +/// still refuses through that clause's own scope. pub fn convert_projection( items: &[ast::SelectItem], scope: &TableScope, ) -> Result> { + let cp_scope = cp_projection_scope(scope); let scope = ColumnScope::Relations(scope); let mut result = Vec::new(); for item in items { match item { ast::SelectItem::UnnamedExpr(expr) => { + let alias = format!("{expr}").to_lowercase(); + if let Some(cp) = convert_cp_item(expr, alias.clone(), cp_scope.as_ref())? { + result.push(cp); + continue; + } let sql_expr = convert_expr(expr, &scope)?; match &sql_expr { SqlExpr::Column { table, name } => { @@ -47,16 +60,21 @@ pub fn convert_projection( _ => { result.push(Projection::Computed { expr: sql_expr, - alias: format!("{expr}").to_lowercase(), + alias, }); } } } ast::SelectItem::ExprWithAlias { expr, alias } => { + let alias = normalize_ident(alias); + if let Some(cp) = convert_cp_item(expr, alias.clone(), cp_scope.as_ref())? { + result.push(cp); + continue; + } let sql_expr = convert_expr(expr, &scope)?; result.push(Projection::Computed { expr: sql_expr, - alias: normalize_ident(alias), + alias, }); } // `expr AS (a, b)` binds one expression to several output names, diff --git a/nodedb-sql/src/planner/select/mod.rs b/nodedb-sql/src/planner/select/mod.rs index f40006b55..a27bff678 100644 --- a/nodedb-sql/src/planner/select/mod.rs +++ b/nodedb-sql/src/planner/select/mod.rs @@ -20,7 +20,7 @@ mod select_stmt; mod where_search; pub(crate) use cte_catalog::CteCatalog; -pub use entry::plan_query; +pub use entry::{plan_query, plan_statement_query}; pub use helpers::{ convert_projection, convert_where_to_filters, extract_float, extract_func_args, extract_string_literal, qualified_name, diff --git a/nodedb-sql/src/planner/select/post_process.rs b/nodedb-sql/src/planner/select/post_process.rs index fa3022587..662d440e6 100644 --- a/nodedb-sql/src/planner/select/post_process.rs +++ b/nodedb-sql/src/planner/select/post_process.rs @@ -129,7 +129,9 @@ fn projects_column(projection: &[Projection], table: Option<&str>, name: &str) - Projection::Column(projected) => { projected.eq_ignore_ascii_case(&qualified) || bare(projected).eq_ignore_ascii_case(name) } - Projection::Computed { alias, .. } => alias.eq_ignore_ascii_case(name), + Projection::Computed { alias, .. } | Projection::CpComputed { alias, .. } => { + alias.eq_ignore_ascii_case(name) + } Projection::Star | Projection::QualifiedStar(_) => true, }) } diff --git a/nodedb-sql/src/planner/select/select_stmt.rs b/nodedb-sql/src/planner/select/select_stmt.rs index 3ab107f9e..147226ef6 100644 --- a/nodedb-sql/src/planner/select/select_stmt.rs +++ b/nodedb-sql/src/planner/select/select_stmt.rs @@ -27,12 +27,17 @@ pub(in crate::planner::select) struct PlannedSelect { /// `tail` carries the enclosing query's ORDER BY / LIMIT so the base scan can /// be built with them in place — see [`QueryTail`] for why the engine rules /// need them before `plan_scan` runs. +/// +/// `statement_output` says whether this SELECT produces the statement's +/// result rows. Only such a SELECT list may hold a Control-Plane-computed +/// item; the flag lands on the scope, and every clause reads it from there. pub(super) fn plan_select( select: &Select, catalog: &dyn SqlCatalog, functions: &FunctionRegistry, temporal: TemporalScope, tail: &QueryTail<'_>, + statement_output: bool, ) -> Result { // 0. Intercept array table-valued functions before catalog resolution // so a name like `ARRAY_SLICE` is not looked up as a collection. @@ -56,12 +61,19 @@ pub(super) fn plan_select( // dropped non-LATERAL derived factors silently, the scope ended // up empty, and the planner errored with "multi-table FROM // without JOIN". - if let Some(planned) = try_plan_derived_from(select, catalog, functions, temporal, tail)? { + if let Some(planned) = + try_plan_derived_from(select, catalog, functions, temporal, tail, statement_output)? + { return Ok(planned); } // 1. Resolve FROM tables. let scope = TableScope::resolve_from(catalog, &select.from)?; + let scope = if statement_output { + scope.as_statement_output() + } else { + scope + }; // 2. Handle constant queries (no FROM clause): SELECT 1, SELECT 'hello', etc. if select.from.is_empty() { @@ -367,7 +379,10 @@ fn has_column_comparison(expr: &SqlExpr) -> bool { } /// Check if a SELECT has aggregation (GROUP BY or aggregate functions in projection). -fn has_aggregation(select: &Select, functions: &FunctionRegistry) -> bool { +pub(in crate::planner::select) fn has_aggregation( + select: &Select, + functions: &FunctionRegistry, +) -> bool { let group_by_non_empty = match &select.group_by { ast::GroupByExpr::All(_) => true, ast::GroupByExpr::Expressions(exprs, _) => !exprs.is_empty(), diff --git a/nodedb-sql/src/resolver/columns.rs b/nodedb-sql/src/resolver/columns.rs index cc5d7f18b..577abd45c 100644 --- a/nodedb-sql/src/resolver/columns.rs +++ b/nodedb-sql/src/resolver/columns.rs @@ -51,6 +51,19 @@ pub struct TableScope { /// `ON CONFLICT DO UPDATE` are both qualified-only, so a bare name in a /// WHEN or SET clause names the target column. qualified_only: Vec, + /// Whether the SELECT this scope backs produces the statement's result + /// rows. Only such a SELECT list may hold a Control-Plane-computed item: + /// the Control Plane evaluates the statement's output rows and nothing + /// deeper. A subquery, CTE, UNION branch, derived table, INSERT source, + /// or MERGE source keeps the default. A scope nested inside this one + /// never inherits the flag. + statement_output: bool, + /// Whether a sequence accessor resolves here although rows are in scope. + /// Set for a SELECT-list item over a relation: the Control Plane + /// evaluates that item once per output row. Every other row-scope clause + /// keeps the default and refuses the call. A scope nested inside this one + /// never inherits the flag. + cp_functions: bool, } impl TableScope { @@ -133,12 +146,40 @@ impl TableScope { } /// A copy of this scope nested inside `outer`, for planning a correlated - /// subquery body. + /// subquery body. The body's own clauses refuse a sequence accessor even + /// when the enclosing SELECT list allows one. pub fn nested_in(mut self, outer: TableScope) -> Self { self.outer = Some(Box::new(outer)); + self.statement_output = false; + self.cp_functions = false; self } + /// This scope marked as backing the SELECT that produces the statement's + /// result rows. + pub fn as_statement_output(mut self) -> Self { + self.statement_output = true; + self + } + + /// Whether the SELECT this scope backs produces the statement's result + /// rows. + pub fn is_statement_output(&self) -> bool { + self.statement_output + } + + /// This scope with sequence accessors allowed: the expression it resolves + /// is a SELECT-list item the Control Plane evaluates per output row. + pub fn allowing_cp_functions(mut self) -> Self { + self.cp_functions = true; + self + } + + /// Whether a sequence accessor resolves here despite rows being in scope. + pub fn allows_cp_functions(&self) -> bool { + self.cp_functions + } + /// A copy of this scope widened with output column names. /// /// An ORDER BY, GROUP BY, or HAVING identifier resolves against input diff --git a/nodedb-sql/src/resolver/expr/functions.rs b/nodedb-sql/src/resolver/expr/functions.rs index 10f0f6b58..2fb3ba866 100644 --- a/nodedb-sql/src/resolver/expr/functions.rs +++ b/nodedb-sql/src/resolver/expr/functions.rs @@ -86,13 +86,18 @@ pub(super) fn convert_function_depth( return Err(SqlError::UndefinedFunction { name }); } - // Per-row gate: a sequence accessor is evaluated at plan time, by - // `planner::catalog_expr_fold` for a FROM-less SELECT and by - // `planner::defaults` for a column DEFAULT. A call over a FROM relation - // reaches the row evaluator instead, which holds no sequence state and - // answers `NULL` for every row. Refuse it here so the statement fails - // loudly at plan time. - if scope.is_row_scope() && crate::functions::sequence_accessor::is_sequence_accessor(&name) { + // Per-row gate. A sequence accessor is evaluated by the planner for a + // FROM-less SELECT (`planner::catalog_expr_fold`) and a column DEFAULT + // (`planner::defaults`), and by the Control Plane once per output row + // for a SELECT-list item over a relation (the scope allows it there). + // Every other row-scope clause (WHERE, ORDER BY, GROUP BY, HAVING, JOIN + // ON, SET, aggregate and window arguments, subqueries) reaches the row + // evaluator, which holds no sequence state and answers `NULL` for every + // row. Refuse the call there so the statement fails at plan time. + if scope.is_row_scope() + && !scope.allows_cp_functions() + && crate::functions::sequence_accessor::is_sequence_accessor(&name) + { return Err(SqlError::SequencePerRowUnsupported { name }); } @@ -344,4 +349,24 @@ mod tests { ); } } + + /// A SELECT-list scope that allows Control-Plane functions resolves the + /// accessor over a relation; a scope nested inside it does not inherit + /// the allowance. + #[test] + fn a_cp_allowing_scope_resolves_the_accessor_and_a_nested_scope_refuses() { + let mut depth = 0; + let func = function_ast("SELECT nextval('s')"); + let allowed = + crate::resolver::columns::test_support::open_scope("t").allowing_cp_functions(); + assert!(allowed.is_row_scope()); + let expr = convert_function_depth(&func, &mut depth, &ColumnScope::Relations(&allowed)) + .expect("a SELECT-list accessor over a relation must resolve"); + assert!(matches!(expr, SqlExpr::Function { ref name, .. } if name == "nextval")); + + let nested = crate::resolver::columns::TableScope::new().nested_in(allowed); + let err = convert_function_depth(&func, &mut depth, &ColumnScope::Relations(&nested)) + .unwrap_err(); + assert!(matches!(err, SqlError::SequencePerRowUnsupported { .. })); + } } diff --git a/nodedb-sql/src/resolver/scope.rs b/nodedb-sql/src/resolver/scope.rs index e829dd79d..146fc28d0 100644 --- a/nodedb-sql/src/resolver/scope.rs +++ b/nodedb-sql/src/resolver/scope.rs @@ -42,4 +42,16 @@ impl ColumnScope<'_> { Self::Relations(scope) => scope.is_row_scope(), } } + + /// Whether a sequence accessor resolves here although rows are in scope. + /// + /// Only a SELECT-list scope over a relation says yes; the Control Plane + /// evaluates that item per output row. `Unchecked` never iterates rows, + /// so the question does not arise there. + pub fn allows_cp_functions(&self) -> bool { + match self { + Self::Unchecked => false, + Self::Relations(scope) => scope.allows_cp_functions(), + } + } } diff --git a/nodedb-sql/src/types/plan/cacheability.rs b/nodedb-sql/src/types/plan/cacheability.rs index 1c2b392c8..90fb79123 100644 --- a/nodedb-sql/src/types/plan/cacheability.rs +++ b/nodedb-sql/src/types/plan/cacheability.rs @@ -3,6 +3,7 @@ use crate::types::query::EngineType; use super::SqlPlan; +use super::expr_scan::projection_is_cp_computed; /// Whether a logical plan may be lowered once and reused from the physical-plan cache. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -44,6 +45,30 @@ impl SqlPlan { match self { Self::ConstantResult { volatile: true, .. } => DataDependent, + // A Control-Plane-computed projection entry holds a sequence + // accessor, which allocates per execution: the lowered tasks + // must not replay one execution's rows for another. + Self::Scan { projection, .. } + | Self::PointGet { projection, .. } + | Self::DocumentIndexLookup { projection, .. } + | Self::RangeScan { projection, .. } + | Self::Join { projection, .. } + | Self::TimeseriesScan { projection, .. } + | Self::VectorSearch { projection, .. } + | Self::MultiVectorSearch { projection, .. } + | Self::SparseSearch { projection, .. } + | Self::TextSearch { projection, .. } + | Self::HybridSearch { projection, .. } + | Self::HybridSearchTriple { projection, .. } + | Self::SpatialScan { projection, .. } + | Self::RecursiveScan { projection, .. } + | Self::Subquery { projection, .. } + | Self::LateralTopK { projection, .. } + | Self::LateralLoop { projection, .. } + if projection_is_cp_computed(projection) => + { + DataDependent + } Self::Insert { volatile_defaults: true, .. @@ -231,6 +256,55 @@ mod tests { ); } + #[test] + fn cp_computed_projection_is_data_dependent() { + use crate::types::query::Projection; + use crate::types_expr::SqlExpr; + let cp = Projection::CpComputed { + expr: SqlExpr::Function { + name: "nextval".into(), + args: vec![SqlExpr::Literal(SqlValue::String("s".into()))], + distinct: false, + }, + alias: "nextval".into(), + }; + let scan = |projection: Vec| SqlPlan::Scan { + collection: "docs".into(), + alias: None, + engine: EngineType::KeyValue, + filters: Vec::new(), + projection, + sort_keys: Vec::new(), + limit: None, + offset: 0, + distinct: false, + window_functions: Vec::new(), + temporal: crate::temporal::TemporalScope::default(), + }; + assert_eq!( + scan(vec![Projection::Column("id".into()), cp.clone()]).cache_eligibility(), + PlanCacheEligibility::DataDependent + ); + assert_eq!( + scan(vec![Projection::Column("id".into())]).cache_eligibility(), + PlanCacheEligibility::Cacheable + ); + let wrapped = SqlPlan::Subquery { + input: Box::new(scan(Vec::new())), + filters: Vec::new(), + projection: vec![cp], + window_functions: Vec::new(), + sort_keys: Vec::new(), + offset: 0, + distinct: false, + limit: None, + }; + assert_eq!( + wrapped.cache_eligibility(), + PlanCacheEligibility::DataDependent + ); + } + #[test] fn nested_point_dependency_propagates() { let plan = SqlPlan::Cte { diff --git a/nodedb-sql/src/types/plan/expr_scan.rs b/nodedb-sql/src/types/plan/expr_scan.rs new file mode 100644 index 000000000..249da2311 --- /dev/null +++ b/nodedb-sql/src/types/plan/expr_scan.rs @@ -0,0 +1,259 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Structural scans over plan expressions: sequence-accessor detection and +//! column-reference collection. +//! +//! A SELECT-list expression holding a sequence accessor is evaluated by the +//! Control Plane once per output row. The planner needs to know that an +//! expression holds one, and which base columns it reads, so the Data Plane +//! projects those columns and the response stage evaluates the expression. + +use crate::functions::sequence_accessor::is_sequence_accessor; +use crate::types::query::Projection; +use crate::types_expr::SqlExpr; + +/// The name of the first sequence accessor called in `expr`, when one is. +/// +/// A `SqlExpr::Subquery` is opaque here: its body resolves in its own scope, +/// which refuses an accessor of its own. +pub fn first_sequence_accessor(expr: &SqlExpr) -> Option<&str> { + match expr { + SqlExpr::Function { name, args, .. } => { + if is_sequence_accessor(name) { + return Some(name.as_str()); + } + args.iter().find_map(first_sequence_accessor) + } + SqlExpr::BinaryOp { left, right, .. } => { + first_sequence_accessor(left).or_else(|| first_sequence_accessor(right)) + } + SqlExpr::UnaryOp { expr, .. } + | SqlExpr::Cast { expr, .. } + | SqlExpr::IsNull { expr, .. } => first_sequence_accessor(expr), + SqlExpr::Case { + operand, + when_then, + else_expr, + } => operand + .as_deref() + .and_then(first_sequence_accessor) + .or_else(|| { + when_then.iter().find_map(|(when, then)| { + first_sequence_accessor(when).or_else(|| first_sequence_accessor(then)) + }) + }) + .or_else(|| else_expr.as_deref().and_then(first_sequence_accessor)), + SqlExpr::InList { expr, list, .. } => { + first_sequence_accessor(expr).or_else(|| list.iter().find_map(first_sequence_accessor)) + } + SqlExpr::Between { + expr, low, high, .. + } => first_sequence_accessor(expr) + .or_else(|| first_sequence_accessor(low)) + .or_else(|| first_sequence_accessor(high)), + SqlExpr::Like { expr, pattern, .. } => { + first_sequence_accessor(expr).or_else(|| first_sequence_accessor(pattern)) + } + SqlExpr::ArrayLiteral(items) => items.iter().find_map(first_sequence_accessor), + SqlExpr::Column { .. } | SqlExpr::Literal(_) | SqlExpr::Subquery(_) | SqlExpr::Wildcard => { + None + } + } +} + +/// Whether `expr` calls a sequence accessor anywhere outside a subquery. +pub fn contains_sequence_accessor(expr: &SqlExpr) -> bool { + first_sequence_accessor(expr).is_some() +} + +/// Every column `expr` references, qualified as `table.name` when the +/// reference carries a table, deduplicated, in first-seen order. +/// +/// A `SqlExpr::Subquery` contributes nothing: its columns belong to its own +/// relation set. +pub fn referenced_columns(expr: &SqlExpr) -> Vec { + let mut out = Vec::new(); + collect_columns(expr, &mut out); + out +} + +fn collect_columns(expr: &SqlExpr, out: &mut Vec) { + match expr { + SqlExpr::Column { table, name } => { + let qualified = match table { + Some(table) => format!("{table}.{name}"), + None => name.clone(), + }; + if !out.contains(&qualified) { + out.push(qualified); + } + } + SqlExpr::Function { args, .. } | SqlExpr::ArrayLiteral(args) => { + for arg in args { + collect_columns(arg, out); + } + } + SqlExpr::BinaryOp { left, right, .. } => { + collect_columns(left, out); + collect_columns(right, out); + } + SqlExpr::UnaryOp { expr, .. } + | SqlExpr::Cast { expr, .. } + | SqlExpr::IsNull { expr, .. } => collect_columns(expr, out), + SqlExpr::Case { + operand, + when_then, + else_expr, + } => { + if let Some(operand) = operand { + collect_columns(operand, out); + } + for (when, then) in when_then { + collect_columns(when, out); + collect_columns(then, out); + } + if let Some(else_expr) = else_expr { + collect_columns(else_expr, out); + } + } + SqlExpr::InList { expr, list, .. } => { + collect_columns(expr, out); + for item in list { + collect_columns(item, out); + } + } + SqlExpr::Between { + expr, low, high, .. + } => { + collect_columns(expr, out); + collect_columns(low, out); + collect_columns(high, out); + } + SqlExpr::Like { expr, pattern, .. } => { + collect_columns(expr, out); + collect_columns(pattern, out); + } + SqlExpr::Literal(_) | SqlExpr::Subquery(_) | SqlExpr::Wildcard => {} + } +} + +/// Whether `projection` holds an entry the Control Plane evaluates per row. +pub fn projection_is_cp_computed(projection: &[Projection]) -> bool { + projection + .iter() + .any(|p| matches!(p, Projection::CpComputed { .. })) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::types_expr::{BinaryOp, SqlValue}; + + fn column(table: Option<&str>, name: &str) -> SqlExpr { + SqlExpr::Column { + table: table.map(str::to_string), + name: name.into(), + } + } + + fn call(name: &str, args: Vec) -> SqlExpr { + SqlExpr::Function { + name: name.into(), + args, + distinct: false, + } + } + + fn nextval() -> SqlExpr { + call( + "nextval", + vec![SqlExpr::Literal(SqlValue::String("s".into()))], + ) + } + + #[test] + fn accessor_found_at_top_level_and_nested() { + assert_eq!(first_sequence_accessor(&nextval()), Some("nextval")); + let nested = SqlExpr::BinaryOp { + left: Box::new(SqlExpr::Literal(SqlValue::Int(2))), + op: BinaryOp::Mul, + right: Box::new(SqlExpr::Cast { + expr: Box::new(call("coalesce", vec![nextval()])), + to_type: "text".into(), + }), + }; + assert!(contains_sequence_accessor(&nested)); + let case = SqlExpr::Case { + operand: None, + when_then: vec![(column(None, "flag"), call("currval", Vec::new()))], + else_expr: None, + }; + assert_eq!(first_sequence_accessor(&case), Some("currval")); + } + + #[test] + fn plain_expressions_hold_no_accessor() { + let expr = SqlExpr::BinaryOp { + left: Box::new(column(Some("t"), "a")), + op: BinaryOp::Add, + right: Box::new(call("upper", vec![column(None, "b")])), + }; + assert!(!contains_sequence_accessor(&expr)); + assert!(!contains_sequence_accessor(&SqlExpr::Wildcard)); + } + + #[test] + fn subquery_is_opaque() { + let subquery = SqlExpr::Subquery(Box::new(crate::types::SqlPlan::ConstantResult { + columns: vec!["n".into()], + values: vec![SqlValue::Int(1)], + volatile: true, + })); + assert!(!contains_sequence_accessor(&subquery)); + assert!(referenced_columns(&subquery).is_empty()); + } + + #[test] + fn referenced_columns_are_qualified_deduplicated_and_ordered() { + let expr = SqlExpr::Case { + operand: Some(Box::new(column(Some("t"), "kind"))), + when_then: vec![( + SqlExpr::InList { + expr: Box::new(column(None, "b")), + list: vec![column(None, "a"), column(Some("t"), "kind")], + negated: false, + }, + SqlExpr::Between { + expr: Box::new(column(None, "c")), + low: Box::new(column(None, "a")), + high: Box::new(SqlExpr::Literal(SqlValue::Int(9))), + negated: false, + }, + )], + else_expr: Some(Box::new(SqlExpr::Like { + expr: Box::new(column(None, "d")), + pattern: Box::new(SqlExpr::Literal(SqlValue::String("x%".into()))), + negated: false, + case_insensitive: false, + })), + }; + assert_eq!(referenced_columns(&expr), ["t.kind", "b", "a", "c", "d"]); + } + + #[test] + fn cp_computed_projection_is_detected() { + let plain = vec![ + Projection::Column("id".into()), + Projection::Computed { + expr: nextval(), + alias: "n".into(), + }, + ]; + assert!(!projection_is_cp_computed(&plain)); + let cp = vec![Projection::CpComputed { + expr: nextval(), + alias: "n".into(), + }]; + assert!(projection_is_cp_computed(&cp)); + } +} diff --git a/nodedb-sql/src/types/plan/mod.rs b/nodedb-sql/src/types/plan/mod.rs index 371c9a13e..beddc1df7 100644 --- a/nodedb-sql/src/types/plan/mod.rs +++ b/nodedb-sql/src/types/plan/mod.rs @@ -3,6 +3,7 @@ //! SqlPlan intermediate representation and supporting types. mod cacheability; +mod expr_scan; mod merge_types; mod row_types; mod variant_name; @@ -11,6 +12,10 @@ mod vector_opts; mod volatility_scan; pub use cacheability::PlanCacheEligibility; +pub use expr_scan::{ + contains_sequence_accessor, first_sequence_accessor, projection_is_cp_computed, + referenced_columns, +}; pub use merge_types::{MergeClauseKind, MergePlanAction, MergePlanClause}; pub use row_types::{KvInsertIntent, VectorPrimaryRow, WriteRoute}; pub use variants::{DistanceMetric, SqlPlan}; diff --git a/nodedb-sql/src/types/query.rs b/nodedb-sql/src/types/query.rs index cf5db8743..a885942e6 100644 --- a/nodedb-sql/src/types/query.rs +++ b/nodedb-sql/src/types/query.rs @@ -66,6 +66,12 @@ pub enum Projection { QualifiedStar(String), /// Computed expression: `SELECT price * qty AS total` Computed { expr: SqlExpr, alias: String }, + /// Expression the Control Plane evaluates per output row after the Data + /// Plane returns the rows (a sequence accessor, possibly inside a larger + /// expression). The Data Plane projects the base columns the expression + /// references; the response stage evaluates `expr`, writes `alias`, and + /// drops those base columns. + CpComputed { expr: SqlExpr, alias: String }, } /// Sort key for ORDER BY. diff --git a/nodedb/src/control/planner/sql_plan_convert/aggregate/projection.rs b/nodedb/src/control/planner/sql_plan_convert/aggregate/projection.rs index bdb5152a5..c4594ebc5 100644 --- a/nodedb/src/control/planner/sql_plan_convert/aggregate/projection.rs +++ b/nodedb/src/control/planner/sql_plan_convert/aggregate/projection.rs @@ -2,50 +2,79 @@ //! Projection-name, computed-column, and window-function serialization helpers. +use nodedb_sql::types::plan::referenced_columns; use nodedb_sql::types::{Projection, SqlExpr, WindowSpec}; use nodedb_physical::physical_plan::JoinProjection; use super::super::expr::{sql_expr_to_bridge_expr, sql_expr_to_bridge_expr_qualified}; +/// The row keys a projection keeps. A Control-Plane-computed entry keeps the +/// base columns its expression reads: the Control Plane evaluates it after +/// the rows return, then drops those columns. pub(in crate::control::planner::sql_plan_convert) fn extract_projection_names( proj: &[Projection], window_functions: &[WindowSpec], ) -> Vec { - proj.iter() - .filter_map(|p| match p { - Projection::Column(name) => Some(name.clone()), + let mut names = Vec::with_capacity(proj.len()); + for p in proj { + match p { + Projection::Column(name) => names.push(name.clone()), Projection::Computed { alias, .. } if window_functions.iter().any(|spec| spec.alias == *alias) => { - Some(alias.clone()) + names.push(alias.clone()); } - _ => None, - }) - .collect() + Projection::CpComputed { expr, .. } => { + for column in referenced_columns(expr) { + if !names.contains(&column) { + names.push(column); + } + } + } + Projection::Computed { .. } | Projection::Star | Projection::QualifiedStar(_) => {} + } + } + names +} + +/// A pass-through join projection for one row key. +fn pass_through(name: &str) -> JoinProjection { + JoinProjection { + source: name.to_string(), + output: name.to_string(), + } } pub(in crate::control::planner::sql_plan_convert) fn extract_join_projection_specs( proj: &[Projection], ) -> Vec { - proj.iter() - .filter_map(|p| match p { - Projection::Column(name) => Some(JoinProjection { - source: name.clone(), - output: name.clone(), - }), + let mut specs = Vec::with_capacity(proj.len()); + for p in proj { + match p { + Projection::Column(name) => specs.push(pass_through(name)), Projection::Computed { expr: SqlExpr::Column { table, name }, alias, - } => Some(JoinProjection { + } => specs.push(JoinProjection { source: table .as_deref() .map_or_else(|| name.clone(), |table| format!("{table}.{name}")), output: alias.clone(), }), - _ => None, - }) - .collect() + // The Control Plane evaluates the entry over the joined row, so + // every base column it reads passes through under its own name. + Projection::CpComputed { expr, .. } => { + for column in referenced_columns(expr) { + if !specs.iter().any(|spec| spec.source == column) { + specs.push(pass_through(&column)); + } + } + } + Projection::Computed { .. } | Projection::Star | Projection::QualifiedStar(_) => {} + } + } + specs } pub(in crate::control::planner::sql_plan_convert) fn serialize_join_computed_projection( @@ -70,25 +99,39 @@ pub(in crate::control::planner::sql_plan_convert) fn serialize_join_computed_pro }); } - let computed = proj - .iter() - .map(|item| match item { - Projection::Column(name) => Some(crate::bridge::expr_eval::ComputedColumn { + let mut computed = Vec::with_capacity(proj.len()); + for item in proj { + match item { + Projection::Column(name) => computed.push(crate::bridge::expr_eval::ComputedColumn { alias: name.clone(), expr: crate::bridge::expr_eval::SqlExpr::Column(name.clone()), }), Projection::Computed { expr, alias } => { - Some(crate::bridge::expr_eval::ComputedColumn { + computed.push(crate::bridge::expr_eval::ComputedColumn { alias: alias.clone(), expr: sql_expr_to_bridge_expr_qualified(expr), - }) + }); } - Projection::Star | Projection::QualifiedStar(_) => None, - }) - .collect::>>() - .ok_or_else(|| crate::Error::BadRequest { - detail: "wildcard join projection reached computed-expression lowering".into(), - })?; + // Nothing is computed on the Data Plane for a Control-Plane + // entry; its base columns pass through under their own names. + Projection::CpComputed { expr, .. } => { + for column in referenced_columns(expr) { + if computed.iter().any(|c| c.alias == column) { + continue; + } + computed.push(crate::bridge::expr_eval::ComputedColumn { + alias: column.clone(), + expr: crate::bridge::expr_eval::SqlExpr::Column(column), + }); + } + } + Projection::Star | Projection::QualifiedStar(_) => { + return Err(crate::Error::BadRequest { + detail: "wildcard join projection reached computed-expression lowering".into(), + }); + } + } + } encode_computed_columns(computed, "join computed projection") } @@ -158,7 +201,13 @@ pub(in crate::control::planner::sql_plan_convert) fn extract_computed_columns( expr: convert(expr), }) } - _ => None, + // A Control-Plane-computed entry never reaches the Data Plane + // evaluator; a window alias is served by its window spec. + Projection::Computed { .. } + | Projection::CpComputed { .. } + | Projection::Column(_) + | Projection::Star + | Projection::QualifiedStar(_) => None, }) .collect(); if computed.is_empty() { diff --git a/nodedb/src/control/planner/sql_plan_convert/expr/inline_cte.rs b/nodedb/src/control/planner/sql_plan_convert/expr/inline_cte.rs index bcd334df5..91cd73102 100644 --- a/nodedb/src/control/planner/sql_plan_convert/expr/inline_cte.rs +++ b/nodedb/src/control/planner/sql_plan_convert/expr/inline_cte.rs @@ -157,11 +157,16 @@ pub(in crate::control::planner::sql_plan_convert) fn inline_cte( } /// `true` if any projection entry is a computed expression (`price * qty AS -/// total`) rather than a bare column or star. +/// total`) rather than a bare column or star. A Control-Plane-computed entry +/// counts too: merging it into a scan body would drop the column the +/// Control Plane evaluates. fn has_computed_projection(projection: &[Projection]) -> bool { - projection - .iter() - .any(|p| matches!(p, Projection::Computed { .. })) + projection.iter().any(|p| { + matches!( + p, + Projection::Computed { .. } | Projection::CpComputed { .. } + ) + }) } /// The outer constraints carried on a `Scan` that references the CTE by diff --git a/nodedb/src/control/planner/sql_plan_convert/group_key_name.rs b/nodedb/src/control/planner/sql_plan_convert/group_key_name.rs index aef5e0f5a..63072643c 100644 --- a/nodedb/src/control/planner/sql_plan_convert/group_key_name.rs +++ b/nodedb/src/control/planner/sql_plan_convert/group_key_name.rs @@ -12,6 +12,10 @@ //! purely internal executor↔shaper handshake. The alias only ever reaches the //! shaper's `display_name`, derived separately from `SqlPlan::Aggregate`'s //! `group_by_aliases`. +//! +//! The rule itself lives in `nodedb_sql::planner::agg_naming`, because the +//! planner projects these names when it wraps an aggregate in a +//! Control-Plane-computed projection. Both sides call the same function. use nodedb_sql::types_expr::SqlExpr; @@ -21,7 +25,7 @@ use nodedb_sql::types_expr::SqlExpr; pub(in crate::control::planner::sql_plan_convert) fn computed_group_key_name( index: usize, ) -> String { - format!("group_{index}") + nodedb_sql::planner::agg_naming::computed_group_key_name(index) } /// The output/lookup name for one GROUP BY key. A bare column keeps its own @@ -31,8 +35,5 @@ pub(in crate::control::planner::sql_plan_convert) fn group_key_output_name( expr: &SqlExpr, index: usize, ) -> String { - match expr { - SqlExpr::Column { name, .. } => name.clone(), - _ => computed_group_key_name(index), - } + nodedb_sql::planner::agg_naming::group_key_row_name(expr, index) } diff --git a/nodedb/src/control/planner/sql_plan_convert/lateral.rs b/nodedb/src/control/planner/sql_plan_convert/lateral.rs index cda271938..4d5508663 100644 --- a/nodedb/src/control/planner/sql_plan_convert/lateral.rs +++ b/nodedb/src/control/planner/sql_plan_convert/lateral.rs @@ -187,18 +187,29 @@ fn sort_keys_to_spec(keys: &[SortKey]) -> Vec { /// Convert `Projection` list to `JoinProjection` list. fn projection_to_join_projections(projection: &[Projection]) -> Vec { - projection - .iter() - .filter_map(|p| match p { - Projection::Column(name) => Some(JoinProjection { - source: name.clone(), - output: name.clone(), - }), - Projection::Computed { alias, .. } => Some(JoinProjection { - source: alias.clone(), - output: alias.clone(), - }), - Projection::Star | Projection::QualifiedStar(_) => None, - }) - .collect() + let pass_through = |name: &str| JoinProjection { + source: name.to_string(), + output: name.to_string(), + }; + let mut specs = Vec::with_capacity(projection.len()); + for p in projection { + match p { + Projection::Column(name) => specs.push(pass_through(name)), + Projection::Computed { alias, .. } => specs.push(pass_through(alias)), + // The Control Plane evaluates the entry over the lateral row, so + // every base column it reads passes through under its own name. + Projection::CpComputed { expr, .. } => { + for column in nodedb_sql::types::plan::referenced_columns(expr) { + if !specs + .iter() + .any(|spec: &JoinProjection| spec.source == column) + { + specs.push(pass_through(&column)); + } + } + } + Projection::Star | Projection::QualifiedStar(_) => {} + } + } + specs } diff --git a/nodedb/src/control/planner/sql_plan_convert/output_schema/columns.rs b/nodedb/src/control/planner/sql_plan_convert/output_schema/columns.rs index 78cea8a45..2b0eb8c92 100644 --- a/nodedb/src/control/planner/sql_plan_convert/output_schema/columns.rs +++ b/nodedb/src/control/planner/sql_plan_convert/output_schema/columns.rs @@ -80,6 +80,37 @@ pub(super) fn projection_to_column( ty: infer_computed_expr_type(expr, types), }) } + // The Control Plane writes the evaluated value under the alias. A + // bare sequence accessor yields a bigint; any wider expression is + // typed like a computed entry. + Projection::CpComputed { expr, alias } => { + let ty = match expr { + SqlExpr::Function { name, .. } + if nodedb_sql::functions::sequence_accessor::is_sequence_accessor(name) => + { + DdlColType::Int8 + } + SqlExpr::Function { .. } + | SqlExpr::Column { .. } + | SqlExpr::Literal(_) + | SqlExpr::BinaryOp { .. } + | SqlExpr::UnaryOp { .. } + | SqlExpr::Case { .. } + | SqlExpr::Cast { .. } + | SqlExpr::Subquery(_) + | SqlExpr::Wildcard + | SqlExpr::IsNull { .. } + | SqlExpr::InList { .. } + | SqlExpr::Between { .. } + | SqlExpr::Like { .. } + | SqlExpr::ArrayLiteral(_) => infer_computed_expr_type(expr, types), + }; + Some(OutputColumn { + display_name: alias.clone(), + lookup_key: alias.clone(), + ty, + }) + } Projection::Star | Projection::QualifiedStar(_) => None, } } @@ -264,6 +295,38 @@ mod tests { assert_eq!(col.ty, DdlColType::Text); } + #[test] + fn cp_computed_accessor_is_int8_under_its_alias() { + let types = HashMap::new(); + let nextval = SqlExpr::Function { + name: "nextval".to_string(), + args: vec![SqlExpr::Literal(nodedb_sql::types_expr::SqlValue::String( + "s".to_string(), + ))], + distinct: false, + }; + let bare = Projection::CpComputed { + expr: nextval.clone(), + alias: "nextval".to_string(), + }; + let col = projection_to_column(&bare, &types).expect("Some for CpComputed"); + assert_eq!(col.display_name, "nextval"); + assert_eq!(col.lookup_key, "nextval"); + assert_eq!(col.ty, DdlColType::Int8); + + let wider = Projection::CpComputed { + expr: SqlExpr::BinaryOp { + left: Box::new(nextval), + op: nodedb_sql::types_expr::BinaryOp::Gt, + right: Box::new(SqlExpr::Literal(nodedb_sql::types_expr::SqlValue::Int(0))), + }, + alias: "positive".to_string(), + }; + let col = projection_to_column(&wider, &types).expect("Some for CpComputed"); + assert_eq!(col.lookup_key, "positive"); + assert_eq!(col.ty, DdlColType::Bool); + } + #[test] fn star_returns_none() { let types = HashMap::new(); diff --git a/nodedb/src/control/planner/sql_plan_convert/set_ops.rs b/nodedb/src/control/planner/sql_plan_convert/set_ops.rs index d8beeea04..1fb9530f3 100644 --- a/nodedb/src/control/planner/sql_plan_convert/set_ops.rs +++ b/nodedb/src/control/planner/sql_plan_convert/set_ops.rs @@ -375,6 +375,10 @@ fn is_merged_doc_body(plan: &PhysicalPlan) -> bool { /// which is the same key the response shaper reads it back by. Every window /// alias is kept too, so a window output the SELECT list does not repeat as a /// computed entry survives the column pruning. +/// +/// A Control-Plane-computed item keeps the base columns its expression reads, +/// not its alias: the Control Plane evaluates it once the tail returns, then +/// drops those columns. fn lower_subquery_projection( projection: &[Projection], window_functions: &[WindowSpec], @@ -387,6 +391,14 @@ fn lower_subquery_projection( } Projection::Star | Projection::QualifiedStar(_) => return Ok(Vec::new()), Projection::Computed { alias, .. } => names.push(alias.clone()), + Projection::CpComputed { expr, .. } => { + for column in nodedb_sql::types::plan::referenced_columns(expr) { + let bare = column.rsplit('.').next().unwrap_or(&column).to_string(); + if !names.contains(&bare) { + names.push(bare); + } + } + } } } for spec in window_functions { diff --git a/nodedb/tests/wire/cases/mod.rs b/nodedb/tests/wire/cases/mod.rs index 257088f8d..92a757c3d 100644 --- a/nodedb/tests/wire/cases/mod.rs +++ b/nodedb/tests/wire/cases/mod.rs @@ -213,6 +213,8 @@ mod sql_rls_predicate_parse; mod sql_schedules; mod sql_search_subquery_composition; mod sql_security_e2e; +mod sql_sequence_row_scope; +mod sql_sequence_row_scope_refusals; mod sql_sequences; mod sql_spatial_index_ddl; mod sql_subquery_from; diff --git a/nodedb/tests/wire/cases/sql_sequence_row_scope.rs b/nodedb/tests/wire/cases/sql_sequence_row_scope.rs new file mode 100644 index 000000000..793b10aca --- /dev/null +++ b/nodedb/tests/wire/cases/sql_sequence_row_scope.rs @@ -0,0 +1,401 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-row evaluation of sequence accessors (`nextval`/`currval`/`setval`) in +//! the SELECT list of a top-level SELECT over a relation. +//! +//! The rule: a sequence accessor in the SELECT list of a top-level SELECT +//! over a relation evaluates once per output row, in output order, on the +//! Control Plane after the rows are final. Every other row-scope clause +//! (ORDER BY, GROUP BY, HAVING, JOIN ON, window, aggregate argument, a +//! nested subquery, UPDATE SET, INSERT ... SELECT source) still refuses with +//! `0A000` — see `sql_sequence_row_scope_refusals.rs`. + +use crate::harness::TestServer; + +/// Create sequence `s` and a kv collection `t` (`id BIGINT PRIMARY KEY, v +/// TEXT`) with rows `(1,'a'), (2,'b'), (3,'c')`. +async fn seed(server: &TestServer) { + server.exec("CREATE SEQUENCE s").await.unwrap(); + server + .exec("CREATE COLLECTION t (id BIGINT PRIMARY KEY, v TEXT) WITH (engine = 'kv')") + .await + .unwrap(); + server + .exec("INSERT INTO t (id, v) VALUES (1, 'a'), (2, 'b'), (3, 'c')") + .await + .unwrap(); +} + +/// `nextval` in the SELECT list of a scan advances once per output row, in +/// `ORDER BY` order, then the session continues from the last row's value. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn nextval_in_a_select_list_advances_once_per_output_row_in_output_order() { + let server = TestServer::start().await; + seed(&server).await; + + let rows = server + .query_rows("SELECT id, nextval('s') AS n FROM t ORDER BY id") + .await + .unwrap(); + assert_eq!( + rows, + vec![ + vec!["1".to_string(), "1".to_string()], + vec!["2".to_string(), "2".to_string()], + vec!["3".to_string(), "3".to_string()], + ] + ); + + let next = server.query_text("SELECT nextval('s')").await.unwrap(); + assert_eq!(next, vec!["4".to_string()]); +} + +/// `LIMIT` truncates the row stream before evaluation, so only the emitted +/// rows consume an allocation. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn nextval_with_limit_consumes_exactly_the_emitted_rows() { + let server = TestServer::start().await; + seed(&server).await; + + let rows = server + .query_text("SELECT nextval('s') AS n FROM t ORDER BY id LIMIT 2") + .await + .unwrap(); + assert_eq!(rows, vec!["1".to_string(), "2".to_string()]); + + let next = server.query_text("SELECT nextval('s')").await.unwrap(); + assert_eq!(next, vec!["3".to_string()]); +} + +/// The accessor's result feeds an arithmetic expression, one evaluation per +/// row. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn nextval_inside_an_arithmetic_expression_evaluates_per_row() { + let server = TestServer::start().await; + seed(&server).await; + + let rows = server + .query_rows("SELECT id, nextval('s') * 10 AS n FROM t ORDER BY id") + .await + .unwrap(); + let n: Vec<&str> = rows.iter().map(|r| r[1].as_str()).collect(); + assert_eq!(n, vec!["10", "20", "30"]); +} + +/// `nextval` referencing a row column sees that row, and the projection +/// carries exactly one output column (`n`) — no leaked pass-through `id`. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn nextval_referencing_a_row_column_sees_that_row() { + let server = TestServer::start().await; + seed(&server).await; + + let rows = server + .query_named_rows("SELECT nextval('s') + id AS n FROM t ORDER BY id") + .await + .unwrap(); + assert_eq!(rows.len(), 3, "three rows expected: {rows:?}"); + for (row, expected) in rows.iter().zip(["2", "4", "6"]) { + assert_eq!(row.len(), 1, "exactly one column expected: {row:?}"); + assert_eq!(row.get("n").map(String::as_str), Some(expected)); + } +} + +/// `currval` inside the same row as a preceding `nextval` reads that row's +/// value, not a stale session-wide one. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn currval_after_nextval_in_the_same_row_reads_that_rows_value() { + let server = TestServer::start().await; + seed(&server).await; + + let rows = server + .query_rows("SELECT nextval('s') AS a, currval('s') AS b FROM t ORDER BY id") + .await + .unwrap(); + assert_eq!( + rows, + vec![ + vec!["1".to_string(), "1".to_string()], + vec!["2".to_string(), "2".to_string()], + vec!["3".to_string(), "3".to_string()], + ] + ); +} + +/// `setval` in the SELECT list positions the sequence per row, and the +/// session continues from the last row's position. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn setval_in_a_select_list_positions_per_row() { + let server = TestServer::start().await; + seed(&server).await; + + let rows = server + .query_text("SELECT setval('s', id * 100) AS v FROM t ORDER BY id") + .await + .unwrap(); + assert_eq!( + rows, + vec!["100".to_string(), "200".to_string(), "300".to_string()] + ); + + let next = server.query_text("SELECT nextval('s')").await.unwrap(); + assert_eq!(next, vec!["301".to_string()]); +} + +/// A SELECT list consisting only of the accessor emits exactly that column, +/// one row per source row. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_bare_accessor_projection_emits_only_its_column() { + let server = TestServer::start().await; + seed(&server).await; + + let rows = server + .query_rows("SELECT nextval('s') FROM t") + .await + .unwrap(); + assert_eq!(rows.len(), 3, "three rows expected: {rows:?}"); + for row in &rows { + assert_eq!(row.len(), 1, "exactly one column expected: {row:?}"); + } +} + +/// A derived table (subquery in FROM) still evaluates the accessor once per +/// outer output row. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn nextval_over_a_derived_table_evaluates_per_outer_row() { + let server = TestServer::start().await; + seed(&server).await; + + let rows = server + .query_rows("SELECT x, nextval('s') AS n FROM (SELECT id AS x FROM t) AS d ORDER BY x") + .await + .unwrap(); + assert_eq!( + rows, + vec![ + vec!["1".to_string(), "1".to_string()], + vec!["2".to_string(), "2".to_string()], + vec!["3".to_string(), "3".to_string()], + ] + ); +} + +/// A `GROUP BY` query evaluates the accessor once per emitted group row, in +/// output order, after grouping has collapsed the input. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn nextval_over_a_grouped_query_evaluates_per_group_row() { + let server = TestServer::start().await; + server.exec("CREATE SEQUENCE s").await.unwrap(); + server + .exec("CREATE COLLECTION t (id BIGINT PRIMARY KEY, grp TEXT) WITH (engine = 'kv')") + .await + .unwrap(); + server + .exec("INSERT INTO t (id, grp) VALUES (1, 'a'), (2, 'a'), (3, 'b')") + .await + .unwrap(); + + let rows = server + .query_rows("SELECT grp, COUNT(*) AS c, nextval('s') AS n FROM t GROUP BY grp ORDER BY grp") + .await + .unwrap(); + assert_eq!( + rows, + vec![ + vec!["a".to_string(), "2".to_string(), "1".to_string()], + vec!["b".to_string(), "1".to_string(), "2".to_string()], + ] + ); +} + +/// The DDL shape per engine, copied from `sql_sequences.rs`'s per-engine +/// tests: `(id BIGINT PRIMARY KEY, v TEXT)` for the document and kv engines, +/// and `COLUMNS (id BIGINT, v TEXT)` (no PRIMARY KEY) for columnar. +fn create_table_sql(name: &str, engine: &str) -> String { + if engine == "columnar" { + format!("CREATE COLLECTION {name} COLUMNS (id BIGINT, v TEXT) WITH (engine='columnar')") + } else { + format!("CREATE COLLECTION {name} (id BIGINT PRIMARY KEY, v TEXT) WITH (engine='{engine}')") + } +} + +/// `nextval` in the SELECT list evaluates on every engine, yielding 3 +/// distinct increasing values in `ORDER BY id` order. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn nextval_in_a_select_list_evaluates_on_every_engine() { + let server = TestServer::start().await; + + for engine in ["document_schemaless", "document_strict", "kv", "columnar"] { + let seq = format!("seq_engine_{engine}"); + let table = format!("t_engine_{engine}"); + + server + .exec(&format!("CREATE SEQUENCE {seq}")) + .await + .unwrap(); + server + .exec(&create_table_sql(&table, engine)) + .await + .unwrap(); + server + .exec(&format!( + "INSERT INTO {table} (id, v) VALUES (1, 'a'), (2, 'b'), (3, 'c')" + )) + .await + .unwrap(); + + let rows = server + .query_text(&format!( + "SELECT nextval('{seq}') AS n FROM {table} ORDER BY id" + )) + .await + .unwrap_or_else(|e| panic!("engine {engine}: {e}")); + + assert_eq!( + rows.len(), + 3, + "engine {engine}: three rows expected: {rows:?}" + ); + let values: Vec = rows + .iter() + .map(|v| { + v.parse() + .unwrap_or_else(|_| panic!("engine {engine}: non-numeric {v}")) + }) + .collect(); + assert_eq!( + values, + vec![values[0], values[0] + 1, values[0] + 2], + "engine {engine}: values must be distinct and increasing, got {values:?}" + ); + } +} + +/// A 500-row scan keeps `n == id` for every row: streaming evaluation must +/// not reorder or skip a row. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn nextval_over_a_large_result_stays_ordered() { + let server = TestServer::start().await; + server.exec("CREATE SEQUENCE s").await.unwrap(); + server + .exec("CREATE COLLECTION t (id BIGINT PRIMARY KEY, v TEXT) WITH (engine = 'kv')") + .await + .unwrap(); + + let values: Vec = (1..=500).map(|i| format!("({i}, 'v{i}')")).collect(); + server + .exec(&format!( + "INSERT INTO t (id, v) VALUES {}", + values.join(", ") + )) + .await + .unwrap(); + + let rows = server + .query_rows("SELECT id, nextval('s') AS n FROM t ORDER BY id") + .await + .unwrap(); + assert_eq!(rows.len(), 500, "500 rows expected: got {}", rows.len()); + for row in &rows { + assert_eq!( + row[0], row[1], + "n must equal id for every row, got id={} n={}", + row[0], row[1] + ); + } +} + +/// `EXPLAIN` plans the statement without executing it, so a per-row accessor +/// inside it must not advance the sequence. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn explain_of_a_per_row_accessor_does_not_advance() { + let server = TestServer::start().await; + seed(&server).await; + + server + .exec("EXPLAIN SELECT nextval('s') FROM t") + .await + .unwrap(); + + let next = server.query_text("SELECT nextval('s')").await.unwrap(); + assert_eq!(next, vec!["1".to_string()]); +} + +/// A statement that fails partway through (division by zero on a later row) +/// leaves no partial result and no partial allocation footprint visible to +/// the caller — the whole statement errors. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_failing_statement_does_not_leave_partial_rows() { + let server = TestServer::start().await; + seed(&server).await; + + server + .expect_error( + "SELECT nextval('s') AS n, 1 / (id - 2) AS d FROM t ORDER BY id", + "22012", + ) + .await; +} + +/// `INSERT ... RETURNING` evaluates the accessor once per inserted row, in +/// insertion order. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn insert_returning_nextval_evaluates_per_inserted_row() { + let server = TestServer::start().await; + seed(&server).await; + + let rows = server + .query_rows( + "INSERT INTO t (id, v) VALUES (10, 'x'), (11, 'y') RETURNING id, nextval('s') AS n", + ) + .await + .unwrap(); + assert_eq!( + rows, + vec![ + vec!["10".to_string(), "1".to_string()], + vec!["11".to_string(), "2".to_string()], + ] + ); +} + +/// A plain arithmetic `RETURNING` expression evaluates against the inserted +/// row, same as any other per-row `RETURNING` expression. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn insert_returning_an_arithmetic_expression_evaluates() { + let server = TestServer::start().await; + seed(&server).await; + + let rows = server + .query_text("INSERT INTO t (id, v) VALUES (12, 'z') RETURNING id * 2 AS d") + .await + .unwrap(); + assert_eq!(rows, vec!["24".to_string()]); +} + +/// `UPDATE ... RETURNING` evaluates the accessor once per updated row. +/// `RETURNING` carries no `ORDER BY`, so the assertion checks the set of +/// `(id, n)` pairs rather than a fixed row order: every id appears exactly +/// once, and `n` takes each of `1..=3` exactly once. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn update_returning_nextval_evaluates_per_updated_row() { + let server = TestServer::start().await; + seed(&server).await; + + let rows = server + .query_rows("UPDATE t SET v = 'touched' RETURNING id, nextval('s') AS n") + .await + .unwrap(); + assert_eq!(rows.len(), 3, "three updated rows expected: {rows:?}"); + + let mut ids: Vec = rows.iter().map(|r| r[0].parse().unwrap()).collect(); + ids.sort_unstable(); + assert_eq!(ids, vec![1, 2, 3], "every id must appear once: {rows:?}"); + + let mut ns: Vec = rows.iter().map(|r| r[1].parse().unwrap()).collect(); + ns.sort_unstable(); + assert_eq!( + ns, + vec![1, 2, 3], + "n must take each allocation exactly once: {rows:?}" + ); +} diff --git a/nodedb/tests/wire/cases/sql_sequence_row_scope_refusals.rs b/nodedb/tests/wire/cases/sql_sequence_row_scope_refusals.rs new file mode 100644 index 000000000..8406b6a7b --- /dev/null +++ b/nodedb/tests/wire/cases/sql_sequence_row_scope_refusals.rs @@ -0,0 +1,142 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Refusals that survive the per-row SELECT-list evaluation rule. +//! +//! Only the SELECT list of a top-level SELECT over a relation evaluates a +//! sequence accessor per row (see `sql_sequence_row_scope.rs`). Every other +//! row-scope clause — ORDER BY, GROUP BY, HAVING, JOIN ON, an aggregate +//! argument, a window function, a nested subquery, UPDATE SET, and the +//! source of `INSERT ... SELECT` — still refuses with `0A000` +//! (feature_not_supported). + +use crate::harness::TestServer; + +/// Create sequence `s` and a kv collection `t` (`id BIGINT PRIMARY KEY, v +/// TEXT`) with rows `(1,'a'), (2,'b'), (3,'c')`. +async fn seed(server: &TestServer) { + server.exec("CREATE SEQUENCE s").await.unwrap(); + server + .exec("CREATE COLLECTION t (id BIGINT PRIMARY KEY, v TEXT) WITH (engine = 'kv')") + .await + .unwrap(); + server + .exec("INSERT INTO t (id, v) VALUES (1, 'a'), (2, 'b'), (3, 'c')") + .await + .unwrap(); +} + +/// `ORDER BY` sorts before the SELECT list is materialized, so the accessor +/// there would decide row order from side-effecting state. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn nextval_in_order_by_is_refused() { + let server = TestServer::start().await; + seed(&server).await; + + server + .expect_error("SELECT id FROM t ORDER BY nextval('s')", "0A000") + .await; +} + +/// `GROUP BY` groups before any per-output-row evaluation exists. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn nextval_in_group_by_is_refused() { + let server = TestServer::start().await; + seed(&server).await; + + server + .expect_error("SELECT COUNT(*) FROM t GROUP BY nextval('s')", "0A000") + .await; +} + +/// `HAVING` filters groups before the SELECT list runs. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn nextval_in_having_is_refused() { + let server = TestServer::start().await; + seed(&server).await; + server + .exec("CREATE COLLECTION g (id BIGINT PRIMARY KEY, grp TEXT) WITH (engine = 'kv')") + .await + .unwrap(); + server + .exec("INSERT INTO g (id, grp) VALUES (1, 'a'), (2, 'a'), (3, 'b')") + .await + .unwrap(); + + server + .expect_error( + "SELECT grp FROM g GROUP BY grp HAVING nextval('s') > 0", + "0A000", + ) + .await; +} + +/// A `JOIN ON` predicate decides which rows exist before the SELECT list +/// runs over them. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn nextval_in_join_on_is_refused() { + let server = TestServer::start().await; + seed(&server).await; + + server + .expect_error( + "SELECT a.id FROM t a JOIN t b ON a.id = nextval('s')", + "0A000", + ) + .await; +} + +/// `UPDATE ... SET` evaluates its expression once per matched row inside +/// the write path, not the read-side per-row evaluator the SELECT-list rule +/// covers. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn nextval_in_update_set_is_refused() { + let server = TestServer::start().await; + seed(&server).await; + + server + .expect_error("UPDATE t SET v = nextval('s')", "0A000") + .await; +} + +/// An aggregate's argument is evaluated inside the aggregation step, before +/// any output row exists to attribute the accessor call to. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn nextval_as_an_aggregate_argument_is_refused() { + let server = TestServer::start().await; + seed(&server).await; + + server + .expect_error("SELECT SUM(nextval('s')) FROM t", "0A000") + .await; +} + +/// A window function's `ORDER BY` runs before the SELECT list, the same +/// reason the top-level `ORDER BY` refuses. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn nextval_in_a_window_order_by_is_refused() { + let server = TestServer::start().await; + seed(&server).await; + + server + .expect_error( + "SELECT ROW_NUMBER() OVER (ORDER BY nextval('s')) FROM t", + "0A000", + ) + .await; +} + +/// `INSERT ... SELECT` accessors live in the source SELECT's list feeding a +/// write target, not a plain top-level SELECT's output. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn nextval_as_an_insert_select_source_is_refused() { + let server = TestServer::start().await; + seed(&server).await; + server + .exec("CREATE COLLECTION t2 (id BIGINT PRIMARY KEY) WITH (engine = 'kv')") + .await + .unwrap(); + + server + .expect_error("INSERT INTO t2 (id) SELECT nextval('s') FROM t", "0A000") + .await; +} diff --git a/nodedb/tests/wire/cases/sql_sequences.rs b/nodedb/tests/wire/cases/sql_sequences.rs index 68d27a6b0..725bcf9bb 100644 --- a/nodedb/tests/wire/cases/sql_sequences.rs +++ b/nodedb/tests/wire/cases/sql_sequences.rs @@ -330,57 +330,6 @@ async fn two_row_collection_with_sequence(server: &TestServer, label: &str) -> ( (sequence, collection) } -/// `nextval` in a SELECT list over a FROM relation raises `0A000` -/// (feature_not_supported). Per-row allocation is not implemented, and a -/// silent NULL for every row is worse than the refusal. -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn nextval_in_a_select_list_over_a_scan_is_refused() { - let server = TestServer::start().await; - let (sequence, collection) = two_row_collection_with_sequence(&server, "nextval").await; - - server - .expect_error( - &format!("SELECT nextval('{sequence}') FROM {collection}"), - "0A000", - ) - .await; -} - -/// `currval` in a SELECT list over a FROM relation raises `0A000`, the same -/// as `nextval`: both read session sequence state the row evaluator lacks. -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn currval_in_a_select_list_over_a_scan_is_refused() { - let server = TestServer::start().await; - let (sequence, collection) = two_row_collection_with_sequence(&server, "currval").await; - - server - .query_text(&format!("SELECT nextval('{sequence}')")) - .await - .unwrap(); - server - .expect_error( - &format!("SELECT currval('{sequence}') FROM {collection}"), - "0A000", - ) - .await; -} - -/// A projection mixing a plain column with a sequence accessor is refused as -/// a whole. Returning the column and a NULL beside it would ship the silent -/// cell this refusal exists to prevent. -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn a_projection_mixing_a_column_and_an_accessor_is_refused() { - let server = TestServer::start().await; - let (sequence, collection) = two_row_collection_with_sequence(&server, "mixed").await; - - server - .expect_error( - &format!("SELECT id, nextval('{sequence}') FROM {collection}"), - "0A000", - ) - .await; -} - /// A sequence accessor in a WHERE clause over a FROM relation is refused for /// the same reason as one in the SELECT list. #[tokio::test(flavor = "multi_thread", worker_threads = 4)] @@ -396,37 +345,6 @@ async fn a_sequence_accessor_in_a_where_clause_is_refused() { .await; } -/// The refusal yields no result set at all, so no row can carry an empty or -/// NULL cell, and the statement allocates nothing from the sequence. -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn a_refused_accessor_produces_no_row_and_no_allocation() { - let server = TestServer::start().await; - let (sequence, collection) = two_row_collection_with_sequence(&server, "guard").await; - - let result = server - .query_text(&format!("SELECT nextval('{sequence}') FROM {collection}")) - .await; - let message = match result { - Ok(rows) => panic!("per-row nextval must not return rows, got {rows:?}"), - Err(message) => message, - }; - assert!( - message.contains("0A000"), - "refusal must carry SQLSTATE 0A000, got: {message}" - ); - - // A refused statement consumes nothing, so the first real allocation is 1. - let first = server - .query_text(&format!("SELECT nextval('{sequence}')")) - .await - .unwrap(); - assert_eq!( - first, - vec!["1".to_string()], - "the refused statement must not have allocated, got {first:?}" - ); -} - /// A from-less projection legally repeats an output name /// (`SELECT nextval('s'), nextval('s')`). The constant row is a single JSON /// object, so keying both cells by the display name collapsed them to the From 39077bda6170585d0ff50c5f07f11e40a133a2f7 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 18 Sep 2026 08:34:45 +0800 Subject: [PATCH 09/15] feat(query): evaluate projections and computed columns over kv scans KvOp::Scan carries projection and serialized computed-column bytes through the clone-source rewriter, planner, and native/RESP plan builders. The kv scan handler applies them to raw msgpack rows before sort, matching the document/columnar/timeseries scan handlers instead of returning NULL for computed expressions. --- nodedb-physical/src/physical_plan/kv/op.rs | 7 ++ nodedb/src/control/clone/resolver/rewrite.rs | 4 + .../planner/sql_plan_convert/scan/core.rs | 6 ++ .../src/control/server/exchange/full_scan.rs | 2 + .../server/native/dispatch/plan_builder/kv.rs | 2 + nodedb/src/control/server/resp/handler.rs | 6 ++ .../shared/ddl/neutral/weighted_pick.rs | 2 + .../predicate/txn_buffering/classify.rs | 2 + .../src/data/executor/handlers/kv/dispatch.rs | 4 + nodedb/src/data/executor/handlers/kv/scan.rs | 72 ++++++++++++++++ .../test_cross_type_join/multi_core_joins.rs | 6 ++ .../inproc/cases/executor_tests/test_kv.rs | 4 + .../cases/executor_tests/test_kv_advanced.rs | 2 + .../executor_tests/test_kv_scan_budget.rs | 2 + nodedb/tests/wire/cases/kv_sql_select.rs | 83 +++++++++++++++++++ 15 files changed, 204 insertions(+) diff --git a/nodedb-physical/src/physical_plan/kv/op.rs b/nodedb-physical/src/physical_plan/kv/op.rs index 1fe9126c7..317761917 100644 --- a/nodedb-physical/src/physical_plan/kv/op.rs +++ b/nodedb-physical/src/physical_plan/kv/op.rs @@ -143,6 +143,13 @@ pub enum KvOp { count: usize, /// Optional filter predicates (same format as DocumentScan filters). filters: Vec, + /// Output column names to keep. Empty = emit the full row. + #[serde(default)] + projection: Vec, + /// Serialized `Vec` (MessagePack), same encoding as + /// `DocumentOp::Scan::computed_columns`. Empty = none. + #[serde(default)] + computed_columns: Vec, /// Optional glob pattern for key matching (e.g., "user:*"). match_pattern: Option, /// ORDER BY terms, each an expression, applied to the scan result diff --git a/nodedb/src/control/clone/resolver/rewrite.rs b/nodedb/src/control/clone/resolver/rewrite.rs index f1189ebbd..c213cb030 100644 --- a/nodedb/src/control/clone/resolver/rewrite.rs +++ b/nodedb/src/control/clone/resolver/rewrite.rs @@ -295,6 +295,8 @@ pub fn rewrite_plan_for_source(params: RewriteForSourceParams<'_>) -> crate::Res cursor, count, filters, + projection, + computed_columns, match_pattern, sort_keys, // The original target-side scan never carries a ceiling @@ -307,6 +309,8 @@ pub fn rewrite_plan_for_source(params: RewriteForSourceParams<'_>) -> crate::Res cursor: cursor.clone(), count: *count, filters: filters.clone(), + projection: projection.clone(), + computed_columns: computed_columns.clone(), match_pattern: match_pattern.clone(), sort_keys: sort_keys.clone(), surrogate_ceiling: kv_surrogate_ceiling, diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/core.rs b/nodedb/src/control/planner/sql_plan_convert/scan/core.rs index 4367a9719..5699c1359 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/core.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/core.rs @@ -132,6 +132,12 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_scan( cursor: Vec::new(), count: limit.unwrap_or(usize::MAX), filters: filter_bytes, + // A kv scan emits the full row: the clone-source merge keys + // tombstone suppression on the row key, so the response shaper + // projects by output schema instead. Computed columns still + // evaluate per row on the Data Plane. + projection: Vec::new(), + computed_columns: computed_bytes, match_pattern: None, sort_keys: sort.clone(), // Original SQL planner output never carries a clone ceiling; diff --git a/nodedb/src/control/server/exchange/full_scan.rs b/nodedb/src/control/server/exchange/full_scan.rs index 58fb851d2..1bfbe89ba 100644 --- a/nodedb/src/control/server/exchange/full_scan.rs +++ b/nodedb/src/control/server/exchange/full_scan.rs @@ -148,6 +148,8 @@ pub fn full_scan_plan_for_collection( cursor: Vec::new(), count: COMPLETE_SCAN, filters: side_filters, + projection: Vec::new(), + computed_columns: Vec::new(), sort_keys: Vec::new(), match_pattern: None, surrogate_ceiling: None, diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs b/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs index d66f38600..36c161d03 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs @@ -24,6 +24,8 @@ pub(crate) fn build_scan( cursor, count, filters, + projection: Vec::new(), + computed_columns: Vec::new(), match_pattern, sort_keys: Vec::new(), surrogate_ceiling: None, diff --git a/nodedb/src/control/server/resp/handler.rs b/nodedb/src/control/server/resp/handler.rs index 0568646e5..356bf3cd1 100644 --- a/nodedb/src/control/server/resp/handler.rs +++ b/nodedb/src/control/server/resp/handler.rs @@ -345,6 +345,8 @@ async fn handle_scan(cmd: &RespCommand, session: &RespSession, state: &SharedSta cursor, count, filters: filter_bytes, + projection: Vec::new(), + computed_columns: Vec::new(), match_pattern, sort_keys: Vec::new(), surrogate_ceiling: None, @@ -379,6 +381,8 @@ async fn handle_keys(cmd: &RespCommand, session: &RespSession, state: &SharedSta cursor: Vec::new(), count: 100_000, filters: Vec::new(), + projection: Vec::new(), + computed_columns: Vec::new(), match_pattern: Some(pattern.to_string()), sort_keys: Vec::new(), surrogate_ceiling: None, @@ -409,6 +413,8 @@ async fn handle_dbsize(session: &RespSession, state: &SharedState) -> RespValue cursor: Vec::new(), count: 0, filters: Vec::new(), + projection: Vec::new(), + computed_columns: Vec::new(), match_pattern: None, sort_keys: Vec::new(), surrogate_ceiling: None, diff --git a/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs b/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs index 53edba9b0..dd713cc8b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs @@ -241,6 +241,8 @@ async fn scan_all_entries( cursor: Vec::new(), count: 100_000, filters: Vec::new(), + projection: Vec::new(), + computed_columns: Vec::new(), match_pattern: None, sort_keys: Vec::new(), surrogate_ceiling: None, diff --git a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs index 8c08d289d..071834fe6 100644 --- a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs +++ b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs @@ -1184,6 +1184,8 @@ mod tests { cursor: Vec::new(), count: 0, filters: Vec::new(), + projection: Vec::new(), + computed_columns: Vec::new(), match_pattern: None, sort_keys: Vec::new(), surrogate_ceiling: None, diff --git a/nodedb/src/data/executor/handlers/kv/dispatch.rs b/nodedb/src/data/executor/handlers/kv/dispatch.rs index e5c6d040b..259d1d422 100644 --- a/nodedb/src/data/executor/handlers/kv/dispatch.rs +++ b/nodedb/src/data/executor/handlers/kv/dispatch.rs @@ -135,6 +135,8 @@ impl CoreLoop { cursor, count, filters, + projection, + computed_columns, match_pattern, sort_keys, surrogate_ceiling, @@ -148,6 +150,8 @@ impl CoreLoop { count: *count, match_pattern: match_pattern.as_deref(), filters, + projection, + computed_columns_bytes: computed_columns, sort_keys, surrogate_ceiling: *surrogate_ceiling, }, diff --git a/nodedb/src/data/executor/handlers/kv/scan.rs b/nodedb/src/data/executor/handlers/kv/scan.rs index 188f489b3..df6681433 100644 --- a/nodedb/src/data/executor/handlers/kv/scan.rs +++ b/nodedb/src/data/executor/handlers/kv/scan.rs @@ -20,6 +20,8 @@ pub(in crate::data::executor) struct KvScanHandlerParams<'a> { pub count: usize, pub match_pattern: Option<&'a str>, pub filters: &'a [u8], + pub projection: &'a [String], + pub computed_columns_bytes: &'a [u8], pub sort_keys: &'a [nodedb_physical::physical_plan::SortKeySpec], pub surrogate_ceiling: Option, } @@ -38,6 +40,8 @@ impl CoreLoop { count, match_pattern, filters, + projection, + computed_columns_bytes, sort_keys, surrogate_ceiling, } = params; @@ -148,6 +152,16 @@ impl CoreLoop { result_entries.push(entry_mp); } + result_entries = match self.apply_kv_projection_and_computed( + task, + result_entries, + projection, + computed_columns_bytes, + ) { + Ok(rows) => rows, + Err(resp) => return resp, + }; + if !sort_keys.is_empty() && let Err(e) = super::super::sort_utils::sort_msgpack_rows(&mut result_entries, sort_keys) @@ -168,6 +182,64 @@ impl CoreLoop { } self.response_with_payload(task, payload) } + + /// Apply projection and computed columns to a kv scan's raw msgpack rows. + /// + /// Runs before sort so `ORDER BY` can name a computed alias. An empty + /// projection with non-empty computed columns keeps the full row (kv has + /// no fixed schema to project against) and appends the computed aliases; + /// a non-empty projection keeps only the named columns plus the computed + /// aliases. Returns `Err(Response)` on malformed computed-column bytes + /// or a per-row evaluation error, ready to return directly to the caller. + fn apply_kv_projection_and_computed( + &self, + task: &ExecutionTask, + result_entries: Vec>, + projection: &[String], + computed_columns_bytes: &[u8], + ) -> Result>, Response> { + if projection.is_empty() && computed_columns_bytes.is_empty() { + return Ok(result_entries); + } + + let computed_cols: Vec = + if computed_columns_bytes.is_empty() { + Vec::new() + } else { + match zerompk::from_msgpack(computed_columns_bytes) { + Ok(cols) => cols, + Err(e) => { + return Err(self.response_error( + task, + ErrorCode::Internal { + detail: format!("kv scan: malformed computed bytes: {e}"), + }, + )); + } + } + }; + + if projection.is_empty() { + crate::data::executor::handlers::provider_scan_compute::apply_windows_and_computed( + result_entries, + &[], + computed_columns_bytes, + ) + .map_err(|e| self.response_error(task, e)) + } else { + let mut projected_entries = Vec::with_capacity(result_entries.len()); + for entry in &result_entries { + let row = crate::data::executor::handlers::document::read::projection::apply_projection_msgpack( + entry, + &computed_cols, + projection, + ) + .map_err(|e| self.response_error(task, e))?; + projected_entries.push(row); + } + Ok(projected_entries) + } + } } /// Extract a single equality filter from serialized ScanFilter bytes. diff --git a/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/multi_core_joins.rs b/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/multi_core_joins.rs index 321116d8b..6a4115058 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/multi_core_joins.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/multi_core_joins.rs @@ -77,6 +77,8 @@ fn multi_core_broadcast_inner_join() { cursor: Vec::new(), count: 100, filters: Vec::new(), + projection: Vec::new(), + computed_columns: Vec::new(), sort_keys: Vec::new(), match_pattern: None, surrogate_ceiling: None, @@ -233,6 +235,8 @@ fn multi_core_broadcast_left_join() { cursor: Vec::new(), count: 100, filters: Vec::new(), + projection: Vec::new(), + computed_columns: Vec::new(), sort_keys: Vec::new(), match_pattern: None, surrogate_ceiling: None, @@ -402,6 +406,8 @@ fn multi_core_broadcast_merge_simulation() { cursor: Vec::new(), count: 100, filters: Vec::new(), + projection: Vec::new(), + computed_columns: Vec::new(), sort_keys: Vec::new(), match_pattern: None, surrogate_ceiling: None, diff --git a/nodedb/tests/inproc/cases/executor_tests/test_kv.rs b/nodedb/tests/inproc/cases/executor_tests/test_kv.rs index 52a8aa244..6c4474402 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_kv.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_kv.rs @@ -231,6 +231,8 @@ fn kv_scan_returns_entries() { cursor: Vec::new(), count: 100, filters: Vec::new(), + projection: Vec::new(), + computed_columns: Vec::new(), match_pattern: None, sort_keys: Vec::new(), surrogate_ceiling: None, @@ -280,6 +282,8 @@ fn kv_scan_with_match_pattern() { cursor: Vec::new(), count: 100, filters: Vec::new(), + projection: Vec::new(), + computed_columns: Vec::new(), match_pattern: Some("user:*".into()), sort_keys: Vec::new(), surrogate_ceiling: None, diff --git a/nodedb/tests/inproc/cases/executor_tests/test_kv_advanced.rs b/nodedb/tests/inproc/cases/executor_tests/test_kv_advanced.rs index 5dba10d7e..5f0bb210e 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_kv_advanced.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_kv_advanced.rs @@ -608,6 +608,8 @@ fn kv_index_write_amp_ratio_matches() { cursor: Vec::new(), count: 200, filters: Vec::new(), + projection: Vec::new(), + computed_columns: Vec::new(), match_pattern: None, sort_keys: Vec::new(), surrogate_ceiling: None, diff --git a/nodedb/tests/inproc/cases/executor_tests/test_kv_scan_budget.rs b/nodedb/tests/inproc/cases/executor_tests/test_kv_scan_budget.rs index 28ba8fc7a..7e1288f58 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_kv_scan_budget.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_kv_scan_budget.rs @@ -45,6 +45,8 @@ fn kv_scan(collection: &str, count: usize) -> PhysicalPlan { cursor: Vec::new(), count, filters: Vec::new(), + projection: Vec::new(), + computed_columns: Vec::new(), match_pattern: None, sort_keys: Vec::new(), surrogate_ceiling: None, diff --git a/nodedb/tests/wire/cases/kv_sql_select.rs b/nodedb/tests/wire/cases/kv_sql_select.rs index 222dab44a..2e32ce547 100644 --- a/nodedb/tests/wire/cases/kv_sql_select.rs +++ b/nodedb/tests/wire/cases/kv_sql_select.rs @@ -165,3 +165,86 @@ async fn kv_sql_single_column_projection() { assert_eq!(row.len(), 1); assert_eq!(row[0], "world"); } + +/// A computed column over a kv scan evaluates per row instead of returning +/// NULL, matching the document/columnar/timeseries scan handlers. +#[tokio::test] +async fn computed_column_over_kv_evaluates_per_row() { + let server = TestServer::start().await; + server + .exec("CREATE COLLECTION t (key STRING PRIMARY KEY, id INT) WITH (engine='kv')") + .await + .unwrap(); + for i in 1..=3i64 { + server + .exec(&format!("INSERT INTO t (key, id) VALUES ('k{i}', {i})")) + .await + .unwrap(); + } + + let rows = server + .query_rows("SELECT id, id * 2 AS d FROM t ORDER BY id") + .await + .expect("computed-column SELECT should succeed"); + + assert_eq!(rows.len(), 3); + let pairs: Vec<(i64, i64)> = rows + .iter() + .map(|r| (r[0].parse().unwrap(), r[1].parse().unwrap())) + .collect(); + assert_eq!(pairs, vec![(1, 2), (2, 4), (3, 6)]); +} + +/// A division-by-zero inside a computed column over a kv scan raises +/// SQLSTATE 22012 instead of materializing NULL into the response. +#[tokio::test] +async fn computed_column_division_by_zero_over_kv_errors_22012() { + let server = TestServer::start().await; + server + .exec("CREATE COLLECTION t (key STRING PRIMARY KEY, id INT) WITH (engine='kv')") + .await + .unwrap(); + for i in 1..=3i64 { + server + .exec(&format!("INSERT INTO t (key, id) VALUES ('k{i}', {i})")) + .await + .unwrap(); + } + + server + .expect_error("SELECT id, 1 / (id - 2) AS d FROM t", "22012") + .await; +} + +/// A function call in the projection list over a kv scan evaluates per row. +#[tokio::test] +async fn function_projection_over_kv_evaluates() { + let server = TestServer::start().await; + server + .exec("CREATE COLLECTION t (key STRING PRIMARY KEY, id INT, v STRING) WITH (engine='kv')") + .await + .unwrap(); + server + .exec("INSERT INTO t (key, id, v) VALUES ('k1', 1, 'a')") + .await + .unwrap(); + server + .exec("INSERT INTO t (key, id, v) VALUES ('k2', 2, 'b')") + .await + .unwrap(); + server + .exec("INSERT INTO t (key, id, v) VALUES ('k3', 3, 'c')") + .await + .unwrap(); + + let rows = server + .query_rows("SELECT upper(v) AS u FROM t ORDER BY id") + .await + .expect("function-projection SELECT should succeed"); + + let values: Vec = rows.iter().map(|r| r[0].clone()).collect(); + assert_eq!( + values, + vec!["A".to_string(), "B".to_string(), "C".to_string()] + ); +} From 744a39a6b06d3ba3bab01ad75258cb0f36baa976 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 18 Sep 2026 09:57:13 +0800 Subject: [PATCH 10/15] feat(query): stamp Control-Plane computed columns across response shaping Thread session sequence access (nextval/currval) through every path that shapes rows: kv/document/columnar scan responses, the HTTP materialized query route, native and pgwire dispatch/streaming, and RETURNING. shape_decoded_rows and shape_returning_rows take a sequences: Option<&dyn SequenceAccess> and stamp a projection's cp_computed columns onto the flat rows after redaction and before projection, so a SELECT-list sequence accessor resolves the same way regardless of which server path served the query. A caller with computed columns in scope but no session access fails the statement instead of shipping NULL under the alias. OutputSchema gains a cp_computed field populated wherever the planner builds an output schema, and a new nodedb::control::sequence::access module (SequenceAccess trait, SessionSequenceAccess) backs it, sharing error mapping extracted into sequence::error_map from the catalog adapter's existing sequence lookup. The HTTP materialized-query handler and the pgwire dispatch loop, both of which need the new sequence wiring, split from a single oversized file into a directory of focused modules (request parsing, row shaping, encoding for the former; task setup, the run loop, and finish handling for the latter). --- .../catalog_adapter/sequence_access.rs | 22 +- .../sql_plan_convert/output_schema/build.rs | 6 + .../sql_plan_convert/output_schema/columns.rs | 56 +- nodedb/src/control/sequence/access.rs | 202 ++++++ nodedb/src/control/sequence/error_map.rs | 80 +++ nodedb/src/control/sequence/mod.rs | 3 + .../server/http/routes/query/materialized.rs | 600 ------------------ .../http/routes/query/materialized/encode.rs | 75 +++ .../http/routes/query/materialized/mod.rs | 13 + .../http/routes/query/materialized/request.rs | 154 +++++ .../http/routes/query/materialized/shape.rs | 470 ++++++++++++++ .../server/http/routes/query/ndjson.rs | 20 +- .../server/http/routes/query_stream.rs | 9 +- .../server/http/routes/result_shape.rs | 31 + .../server/native/dispatch/conversion.rs | 2 + .../server/native/dispatch/response.rs | 2 + .../src/control/server/native/dispatch/sql.rs | 2 +- .../server/native/dispatch/sql_loop.rs | 12 + .../server/native/dispatch/streaming.rs | 6 +- .../src/control/server/native/session/run.rs | 2 +- .../server/native/session/session_stream.rs | 5 +- .../server/pgwire/handler/prepared/execute.rs | 3 + .../pgwire/handler/routing/calvin_response.rs | 3 + .../pgwire/handler/routing/cluster_array.rs | 7 +- .../handler/routing/dispatch_loop/finish.rs | 103 +++ .../handler/routing/dispatch_loop/mod.rs | 9 + .../run.rs} | 153 ++--- .../handler/routing/dispatch_loop/task.rs | 115 ++++ .../server/pgwire/handler/routing/execute.rs | 17 +- .../handler/routing/execute_dml_hooks.rs | 7 +- .../handler/routing/gateway_dispatch.rs | 10 +- .../pgwire/handler/routing/result_shaping.rs | 2 +- .../server/pgwire/handler/routing/set_ops.rs | 17 +- .../pgwire/handler/routing/streaming.rs | 13 +- .../pgwire/handler/shape_encode/cell.rs | 4 +- .../server/pgwire/handler/stream_response.rs | 5 + .../control/server/pgwire/types/error_map.rs | 12 + nodedb/src/control/server/pgwire/types/mod.rs | 3 +- .../response_shape/compose/array_slice.rs | 8 +- .../server/response_shape/compose/kernel.rs | 106 +++- .../response_shape/compose/materialized.rs | 33 +- .../src/control/server/response_shape/mod.rs | 1 + .../control/server/response_shape/request.rs | 6 + .../server/response_shape/returning.rs | 37 +- .../control/server/response_shape/schema.rs | 17 +- .../server/response_shape/stamp/evaluate.rs | 238 +++++++ .../server/response_shape/stamp/mod.rs | 8 + .../server/response_shape/stamp/substitute.rs | 367 +++++++++++ .../neutral/collection/dml/parse/dispatch.rs | 3 + 49 files changed, 2308 insertions(+), 771 deletions(-) create mode 100644 nodedb/src/control/sequence/access.rs create mode 100644 nodedb/src/control/sequence/error_map.rs delete mode 100644 nodedb/src/control/server/http/routes/query/materialized.rs create mode 100644 nodedb/src/control/server/http/routes/query/materialized/encode.rs create mode 100644 nodedb/src/control/server/http/routes/query/materialized/mod.rs create mode 100644 nodedb/src/control/server/http/routes/query/materialized/request.rs create mode 100644 nodedb/src/control/server/http/routes/query/materialized/shape.rs create mode 100644 nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/finish.rs create mode 100644 nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/mod.rs rename nodedb/src/control/server/pgwire/handler/routing/{dispatch_loop.rs => dispatch_loop/run.rs} (79%) create mode 100644 nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/task.rs create mode 100644 nodedb/src/control/server/response_shape/stamp/evaluate.rs create mode 100644 nodedb/src/control/server/response_shape/stamp/mod.rs create mode 100644 nodedb/src/control/server/response_shape/stamp/substitute.rs diff --git a/nodedb/src/control/planner/catalog_adapter/sequence_access.rs b/nodedb/src/control/planner/catalog_adapter/sequence_access.rs index 020fdd543..acea5ad60 100644 --- a/nodedb/src/control/planner/catalog_adapter/sequence_access.rs +++ b/nodedb/src/control/planner/catalog_adapter/sequence_access.rs @@ -8,7 +8,8 @@ use nodedb_sql::SqlError; -use crate::control::sequence::{SequenceError, SequenceRegistry}; +use crate::control::sequence::SequenceRegistry; +use crate::control::sequence::error_map::{map_sequence_error, undefined_sequence}; use super::adapter::OriginCatalog; @@ -63,22 +64,3 @@ impl OriginCatalog { }) } } - -/// The function exists, the object does not — SQLSTATE `42704`, never `42883`. -fn undefined_sequence(name: &str) -> SqlError { - SqlError::UndefinedObject { - kind: "sequence", - name: name.to_string(), - } -} - -/// Map a registry error onto its planner equivalent. -fn map_sequence_error(name: &str, error: SequenceError) -> SqlError { - match error { - SequenceError::NotFound { .. } => undefined_sequence(name), - other => SqlError::ObjectNotInPrerequisiteState { - object: name.to_string(), - detail: other.to_string(), - }, - } -} diff --git a/nodedb/src/control/planner/sql_plan_convert/output_schema/build.rs b/nodedb/src/control/planner/sql_plan_convert/output_schema/build.rs index 1d9a5dd14..80be200fb 100644 --- a/nodedb/src/control/planner/sql_plan_convert/output_schema/build.rs +++ b/nodedb/src/control/planner/sql_plan_convert/output_schema/build.rs @@ -44,6 +44,7 @@ pub fn build_output_schema( return OutputSchema { columns: Vec::new(), is_star: false, + cp_computed: Vec::new(), }; }; @@ -87,6 +88,7 @@ pub fn build_output_schema( OutputSchema { columns, is_star: false, + cp_computed: Vec::new(), } } SqlPlan::Scan { @@ -199,6 +201,7 @@ pub fn build_output_schema( }) .collect(), is_star: false, + cp_computed: Vec::new(), } } SqlPlan::Aggregate { @@ -264,6 +267,7 @@ pub fn build_output_schema( OutputSchema { columns, is_star: false, + cp_computed: Vec::new(), } } // Set operations take their column names/types from the first @@ -290,6 +294,7 @@ pub fn build_output_schema( }) .collect(), is_star: false, + cp_computed: Vec::new(), }, // The outer query determines the final projected shape; the CTE // definitions themselves are only inputs to it. @@ -342,6 +347,7 @@ pub fn build_output_schema( }) .collect(), is_star: false, + cp_computed: Vec::new(), }, // A write announces exactly what its `RETURNING` clause projects, from // the target collection's declared columns. `RETURNING` is a diff --git a/nodedb/src/control/planner/sql_plan_convert/output_schema/columns.rs b/nodedb/src/control/planner/sql_plan_convert/output_schema/columns.rs index 2b0eb8c92..13b2f27b8 100644 --- a/nodedb/src/control/planner/sql_plan_convert/output_schema/columns.rs +++ b/nodedb/src/control/planner/sql_plan_convert/output_schema/columns.rs @@ -18,7 +18,7 @@ use nodedb_sql::types_expr::SqlExpr; use crate::control::planner::sql_plan_convert::group_key_name::computed_group_key_name; use crate::control::planner::sql_plan_convert::output_schema_types::infer_computed_expr_type; use crate::control::server::response_shape::schema::{ - OutputColumn, OutputSchema, sql_data_type_to_ddl_col_type_with_width, + CpComputedColumn, OutputColumn, OutputSchema, sql_data_type_to_ddl_col_type_with_width, }; use crate::control::server::response_shape::types::DdlColType; @@ -234,14 +234,25 @@ pub(super) fn ordered_columns_for( /// appending only entries not already produced by a named projection. When /// the projection has no star, `ordered_cols` is ignored and behavior is the /// named-columns-only, `is_star=false` case. +/// +/// A `CpComputed` entry is announced under its alias like any other column +/// and also recorded in `cp_computed`, in SELECT-list order, so the response +/// shaper evaluates it on the Control Plane before the projection runs. pub(super) fn schema_from_projection( projection: &[Projection], types: &HashMap, ordered_cols: &[OutputColumn], ) -> OutputSchema { let mut columns = Vec::with_capacity(projection.len()); + let mut cp_computed = Vec::new(); let mut is_star = false; for p in projection { + if let Projection::CpComputed { expr, alias } = p { + cp_computed.push(CpComputedColumn { + alias: alias.clone(), + expr: super::super::expr::sql_expr_to_bridge_expr(expr), + }); + } match projection_to_column(p, types) { Some(col) => columns.push(col), None => { @@ -254,7 +265,11 @@ pub(super) fn schema_from_projection( } } } - OutputSchema { columns, is_star } + OutputSchema { + columns, + is_star, + cp_computed, + } } #[cfg(test)] @@ -327,6 +342,43 @@ mod tests { assert_eq!(col.ty, DdlColType::Bool); } + #[test] + fn cp_computed_entries_are_collected_in_select_list_order() { + let types = HashMap::new(); + let accessor = |name: &str| SqlExpr::Function { + name: name.to_string(), + args: vec![SqlExpr::Literal(nodedb_sql::types_expr::SqlValue::String( + "s".to_string(), + ))], + distinct: false, + }; + let projection = vec![ + Projection::Column("id".to_string()), + Projection::CpComputed { + expr: accessor("nextval"), + alias: "n".to_string(), + }, + Projection::CpComputed { + expr: accessor("currval"), + alias: "c".to_string(), + }, + ]; + let schema = schema_from_projection(&projection, &types, &[]); + assert_eq!(schema.columns.len(), 3); + assert_eq!(schema.columns[1].display_name, "n"); + assert_eq!(schema.columns[2].display_name, "c"); + let aliases: Vec<&str> = schema + .cp_computed + .iter() + .map(|c| c.alias.as_str()) + .collect(); + assert_eq!(aliases, ["n", "c"]); + assert!(matches!( + &schema.cp_computed[0].expr, + crate::bridge::expr_eval::SqlExpr::Function { name, .. } if name == "nextval" + )); + } + #[test] fn star_returns_none() { let types = HashMap::new(); diff --git a/nodedb/src/control/sequence/access.rs b/nodedb/src/control/sequence/access.rs new file mode 100644 index 000000000..adf1938c1 --- /dev/null +++ b/nodedb/src/control/sequence/access.rs @@ -0,0 +1,202 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Session-scoped sequence accessor evaluation for the response shaper. +//! +//! A SELECT-list accessor over a relation runs once per output row on the +//! Control Plane, after the Data Plane returns the rows. Each call resolves +//! against the node registry and the calling session's `currval` map with +//! the same semantics the plan-time catalog adapter applies to a constant +//! `SELECT nextval('s')`. + +use std::sync::Arc; + +use nodedb_types::{DatabaseId, TenantId}; + +use super::error_map::sequence_error_to_error; +use super::registry::SequenceRegistry; +use super::session_values::SessionSequenceValues; +use crate::control::state::SharedState; + +/// Session-scoped sequence accessor evaluation for the response shaper. +pub trait SequenceAccess { + /// Advance `name` and record the value as this session's `currval`. + fn nextval(&self, name: &str) -> crate::Result; + /// The last value THIS SESSION obtained from `nextval` on `name`. + fn currval(&self, name: &str) -> crate::Result; + /// Position `name` so the next `nextval` returns `value + increment`. + fn setval(&self, name: &str, value: i64) -> crate::Result; +} + +/// [`SequenceAccess`] over one session's view of the node registry. +/// +/// `session` is `None` on a transport with no session state (HTTP): `nextval` +/// still advances the registry, and `currval` reports that this session never +/// called `nextval`, matching the plan-time adapter's behavior for the same +/// caller. +pub struct SessionSequenceAccess<'a> { + registry: &'a SequenceRegistry, + session: Option>, + database_id: DatabaseId, + tenant_id: TenantId, +} + +impl<'a> SessionSequenceAccess<'a> { + pub fn new( + registry: &'a SequenceRegistry, + session: Option>, + database_id: DatabaseId, + tenant_id: TenantId, + ) -> Self { + Self { + registry, + session, + database_id, + tenant_id, + } + } + + /// Build over `state`'s sequence registry for one request's session, + /// database, and tenant scope. Every call site (pgwire task/finish, + /// native `sql_loop`, HTTP materialized/ndjson) derives its accessor + /// from these same four inputs; this is the one constructor for that. + pub fn for_session( + state: &'a SharedState, + session: Option>, + database_id: DatabaseId, + tenant_id: TenantId, + ) -> Self { + Self::new(&state.sequence_registry, session, database_id, tenant_id) + } +} + +impl SequenceAccess for SessionSequenceAccess<'_> { + fn nextval(&self, name: &str) -> crate::Result { + let database_id = self.database_id.as_u64(); + let tenant_id = self.tenant_id.as_u64(); + let value = self + .registry + .nextval(database_id, tenant_id, name) + .map_err(|e| sequence_error_to_error(name, e))?; + if let Some(session) = &self.session { + session.record(database_id, tenant_id, name, value); + } + Ok(value) + } + + fn currval(&self, name: &str) -> crate::Result { + let database_id = self.database_id.as_u64(); + let tenant_id = self.tenant_id.as_u64(); + // The sequence must exist before its absence from the session map can + // mean "not yet called in this session" rather than "no such object". + if !self.registry.exists(database_id, tenant_id, name) { + return Err(crate::Error::UndefinedObject { + kind: "sequence", + name: name.to_string(), + }); + } + self.session + .as_ref() + .and_then(|session| session.last(database_id, tenant_id, name)) + .ok_or_else(|| crate::Error::ObjectNotInPrerequisiteState { + object: name.to_string(), + detail: format!( + "currval of sequence \"{name}\" is not yet defined in this session" + ), + }) + } + + fn setval(&self, name: &str, value: i64) -> crate::Result { + self.registry + .setval( + self.database_id.as_u64(), + self.tenant_id.as_u64(), + name, + value, + ) + .map_err(|e| sequence_error_to_error(name, e)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::security::catalog::sequence_types::StoredSequence; + + const DB: DatabaseId = DatabaseId::DEFAULT; + const TENANT: TenantId = TenantId::new(7); + + fn registry_with(name: &str) -> SequenceRegistry { + let registry = SequenceRegistry::new(); + registry + .create(StoredSequence::new( + DB.as_u64(), + TENANT.as_u64(), + name.to_string(), + "alice".to_string(), + )) + .expect("create sequence"); + registry + } + + #[test] + fn nextval_records_currval_for_the_session() { + let registry = registry_with("s"); + let session = Arc::new(SessionSequenceValues::new()); + let access = SessionSequenceAccess::new(®istry, Some(Arc::clone(&session)), DB, TENANT); + let first = access.nextval("s").expect("nextval"); + assert_eq!(access.currval("s").expect("currval"), first); + let second = access.nextval("s").expect("nextval"); + assert!(second > first); + assert_eq!(access.currval("s").expect("currval"), second); + } + + #[test] + fn currval_before_nextval_is_prerequisite_state() { + let registry = registry_with("s"); + let access = SessionSequenceAccess::new( + ®istry, + Some(Arc::new(SessionSequenceValues::new())), + DB, + TENANT, + ); + assert!(matches!( + access.currval("s"), + Err(crate::Error::ObjectNotInPrerequisiteState { .. }) + )); + } + + #[test] + fn unknown_sequence_is_undefined_object() { + let registry = registry_with("s"); + let access = SessionSequenceAccess::new(®istry, None, DB, TENANT); + assert!(matches!( + access.nextval("missing"), + Err(crate::Error::UndefinedObject { + kind: "sequence", + .. + }) + )); + assert!(matches!( + access.currval("missing"), + Err(crate::Error::UndefinedObject { + kind: "sequence", + .. + }) + )); + assert!(matches!( + access.setval("missing", 5), + Err(crate::Error::UndefinedObject { + kind: "sequence", + .. + }) + )); + } + + #[test] + fn setval_positions_the_next_value() { + let registry = registry_with("s"); + let access = SessionSequenceAccess::new(®istry, None, DB, TENANT); + access.setval("s", 41).expect("setval"); + assert_eq!(access.nextval("s").expect("nextval"), 42); + } +} diff --git a/nodedb/src/control/sequence/error_map.rs b/nodedb/src/control/sequence/error_map.rs new file mode 100644 index 000000000..5fd09090f --- /dev/null +++ b/nodedb/src/control/sequence/error_map.rs @@ -0,0 +1,80 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Maps a [`SequenceError`] onto the planner's `SqlError` and the crate's +//! `Error`, with one SQLSTATE per cause on both surfaces: a missing sequence +//! is `42704` (`undefined_object`) and every other registry refusal is +//! `55000` (`object_not_in_prerequisite_state`). + +use nodedb_sql::SqlError; + +use super::types::SequenceError; + +/// The function exists, the object does not — SQLSTATE `42704`, never `42883`. +pub(crate) fn undefined_sequence(name: &str) -> SqlError { + SqlError::UndefinedObject { + kind: "sequence", + name: name.to_string(), + } +} + +/// Map a registry error onto its planner equivalent. +pub(crate) fn map_sequence_error(name: &str, error: SequenceError) -> SqlError { + match error { + SequenceError::NotFound { .. } => undefined_sequence(name), + other => SqlError::ObjectNotInPrerequisiteState { + object: name.to_string(), + detail: other.to_string(), + }, + } +} + +/// Map a registry error onto the crate error the response shaper returns. +/// +/// Mirrors [`map_sequence_error`] variant for variant, the way +/// `plan_error_map` carries `SqlError::UndefinedObject` and +/// `SqlError::ObjectNotInPrerequisiteState` into `crate::Error`, so a +/// per-row accessor error renders the same SQLSTATE a plan-time one does. +pub(crate) fn sequence_error_to_error(name: &str, error: SequenceError) -> crate::Error { + match error { + SequenceError::NotFound { .. } => crate::Error::UndefinedObject { + kind: "sequence", + name: name.to_string(), + }, + other => crate::Error::ObjectNotInPrerequisiteState { + object: name.to_string(), + detail: other.to_string(), + }, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn not_found_is_undefined_object_on_both_surfaces() { + let sql = map_sequence_error("s", SequenceError::NotFound { name: "s".into() }); + assert!(matches!( + sql, + SqlError::UndefinedObject { + kind: "sequence", + .. + } + )); + let err = sequence_error_to_error("s", SequenceError::NotFound { name: "s".into() }); + assert!( + matches!(err, crate::Error::UndefinedObject { kind: "sequence", ref name } if name == "s") + ); + } + + #[test] + fn exhausted_is_prerequisite_state_on_both_surfaces() { + let sql = map_sequence_error("s", SequenceError::Exhausted { name: "s".into() }); + assert!(matches!(sql, SqlError::ObjectNotInPrerequisiteState { .. })); + let err = sequence_error_to_error("s", SequenceError::Exhausted { name: "s".into() }); + assert!(matches!( + err, + crate::Error::ObjectNotInPrerequisiteState { ref object, .. } if object == "s" + )); + } +} diff --git a/nodedb/src/control/sequence/mod.rs b/nodedb/src/control/sequence/mod.rs index 16ba6f427..2d617f6b6 100644 --- a/nodedb/src/control/sequence/mod.rs +++ b/nodedb/src/control/sequence/mod.rs @@ -1,6 +1,8 @@ // SPDX-License-Identifier: BUSL-1.1 +pub mod access; mod ddl_overlay; +pub mod error_map; pub mod format; pub mod gap_free; pub mod log; @@ -9,6 +11,7 @@ pub mod registry; pub mod session_values; pub mod types; +pub use self::access::{SequenceAccess, SessionSequenceAccess}; pub use self::format::{FormatToken, ResetScope}; pub use self::gap_free::GapFreeManager; pub use self::range_alloc::RangeAllocator; diff --git a/nodedb/src/control/server/http/routes/query/materialized.rs b/nodedb/src/control/server/http/routes/query/materialized.rs deleted file mode 100644 index 91e1344da..000000000 --- a/nodedb/src/control/server/http/routes/query/materialized.rs +++ /dev/null @@ -1,600 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -use std::sync::Arc; - -use axum::extract::{Query as QueryParams, State}; -use axum::http::HeaderMap; -use axum::response::IntoResponse; - -use crate::bridge::envelope::Status; -use crate::control::gateway::GatewayErrorMap; -use crate::control::gateway::core::QueryContext; -use crate::control::security::audit::ArcAuditEmitter; -use crate::control::server::response_shape::redaction::QueryRedaction; -use crate::control::server::response_shape::request::MaterializedShapeRequest; -use crate::control::server::response_shape::types::describe_plan; -use crate::control::server::shared::authorization::authorize_database; -use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; -use crate::control::server::shared::plan_admission::{ - PlanAdmissionRequest, plan_authorize_and_admit, -}; -use crate::control::server::shared::quota_admission::admit_quota_for_dispatch; - -use super::super::super::auth::{ApiError, AppState, build_request_scope, resolve_auth_parts}; -use super::super::super::peer::PeerAddr; -use super::super::super::transport::ClientTransport; -use super::super::super::types::{HttpQueryRequest, HttpQueryResponse}; -use super::super::result_shape::{ - HttpShaped, ddl_results_to_json, passthrough_json_row, shape_http_payload, -}; -use super::{DatabaseQueryParam, resolve_database_id}; - -/// POST /v1/query — execute a SQL/DDL statement. -/// -/// Body: `{ "sql": "..." }`. Database context via `X-NodeDB-Database` header -/// or `?database=` param. -pub async fn query( - headers: HeaderMap, - peer: PeerAddr, - transport: ClientTransport, - QueryParams(db_param): QueryParams, - State(state): State, - axum::Json(body): axum::Json, -) -> Result { - let (identity, verified_jwt) = - resolve_auth_parts(&headers, &state, peer.as_str(), transport.security()).await?; - let database_id = resolve_database_id(&headers, &db_param, &state)?; - let trace_id = crate::control::trace_context::extract_from_headers(&headers); - let emitter = ArcAuditEmitter(Arc::clone(&state.shared.audit)); - authorize_database(&identity, database_id, &emitter).map_err(crate::Error::from)?; - - let sql = body.sql.as_str(); - - // Request-selected database is authoritative for RLS vars; passing it as the - // session database makes `scope.database_id()` resolve to `database_id`. - let request = build_request_scope( - &identity, - verified_jwt.as_ref(), - &headers, - &state, - database_id, - peer.as_str(), - ); - - // Admission gate runs once, before either DDL dispatch or DML planning, so both - // are covered. `Some(result)` carries the outcome as `X-RateLimit-*` headers below. - let rate_limit_result = crate::control::server::session_auth::check_request_admission( - &state.shared, - &request, - "sql", - )?; - let scope = request.into_resolved_scope(); - let rate_limit_headers = - super::super::super::rate_limit_headers::rate_limit_headers(&rate_limit_result); - - // HTTP is stateless — no BEGIN/COMMIT session concept — so a session-less scope - // satisfies the DDL dispatch signature and always takes the autocommit branch. - let http_scope = crate::control::server::shared::session::DetachedTxnScope::new(); - let txn_ctx = http_scope.ctx(); - - // Try DDL commands first. Reached only after the admission call above, so - // `shared::ddl::user_dispatch` must not admit this request a second time. - if let Some(result) = crate::control::server::shared::ddl::dispatch( - &state.shared, - &identity, - sql.trim(), - database_id, - &txn_ctx, - ) - .await - { - return match result { - Ok(results) => { - let json_rows = ddl_results_to_json(results); - Ok(( - rate_limit_headers, - axum::Json(HttpQueryResponse::ok(json_rows)), - )) - } - Err(e) => Err(ddl_error_to_api(e)), - }; - } - - // Extract per-query ON DENY override + plan SQL with RLS injection. - let tenant_id = identity.tenant_id; - - // Quota enforcement — reject before any planning or dispatch. - state - .shared - .check_tenant_quota(tenant_id) - .map_err(|e| ApiError::RateLimited { - message: e.to_string(), - retry_after_secs: 1, - })?; - - let (clean_sql, scope) = - crate::control::server::session_auth::apply_per_query_on_deny(sql, scope); - // Planning and lease admission run as one retried unit so a descriptor drain - // starting between them is absorbed rather than surfaced. - let admission = plan_authorize_and_admit(PlanAdmissionRequest { - state: &state.shared, - query_ctx: &state.query_ctx, - scope: &scope, - sql: &clean_sql, - trace_id: crate::types::TraceId::ZERO, - }) - .await - .map_err(ApiError::from)?; - let tasks = admission.tasks; - let output_schema = admission.output_schema; - let _lease_scope = admission.lease_scope; - - if tasks.is_empty() { - return Ok(( - rate_limit_headers, - axum::Json(HttpQueryResponse::ok(vec![])), - )); - } - - // Track active request for quota accounting. - let _request = state.shared.tenant_request_guard(tenant_id); - - // Execute each task via the SPSC bridge. - let mut result_rows = Vec::new(); - // Checked once, not per task: keeps the per-task extraction below a true - // no-op when metering is disabled (the default). - let metering_enabled = state.shared.metering_config.enabled; - - async { - for task in tasks { - // Extracted before `task.plan` is cloned/moved into any branch below. - let plan_metering_info = - metering_enabled.then(|| PlanMeteringInfo::extract(&task.plan)); - // A spent hard quota refuses the task before it runs; charging below is - // success-path only and never refuses. - if let Some(info) = &plan_metering_info { - admit_quota_for_dispatch(&state.shared, &scope, info).map_err(gateway_error)?; - } - let rows_before = result_rows.len(); - // `INSERT ... SELECT` orchestrates on the Control Plane and issues its own - // WAL-backed writes, so the outer per-task WAL append is skipped for it. - // Never a clone-write shape. - if let crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::InsertSelect { .. }, - ) = &task.plan - { - let plan_kind = describe_plan(&task.plan); - let plan_for_shape = task.plan.clone(); - let authorized_task = - authorize_materialized_task(&state.shared, &identity, &task) - .map_err(gateway_error)?; - let resp = crate::control::insert_select::run_authorized_insert_select( - &state.shared, - authorized_task, - ) - .await - .map_err(gateway_error)?; - append_response( - &mut result_rows, - resp, - ShapedAppend { - plan: &plan_for_shape, - plan_kind, - output_schema: &output_schema, - state: &state, - database_id, - tenant_id, - redaction: &QueryRedaction::for_plan( - tenant_id, - scope.auth(), - &plan_for_shape, - ), - }, - )?; - meter_task_dispatch(&state.shared, &scope, &plan_metering_info, rows_before, &result_rows); - continue; - } - - // Autocommit `MERGE` orchestrates on the Control Plane and issues its own - // writes, so the per-task WAL append below is skipped for it. - if let crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::Merge { - target_collection: _, - source_collection: _, - source_alias: _, - target_join_col: _, - source_join_col: _, - clauses: _, - returning: _, - resolved_inserts: None, - source_rows: _, - rls_filters: _, - rls_write_check: _, - resolved_sum_targets: _, - declared_primary_key: _, - }, - ) = &task.plan - { - let plan_kind = describe_plan(&task.plan); - let plan_for_shape = task.plan.clone(); - let authorized_task = - authorize_materialized_task(&state.shared, &identity, &task) - .map_err(gateway_error)?; - let resp = crate::control::merge_orchestrator::run_authorized_merge( - &state.shared, - authorized_task, - ) - .await - .map_err(gateway_error)?; - append_response( - &mut result_rows, - resp, - ShapedAppend { - plan: &plan_for_shape, - plan_kind, - output_schema: &output_schema, - state: &state, - database_id, - tenant_id, - redaction: &QueryRedaction::for_plan( - tenant_id, - scope.auth(), - &plan_for_shape, - ), - }, - )?; - meter_task_dispatch(&state.shared, &scope, &plan_metering_info, rows_before, &result_rows); - continue; - } - - // Autocommit `UPDATE ... FROM ` scans the source on its own core and - // ships it into the plan; the orchestrator's own write skips the WAL append below. - if let crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::UpdateFromJoin { - target_collection: _, - source_collection: _, - source_alias: _, - target_join_col: _, - source_join_col: _, - updates: _, - target_filters: _, - returning: _, - source_rows: None, - rls_filters: _, - rls_write_check: _, - resolved_sum_targets: _, - declared_primary_key: _, - }, - ) = &task.plan - { - let plan_kind = describe_plan(&task.plan); - let plan_for_shape = task.plan.clone(); - let authorized_task = - authorize_materialized_task(&state.shared, &identity, &task) - .map_err(gateway_error)?; - let resp = crate::control::update_from_join_orchestrator::run_authorized_update_from_join( - &state.shared, - authorized_task, - ) - .await - .map_err(gateway_error)?; - append_response( - &mut result_rows, - resp, - ShapedAppend { - plan: &plan_for_shape, - plan_kind, - output_schema: &output_schema, - state: &state, - database_id, - tenant_id, - redaction: &QueryRedaction::for_plan( - tenant_id, - scope.auth(), - &plan_for_shape, - ), - }, - )?; - meter_task_dispatch(&state.shared, &scope, &plan_metering_info, rows_before, &result_rows); - continue; - } - - // A governed columnar predicate UPDATE/DELETE resolves to a concrete row set - // before proposing, skipping the WAL append below; local (non-Raft) path skips this. - if let Some(resolver) = crate::control::write_resolve::resolver_for_plan(&task.plan) - && state.shared.async_raft_proposer().is_some() - { - let plan_kind = describe_plan(&task.plan); - let plan_for_shape = task.plan.clone(); - let authorized_task = - authorize_materialized_task(&state.shared, &identity, &task) - .map_err(gateway_error)?; - let resp = crate::control::write_resolve::run_authorized_write_resolve( - &state.shared, - authorized_task, - resolver, - ) - .await - .map_err(gateway_error)?; - append_response( - &mut result_rows, - resp, - ShapedAppend { - plan: &plan_for_shape, - plan_kind, - output_schema: &output_schema, - state: &state, - database_id, - tenant_id, - redaction: &QueryRedaction::for_plan( - tenant_id, - scope.auth(), - &plan_for_shape, - ), - }, - )?; - meter_task_dispatch(&state.shared, &scope, &plan_metering_info, rows_before, &result_rows); - continue; - } - - // Captured before dispatch moves `task.plan` — needed by shaping below. - let plan_kind = describe_plan(&task.plan); - let plan_for_shape = task.plan.clone(); - // Resolved once per task, reused for every payload it produced. - let redaction = QueryRedaction::for_plan(tenant_id, scope.auth(), &plan_for_shape); - - // Clone CoW write-path interception, then authorization, run once - // per task before dispatch — same protocol-neutral gate every - // transport runs. - let emitter = crate::control::security::audit::ArcAuditEmitter(Arc::clone( - &state.shared.audit, - )); - let checked = match crate::control::server::shared::clone_write::intercept_and_authorize( - crate::control::server::shared::clone_write::InterceptAndAuthorizeParams { - state: &state.shared, - task, - identity: &identity, - tenant_id, - permissions: &state.shared.permissions, - roles: &state.shared.roles, - emitter: &emitter, - }, - ) - .await - .map_err(gateway_error)? - { - crate::control::server::shared::clone_write::CloneCheckedOutcome::Handled(resp) => { - append_response( - &mut result_rows, - resp, - ShapedAppend { - plan: &plan_for_shape, - plan_kind, - output_schema: &output_schema, - state: &state, - database_id, - tenant_id, - redaction: &redaction, - }, - )?; - meter_task_dispatch( - &state.shared, - &scope, - &plan_metering_info, - rows_before, - &result_rows, - ); - continue; - } - crate::control::server::shared::clone_write::CloneCheckedOutcome::Proceed( - checked, - ) => checked, - }; - - // Prefer gateway (cluster-aware, owns WAL durability), else fall back to - // local SPSC dispatch, where WAL append precedes enqueue so LSN order matches. - let payloads = match state.shared.gateway.get() { - Some(gw) => { - let gw_ctx = QueryContext { - tenant_id: checked.tenant_id(), - trace_id, - database_id, - txn_id: None, - }; - gw.execute(&gw_ctx, checked) - .await - .map_err(gateway_error)? - } - None => { - // Single-node boot: gateway not yet initialised — dispatch locally. - let response = crate::control::server::dispatch_utils::dispatch_authorized_autocommit_write( - &state.shared, - checked, - trace_id, - ) - .await - .map_err(gateway_error)?; - if response.status != Status::Ok { - return Err(response_error(&response)); - } - vec![response.payload.to_vec()] - } - }; - - for payload in &payloads { - if payload.is_empty() { - continue; - } - match shape_http_payload(MaterializedShapeRequest { - payload, - plan: &plan_for_shape, - plan_kind, - projection: Some(&output_schema), - state: &state.shared, - database_id, - tenant_id, - redaction: Some(redaction.ctx(&state.shared.redaction)), - }) { - Ok(HttpShaped::Rows(rows)) => result_rows.extend(rows), - Ok(HttpShaped::Passthrough) => result_rows.push(passthrough_json_row(payload)), - Err(e) => return Err(ApiError::Internal(e.message().to_string())), - } - } - meter_task_dispatch(&state.shared, &scope, &plan_metering_info, rows_before, &result_rows); - } - - Ok((rate_limit_headers, axum::Json(HttpQueryResponse::ok(result_rows)))) - } - .await -} - -/// Authorize one task with no clone-write check — used only by the -/// Control-Plane orchestrator branches ahead of the general dispatch tail, -/// whose plan shapes (`InsertSelect`, `Merge`, `UpdateFromJoin`, a governed -/// predicate resolution) are never clone-write shapes. -fn authorize_materialized_task( - shared: &crate::control::state::SharedState, - identity: &crate::control::security::identity::AuthenticatedIdentity, - task: &nodedb_physical::physical_task::PhysicalTask, -) -> crate::Result { - let emitter = ArcAuditEmitter(Arc::clone(&shared.audit)); - crate::control::server::shared::authorization::authorize_task_set( - identity, - std::slice::from_ref(task), - &shared.permissions, - &shared.roles, - &emitter, - ) - .map_err(crate::Error::from)? - .into_tasks() - .into_iter() - .next() - .ok_or_else(|| crate::Error::Internal { - detail: "authorization returned an empty capability set".into(), - }) -} - -fn ddl_error_to_api(error: crate::control::server::shared::ddl::DdlError) -> ApiError { - let status = if error.sqlstate == "42501" { - axum::http::StatusCode::FORBIDDEN - } else { - axum::http::StatusCode::BAD_REQUEST - }; - ApiError::Coded { - status, - message: error.message, - code: error.code, - } -} - -fn gateway_error(error: crate::Error) -> ApiError { - let (status, msg) = GatewayErrorMap::to_http(&error); - ApiError::HttpStatus(status, msg) -} - -fn response_error(response: &crate::bridge::envelope::Response) -> ApiError { - let detail = response - .error_code - .as_ref() - .map(|code| format!("{code:?}")) - .unwrap_or_else(|| "unknown error".into()); - ApiError::Internal(detail) -} - -/// Meter one task's dispatch after its rows are appended to `result_rows` — -/// the row count is the delta since `rows_before`. -fn meter_task_dispatch( - state: &crate::control::state::SharedState, - scope: &crate::control::security::request_scope::RequestAuthScope<'_>, - info: &Option, - rows_before: usize, - result_rows: &[serde_json::Value], -) { - if let Some(info) = info { - let task_rows = (result_rows.len() - rows_before) as u64; - meter_dispatch(state, scope, info, Some(task_rows)); - } -} - -/// Everything one orchestrated task's response needs to be shaped and -/// appended. Grouped so the append helper stays within the argument budget as -/// it gained the per-statement redaction resolution. -struct ShapedAppend<'a> { - plan: &'a crate::bridge::envelope::PhysicalPlan, - plan_kind: crate::control::server::response_shape::types::PlanKind, - output_schema: &'a crate::control::server::response_shape::schema::OutputSchema, - state: &'a AppState, - database_id: nodedb_types::DatabaseId, - tenant_id: crate::types::TenantId, - redaction: &'a QueryRedaction, -} - -fn append_response( - result_rows: &mut Vec, - response: crate::bridge::envelope::Response, - append: ShapedAppend<'_>, -) -> Result<(), ApiError> { - if response.status != Status::Ok { - return Err(response_error(&response)); - } - let payload = response.payload.to_vec(); - if payload.is_empty() { - return Ok(()); - } - match shape_http_payload(MaterializedShapeRequest { - payload: &payload, - plan: append.plan, - plan_kind: append.plan_kind, - projection: Some(append.output_schema), - state: &append.state.shared, - database_id: append.database_id, - tenant_id: append.tenant_id, - redaction: Some(append.redaction.ctx(&append.state.shared.redaction)), - }) { - Ok(HttpShaped::Rows(rows)) => result_rows.extend(rows), - Ok(HttpShaped::Passthrough) => result_rows.push(passthrough_json_row(&payload)), - Err(e) => return Err(ApiError::Internal(e.message().to_string())), - } - Ok(()) -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn ddl_insufficient_privilege_maps_to_forbidden() { - let error = - crate::control::server::shared::ddl::DdlError::new("42501", "write permission denied"); - - assert!(matches!( - ddl_error_to_api(error), - ApiError::Coded { status, message, code } - if status == axum::http::StatusCode::FORBIDDEN - && message == "write permission denied" - && code == nodedb_types::error::ErrorCode::AUTHORIZATION_DENIED - )); - } - - /// Round-trips through the actual JSON response body `into_response()` - /// produces — not just the pre-serialization `ApiError` — so this proves - /// the code reaches the client, not merely that the server set it. - #[tokio::test] - async fn ddl_error_code_survives_into_response_json() { - let error = - crate::control::server::shared::ddl::DdlError::new("42501", "write permission denied"); - - let response = ddl_error_to_api(error).into_response(); - assert_eq!(response.status(), axum::http::StatusCode::FORBIDDEN); - - let body = axum::body::to_bytes(response.into_body(), usize::MAX) - .await - .expect("read response body"); - let json: serde_json::Value = serde_json::from_slice(&body).expect("valid JSON body"); - assert_eq!( - json["code"], - nodedb_types::error::ErrorCode::AUTHORIZATION_DENIED.to_string() - ); - assert_eq!(json["error"], "write permission denied"); - } -} diff --git a/nodedb/src/control/server/http/routes/query/materialized/encode.rs b/nodedb/src/control/server/http/routes/query/materialized/encode.rs new file mode 100644 index 000000000..f6d4be59c --- /dev/null +++ b/nodedb/src/control/server/http/routes/query/materialized/encode.rs @@ -0,0 +1,75 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Maps internal errors and DDL results onto the HTTP `ApiError` surface. + +use super::super::super::super::auth::ApiError; + +pub(super) fn ddl_error_to_api(error: crate::control::server::shared::ddl::DdlError) -> ApiError { + let status = if error.sqlstate == "42501" { + axum::http::StatusCode::FORBIDDEN + } else { + axum::http::StatusCode::BAD_REQUEST + }; + ApiError::Coded { + status, + message: error.message, + code: error.code, + } +} + +pub(super) fn gateway_error(error: crate::Error) -> ApiError { + let (status, msg) = crate::control::gateway::GatewayErrorMap::to_http(&error); + ApiError::HttpStatus(status, msg) +} + +pub(super) fn response_error(response: &crate::bridge::envelope::Response) -> ApiError { + let detail = response + .error_code + .as_ref() + .map(|code| format!("{code:?}")) + .unwrap_or_else(|| "unknown error".into()); + ApiError::Internal(detail) +} + +#[cfg(test)] +mod tests { + use axum::response::IntoResponse; + + use super::*; + + #[test] + fn ddl_insufficient_privilege_maps_to_forbidden() { + let error = + crate::control::server::shared::ddl::DdlError::new("42501", "write permission denied"); + + assert!(matches!( + ddl_error_to_api(error), + ApiError::Coded { status, message, code } + if status == axum::http::StatusCode::FORBIDDEN + && message == "write permission denied" + && code == nodedb_types::error::ErrorCode::AUTHORIZATION_DENIED + )); + } + + /// Round-trips through the actual JSON response body `into_response()` + /// produces — not just the pre-serialization `ApiError` — so this proves + /// the code reaches the client, not merely that the server set it. + #[tokio::test] + async fn ddl_error_code_survives_into_response_json() { + let error = + crate::control::server::shared::ddl::DdlError::new("42501", "write permission denied"); + + let response = ddl_error_to_api(error).into_response(); + assert_eq!(response.status(), axum::http::StatusCode::FORBIDDEN); + + let body = axum::body::to_bytes(response.into_body(), usize::MAX) + .await + .expect("read response body"); + let json: serde_json::Value = serde_json::from_slice(&body).expect("valid JSON body"); + assert_eq!( + json["code"], + nodedb_types::error::ErrorCode::AUTHORIZATION_DENIED.to_string() + ); + assert_eq!(json["error"], "write permission denied"); + } +} diff --git a/nodedb/src/control/server/http/routes/query/materialized/mod.rs b/nodedb/src/control/server/http/routes/query/materialized/mod.rs new file mode 100644 index 000000000..a277777ae --- /dev/null +++ b/nodedb/src/control/server/http/routes/query/materialized/mod.rs @@ -0,0 +1,13 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `/v1/query`: materialized (buffer-then-respond) SQL execution. +//! +//! `request.rs` holds the entry point through admission; `shape.rs` runs +//! the per-task dispatch loop and shapes each task's response; `encode.rs` +//! maps errors onto the HTTP surface. + +mod encode; +mod request; +mod shape; + +pub use request::query; diff --git a/nodedb/src/control/server/http/routes/query/materialized/request.rs b/nodedb/src/control/server/http/routes/query/materialized/request.rs new file mode 100644 index 000000000..71d0bf71b --- /dev/null +++ b/nodedb/src/control/server/http/routes/query/materialized/request.rs @@ -0,0 +1,154 @@ +// SPDX-License-Identifier: BUSL-1.1 + +use std::sync::Arc; + +use axum::extract::{Query as QueryParams, State}; +use axum::http::HeaderMap; +use axum::response::IntoResponse; + +use crate::control::security::audit::ArcAuditEmitter; +use crate::control::server::shared::authorization::authorize_database; +use crate::control::server::shared::plan_admission::{ + PlanAdmissionRequest, plan_authorize_and_admit, +}; + +use super::super::super::super::auth::{ + ApiError, AppState, build_request_scope, resolve_auth_parts, +}; +use super::super::super::super::peer::PeerAddr; +use super::super::super::super::transport::ClientTransport; +use super::super::super::super::types::{HttpQueryRequest, HttpQueryResponse}; +use super::super::super::result_shape::ddl_results_to_json; +use super::super::{DatabaseQueryParam, resolve_database_id}; +use super::encode::ddl_error_to_api; +use super::shape::{TaskLoopParams, run_task_loop}; + +/// POST /v1/query — execute a SQL/DDL statement. +/// +/// Body: `{ "sql": "..." }`. Database context via `X-NodeDB-Database` header +/// or `?database=` param. +pub async fn query( + headers: HeaderMap, + peer: PeerAddr, + transport: ClientTransport, + QueryParams(db_param): QueryParams, + State(state): State, + axum::Json(body): axum::Json, +) -> Result { + let (identity, verified_jwt) = + resolve_auth_parts(&headers, &state, peer.as_str(), transport.security()).await?; + let database_id = resolve_database_id(&headers, &db_param, &state)?; + let trace_id = crate::control::trace_context::extract_from_headers(&headers); + let emitter = ArcAuditEmitter(Arc::clone(&state.shared.audit)); + authorize_database(&identity, database_id, &emitter).map_err(crate::Error::from)?; + + let sql = body.sql.as_str(); + + // Request-selected database is authoritative for RLS vars; passing it as the + // session database makes `scope.database_id()` resolve to `database_id`. + let request = build_request_scope( + &identity, + verified_jwt.as_ref(), + &headers, + &state, + database_id, + peer.as_str(), + ); + + // Admission gate runs once, before either DDL dispatch or DML planning, so both + // are covered. `Some(result)` carries the outcome as `X-RateLimit-*` headers below. + let rate_limit_result = crate::control::server::session_auth::check_request_admission( + &state.shared, + &request, + "sql", + )?; + let scope = request.into_resolved_scope(); + let rate_limit_headers = + super::super::super::super::rate_limit_headers::rate_limit_headers(&rate_limit_result); + + // HTTP is stateless — no BEGIN/COMMIT session concept — so a session-less scope + // satisfies the DDL dispatch signature and always takes the autocommit branch. + let http_scope = crate::control::server::shared::session::DetachedTxnScope::new(); + let txn_ctx = http_scope.ctx(); + + // Try DDL commands first. Reached only after the admission call above, so + // `shared::ddl::user_dispatch` must not admit this request a second time. + if let Some(result) = crate::control::server::shared::ddl::dispatch( + &state.shared, + &identity, + sql.trim(), + database_id, + &txn_ctx, + ) + .await + { + return match result { + Ok(results) => { + let json_rows = ddl_results_to_json(results); + Ok(( + rate_limit_headers, + axum::Json(HttpQueryResponse::ok(json_rows)), + )) + } + Err(e) => Err(ddl_error_to_api(e)), + }; + } + + // Extract per-query ON DENY override + plan SQL with RLS injection. + let tenant_id = identity.tenant_id; + + // Quota enforcement — reject before any planning or dispatch. + state + .shared + .check_tenant_quota(tenant_id) + .map_err(|e| ApiError::RateLimited { + message: e.to_string(), + retry_after_secs: 1, + })?; + + let (clean_sql, scope) = + crate::control::server::session_auth::apply_per_query_on_deny(sql, scope); + // Planning and lease admission run as one retried unit so a descriptor drain + // starting between them is absorbed rather than surfaced. + let admission = plan_authorize_and_admit(PlanAdmissionRequest { + state: &state.shared, + query_ctx: &state.query_ctx, + scope: &scope, + sql: &clean_sql, + trace_id: crate::types::TraceId::ZERO, + }) + .await + .map_err(ApiError::from)?; + let tasks = admission.tasks; + let output_schema = admission.output_schema; + let _lease_scope = admission.lease_scope; + + if tasks.is_empty() { + return Ok(( + rate_limit_headers, + axum::Json(HttpQueryResponse::ok(vec![])), + )); + } + + // Track active request for quota accounting. + let _request = state.shared.tenant_request_guard(tenant_id); + + let result_rows = run_task_loop( + tasks, + TaskLoopParams { + state: &state, + identity: &identity, + scope, + output_schema, + database_id, + tenant_id, + trace_id, + }, + ) + .await?; + + Ok(( + rate_limit_headers, + axum::Json(HttpQueryResponse::ok(result_rows)), + )) +} diff --git a/nodedb/src/control/server/http/routes/query/materialized/shape.rs b/nodedb/src/control/server/http/routes/query/materialized/shape.rs new file mode 100644 index 000000000..bbd8b28a3 --- /dev/null +++ b/nodedb/src/control/server/http/routes/query/materialized/shape.rs @@ -0,0 +1,470 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The per-task dispatch loop: runs each admitted task, orchestrating the +//! Control-Plane-side plan shapes (`InsertSelect`, `Merge`, +//! `UpdateFromJoin`, a governed predicate resolution) directly and routing +//! everything else through the general clone-write / gateway / local-SPSC +//! dispatch path, then shapes each response into the JSON row set. + +use std::sync::Arc; + +use nodedb_physical::physical_task::PhysicalTask; + +use crate::bridge::envelope::Status; +use crate::control::gateway::core::QueryContext; +use crate::control::security::audit::ArcAuditEmitter; +use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::security::request_scope::RequestAuthScope; +use crate::control::server::response_shape::redaction::QueryRedaction; +use crate::control::server::response_shape::request::MaterializedShapeRequest; +use crate::control::server::response_shape::schema::OutputSchema; +use crate::control::server::response_shape::types::{PlanKind, describe_plan}; +use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; +use crate::control::server::shared::quota_admission::admit_quota_for_dispatch; +use crate::types::{DatabaseId, TenantId, TraceId}; + +use super::super::super::super::auth::{ApiError, AppState}; +use super::super::super::result_shape::{ + HttpShaped, passthrough_json_row, shape_error_to_api, shape_http_payload, +}; +use super::encode::{gateway_error, response_error}; + +/// Everything the per-task loop needs, beyond the tasks themselves. +pub(super) struct TaskLoopParams<'a> { + pub(super) state: &'a AppState, + pub(super) identity: &'a AuthenticatedIdentity, + pub(super) scope: RequestAuthScope<'a>, + pub(super) output_schema: OutputSchema, + pub(super) database_id: DatabaseId, + pub(super) tenant_id: TenantId, + pub(super) trace_id: TraceId, +} + +/// Run every admitted task and return the statement's JSON rows. +pub(super) async fn run_task_loop( + tasks: Vec, + params: TaskLoopParams<'_>, +) -> Result, ApiError> { + let TaskLoopParams { + state, + identity, + scope, + output_schema, + database_id, + tenant_id, + trace_id, + } = params; + + let mut result_rows = Vec::new(); + // Checked once, not per task: keeps the per-task extraction below a true + // no-op when metering is disabled (the default). + let metering_enabled = state.shared.metering_config.enabled; + + for task in tasks { + // Extracted before `task.plan` is cloned/moved into any branch below. + let plan_metering_info = metering_enabled.then(|| PlanMeteringInfo::extract(&task.plan)); + // A spent hard quota refuses the task before it runs; charging below is + // success-path only and never refuses. + if let Some(info) = &plan_metering_info { + admit_quota_for_dispatch(&state.shared, &scope, info).map_err(gateway_error)?; + } + let rows_before = result_rows.len(); + // `INSERT ... SELECT` orchestrates on the Control Plane and issues its own + // WAL-backed writes, so the outer per-task WAL append is skipped for it. + // Never a clone-write shape. + if let crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::InsertSelect { .. }, + ) = &task.plan + { + let plan_kind = describe_plan(&task.plan); + let plan_for_shape = task.plan.clone(); + let authorized_task = authorize_materialized_task(&state.shared, identity, &task) + .map_err(gateway_error)?; + let resp = crate::control::insert_select::run_authorized_insert_select( + &state.shared, + authorized_task, + ) + .await + .map_err(gateway_error)?; + append_response( + &mut result_rows, + resp, + ShapedAppend { + plan: &plan_for_shape, + plan_kind, + output_schema: &output_schema, + state, + database_id, + tenant_id, + redaction: &QueryRedaction::for_plan(tenant_id, scope.auth(), &plan_for_shape), + }, + )?; + meter_task_dispatch( + &state.shared, + &scope, + &plan_metering_info, + rows_before, + &result_rows, + ); + continue; + } + + // Autocommit `MERGE` orchestrates on the Control Plane and issues its own + // writes, so the per-task WAL append below is skipped for it. + if let crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::Merge { + target_collection: _, + source_collection: _, + source_alias: _, + target_join_col: _, + source_join_col: _, + clauses: _, + returning: _, + resolved_inserts: None, + source_rows: _, + rls_filters: _, + rls_write_check: _, + resolved_sum_targets: _, + declared_primary_key: _, + }, + ) = &task.plan + { + let plan_kind = describe_plan(&task.plan); + let plan_for_shape = task.plan.clone(); + let authorized_task = authorize_materialized_task(&state.shared, identity, &task) + .map_err(gateway_error)?; + let resp = crate::control::merge_orchestrator::run_authorized_merge( + &state.shared, + authorized_task, + ) + .await + .map_err(gateway_error)?; + append_response( + &mut result_rows, + resp, + ShapedAppend { + plan: &plan_for_shape, + plan_kind, + output_schema: &output_schema, + state, + database_id, + tenant_id, + redaction: &QueryRedaction::for_plan(tenant_id, scope.auth(), &plan_for_shape), + }, + )?; + meter_task_dispatch( + &state.shared, + &scope, + &plan_metering_info, + rows_before, + &result_rows, + ); + continue; + } + + // Autocommit `UPDATE ... FROM ` scans the source on its own core and + // ships it into the plan; the orchestrator's own write skips the WAL append below. + if let crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::UpdateFromJoin { + target_collection: _, + source_collection: _, + source_alias: _, + target_join_col: _, + source_join_col: _, + updates: _, + target_filters: _, + returning: _, + source_rows: None, + rls_filters: _, + rls_write_check: _, + resolved_sum_targets: _, + declared_primary_key: _, + }, + ) = &task.plan + { + let plan_kind = describe_plan(&task.plan); + let plan_for_shape = task.plan.clone(); + let authorized_task = authorize_materialized_task(&state.shared, identity, &task) + .map_err(gateway_error)?; + let resp = + crate::control::update_from_join_orchestrator::run_authorized_update_from_join( + &state.shared, + authorized_task, + ) + .await + .map_err(gateway_error)?; + append_response( + &mut result_rows, + resp, + ShapedAppend { + plan: &plan_for_shape, + plan_kind, + output_schema: &output_schema, + state, + database_id, + tenant_id, + redaction: &QueryRedaction::for_plan(tenant_id, scope.auth(), &plan_for_shape), + }, + )?; + meter_task_dispatch( + &state.shared, + &scope, + &plan_metering_info, + rows_before, + &result_rows, + ); + continue; + } + + // A governed columnar predicate UPDATE/DELETE resolves to a concrete row set + // before proposing, skipping the WAL append below; local (non-Raft) path skips this. + if let Some(resolver) = crate::control::write_resolve::resolver_for_plan(&task.plan) + && state.shared.async_raft_proposer().is_some() + { + let plan_kind = describe_plan(&task.plan); + let plan_for_shape = task.plan.clone(); + let authorized_task = authorize_materialized_task(&state.shared, identity, &task) + .map_err(gateway_error)?; + let resp = crate::control::write_resolve::run_authorized_write_resolve( + &state.shared, + authorized_task, + resolver, + ) + .await + .map_err(gateway_error)?; + append_response( + &mut result_rows, + resp, + ShapedAppend { + plan: &plan_for_shape, + plan_kind, + output_schema: &output_schema, + state, + database_id, + tenant_id, + redaction: &QueryRedaction::for_plan(tenant_id, scope.auth(), &plan_for_shape), + }, + )?; + meter_task_dispatch( + &state.shared, + &scope, + &plan_metering_info, + rows_before, + &result_rows, + ); + continue; + } + + // Captured before dispatch moves `task.plan` — needed by shaping below. + let plan_kind = describe_plan(&task.plan); + let plan_for_shape = task.plan.clone(); + // Resolved once per task, reused for every payload it produced. + let redaction = QueryRedaction::for_plan(tenant_id, scope.auth(), &plan_for_shape); + + // Clone CoW write-path interception, then authorization, run once + // per task before dispatch — same protocol-neutral gate every + // transport runs. + let emitter = ArcAuditEmitter(Arc::clone(&state.shared.audit)); + let checked = match crate::control::server::shared::clone_write::intercept_and_authorize( + crate::control::server::shared::clone_write::InterceptAndAuthorizeParams { + state: &state.shared, + task, + identity, + tenant_id, + permissions: &state.shared.permissions, + roles: &state.shared.roles, + emitter: &emitter, + }, + ) + .await + .map_err(gateway_error)? + { + crate::control::server::shared::clone_write::CloneCheckedOutcome::Handled(resp) => { + append_response( + &mut result_rows, + resp, + ShapedAppend { + plan: &plan_for_shape, + plan_kind, + output_schema: &output_schema, + state, + database_id, + tenant_id, + redaction: &redaction, + }, + )?; + meter_task_dispatch( + &state.shared, + &scope, + &plan_metering_info, + rows_before, + &result_rows, + ); + continue; + } + crate::control::server::shared::clone_write::CloneCheckedOutcome::Proceed(checked) => { + checked + } + }; + + // Prefer gateway (cluster-aware, owns WAL durability), else fall back to + // local SPSC dispatch, where WAL append precedes enqueue so LSN order matches. + let payloads = match state.shared.gateway.get() { + Some(gw) => { + let gw_ctx = QueryContext { + tenant_id: checked.tenant_id(), + trace_id, + database_id, + txn_id: None, + }; + gw.execute(&gw_ctx, checked).await.map_err(gateway_error)? + } + None => { + // Single-node boot: gateway not yet initialised — dispatch locally. + let response = + crate::control::server::dispatch_utils::dispatch_authorized_autocommit_write( + &state.shared, + checked, + trace_id, + ) + .await + .map_err(gateway_error)?; + if response.status != Status::Ok { + return Err(response_error(&response)); + } + vec![response.payload.to_vec()] + } + }; + + // HTTP carries no session, so `currval` reports "not yet called + // in this session" while `nextval` still advances the registry — + // the same answer the plan-time adapter gives this transport. + let sequences = crate::control::sequence::SessionSequenceAccess::for_session( + &state.shared, + None, + database_id, + tenant_id, + ); + for payload in &payloads { + if payload.is_empty() { + continue; + } + match shape_http_payload(MaterializedShapeRequest { + payload, + plan: &plan_for_shape, + plan_kind, + projection: Some(&output_schema), + state: &state.shared, + database_id, + tenant_id, + redaction: Some(redaction.ctx(&state.shared.redaction)), + sequences: Some(&sequences), + }) { + Ok(HttpShaped::Rows(rows)) => result_rows.extend(rows), + Ok(HttpShaped::Passthrough) => result_rows.push(passthrough_json_row(payload)), + Err(e) => return Err(shape_error_to_api(e)), + } + } + meter_task_dispatch( + &state.shared, + &scope, + &plan_metering_info, + rows_before, + &result_rows, + ); + } + + Ok(result_rows) +} + +/// Authorize one task with no clone-write check — used only by the +/// Control-Plane orchestrator branches ahead of the general dispatch tail, +/// whose plan shapes (`InsertSelect`, `Merge`, `UpdateFromJoin`, a governed +/// predicate resolution) are never clone-write shapes. +fn authorize_materialized_task( + shared: &crate::control::state::SharedState, + identity: &AuthenticatedIdentity, + task: &PhysicalTask, +) -> crate::Result { + let emitter = ArcAuditEmitter(Arc::clone(&shared.audit)); + crate::control::server::shared::authorization::authorize_task_set( + identity, + std::slice::from_ref(task), + &shared.permissions, + &shared.roles, + &emitter, + ) + .map_err(crate::Error::from)? + .into_tasks() + .into_iter() + .next() + .ok_or_else(|| crate::Error::Internal { + detail: "authorization returned an empty capability set".into(), + }) +} + +/// Meter one task's dispatch after its rows are appended to `result_rows` — +/// the row count is the delta since `rows_before`. +fn meter_task_dispatch( + state: &crate::control::state::SharedState, + scope: &RequestAuthScope<'_>, + info: &Option, + rows_before: usize, + result_rows: &[serde_json::Value], +) { + if let Some(info) = info { + let task_rows = (result_rows.len() - rows_before) as u64; + meter_dispatch(state, scope, info, Some(task_rows)); + } +} + +/// Everything one orchestrated task's response needs to be shaped and +/// appended. Grouped so the append helper stays within the argument budget as +/// it gained the per-statement redaction resolution. +struct ShapedAppend<'a> { + plan: &'a crate::bridge::envelope::PhysicalPlan, + plan_kind: PlanKind, + output_schema: &'a OutputSchema, + state: &'a AppState, + database_id: DatabaseId, + tenant_id: TenantId, + redaction: &'a QueryRedaction, +} + +fn append_response( + result_rows: &mut Vec, + response: crate::bridge::envelope::Response, + append: ShapedAppend<'_>, +) -> Result<(), ApiError> { + if response.status != Status::Ok { + return Err(response_error(&response)); + } + let payload = response.payload.to_vec(); + if payload.is_empty() { + return Ok(()); + } + // HTTP carries no session: `nextval` advances the registry, `currval` + // reports "not yet called in this session". + let sequences = crate::control::sequence::SessionSequenceAccess::for_session( + &append.state.shared, + None, + append.database_id, + append.tenant_id, + ); + match shape_http_payload(MaterializedShapeRequest { + payload: &payload, + plan: append.plan, + plan_kind: append.plan_kind, + projection: Some(append.output_schema), + state: &append.state.shared, + database_id: append.database_id, + tenant_id: append.tenant_id, + redaction: Some(append.redaction.ctx(&append.state.shared.redaction)), + sequences: Some(&sequences), + }) { + Ok(HttpShaped::Rows(rows)) => result_rows.extend(rows), + Ok(HttpShaped::Passthrough) => result_rows.push(passthrough_json_row(&payload)), + Err(e) => return Err(shape_error_to_api(e)), + } + Ok(()) +} diff --git a/nodedb/src/control/server/http/routes/query/ndjson.rs b/nodedb/src/control/server/http/routes/query/ndjson.rs index eeaa46697..097f16619 100644 --- a/nodedb/src/control/server/http/routes/query/ndjson.rs +++ b/nodedb/src/control/server/http/routes/query/ndjson.rs @@ -128,7 +128,16 @@ pub async fn query_ndjson( // `Body::from_stream` polls the data-plane stream under normal HTTP backpressure // while its captured lease scope stays alive until body completion or disconnect. - match try_open_stream(&state, &tasks, &identity, database_id, trace_id).await { + match try_open_stream( + &state, + &tasks, + &identity, + database_id, + &output_schema, + trace_id, + ) + .await + { Ok(Some((stream, limit))) => { let Some(lease_scope) = lease_scope.take() else { return ApiError::from(crate::Error::Internal { @@ -317,6 +326,14 @@ pub async fn query_ndjson( // Row count for metering below — a per-row shaping error doesn't // change whether the task is billed, only how many rows count. let mut task_rows: u64 = 0; + // HTTP carries no session: `nextval` advances the registry, + // `currval` reports "not yet called in this session". + let sequences = crate::control::sequence::SessionSequenceAccess::for_session( + &state.shared, + None, + database_id, + tenant_id, + ); for payload in &payloads { if payload.is_empty() { continue; @@ -330,6 +347,7 @@ pub async fn query_ndjson( database_id, tenant_id, redaction: Some(redaction.ctx(&state.shared.redaction)), + sequences: Some(&sequences), }) { Ok(HttpShaped::Rows(rows)) => { task_rows += rows.len() as u64; diff --git a/nodedb/src/control/server/http/routes/query_stream.rs b/nodedb/src/control/server/http/routes/query_stream.rs index c05149c97..9ca180aab 100644 --- a/nodedb/src/control/server/http/routes/query_stream.rs +++ b/nodedb/src/control/server/http/routes/query_stream.rs @@ -42,7 +42,10 @@ use super::super::auth::AppState; /// /// Mirrors the pgwire `maybe_stream_select` plan-shape predicate via /// [`streamable_gather_child`] plus the single-task and no-set-op gates. HTTP -/// is stateless, so there is no autocommit / transaction-block check. +/// is stateless, so there is no autocommit / transaction-block check. A +/// projection with Control-Plane computed columns (`cp_computed`) is not +/// eligible either: those columns need per-row sequence access the per-batch +/// shaper does not carry, so the materialized path answers instead. /// /// Returns `Ok(Some((stream, limit)))` when eligible, `Ok(None)` when the /// caller should fall back to the materialized path, or `Err` when the @@ -52,12 +55,13 @@ pub(super) async fn try_open_stream( tasks: &[PhysicalTask], identity: &AuthenticatedIdentity, database_id: nodedb_types::DatabaseId, + output_schema: &OutputSchema, trace_id: crate::types::TraceId, ) -> crate::Result> { let [task] = tasks else { return Ok(None); }; - if task.post_set_op != PostSetOp::None { + if task.post_set_op != PostSetOp::None || !output_schema.cp_computed.is_empty() { return Ok(None); } let Some((child_plan, limit)) = streamable_gather_child(&task.plan) else { @@ -196,6 +200,7 @@ pub(super) fn ndjson_body_stream( value, projection.as_ref(), redaction.as_ref().map(|r| r.ctx(&state.redaction)), + None, ) { Ok(s) => s, Err(e) => { diff --git a/nodedb/src/control/server/http/routes/result_shape.rs b/nodedb/src/control/server/http/routes/result_shape.rs index 32ee1f6ae..a3022bd33 100644 --- a/nodedb/src/control/server/http/routes/result_shape.rs +++ b/nodedb/src/control/server/http/routes/result_shape.rs @@ -10,10 +10,41 @@ //! (`Execution`, `DmlResult`) come back as [`HttpShaped::Passthrough`]; the //! caller keeps its existing raw decode/base64 fallback for those. +use axum::http::StatusCode; + use crate::control::server::response_shape::cell::row_to_wire_json; use crate::control::server::response_shape::compose::{ShapeOutcome, shape_response_materialized}; use crate::control::server::response_shape::request::MaterializedShapeRequest; use nodedb_types::NodeDbError; +use nodedb_types::error::ErrorCode; + +use super::super::auth::ApiError; + +/// Map a shaping error to the HTTP error the client reads, keeping its +/// numeric code. A statement-level refusal the shaper raises per row — an +/// unknown sequence, `currval` before `nextval`, a bad accessor argument, +/// division by zero — is the caller's error and answers `400`; anything +/// else is the server's and answers `500`. +pub(super) fn shape_error_to_api(e: NodeDbError) -> ApiError { + let code = e.code(); + let status = if matches!( + code, + ErrorCode::UNDEFINED_OBJECT + | ErrorCode::OBJECT_NOT_READY + | ErrorCode::PLAN_ERROR + | ErrorCode::DIVISION_BY_ZERO + | ErrorCode::BAD_REQUEST + ) { + StatusCode::BAD_REQUEST + } else { + StatusCode::INTERNAL_SERVER_ERROR + }; + ApiError::Coded { + status, + message: e.message().to_string(), + code, + } +} /// Outcome of shaping one Data-Plane payload for an HTTP response. pub(super) enum HttpShaped { diff --git a/nodedb/src/control/server/native/dispatch/conversion.rs b/nodedb/src/control/server/native/dispatch/conversion.rs index 3c4decf63..3df2f2c86 100644 --- a/nodedb/src/control/server/native/dispatch/conversion.rs +++ b/nodedb/src/control/server/native/dispatch/conversion.rs @@ -236,6 +236,8 @@ pub(crate) fn calvin_native_response( database_id, tenant_id, redaction: redaction.as_ref().map(|r| r.ctx(&state.redaction)), + // No projection, so no Control-Plane computed column to resolve. + sequences: None, }) { let (cols, rows) = to_native_columns_rows(&shaped); diff --git a/nodedb/src/control/server/native/dispatch/response.rs b/nodedb/src/control/server/native/dispatch/response.rs index aeba1e6e5..ac5f31cad 100644 --- a/nodedb/src/control/server/native/dispatch/response.rs +++ b/nodedb/src/control/server/native/dispatch/response.rs @@ -36,6 +36,8 @@ pub(crate) fn data_plane_response_to_native( database_id: ctx.database_id(), tenant_id: ctx.tenant_id(), redaction: Some(redaction.ctx(&ctx.state.redaction)), + // No projection, so no Control-Plane computed column to resolve. + sequences: None, }) { Ok(ShapeOutcome::Rows(shaped)) => { let (columns, rows) = to_native_columns_rows(&shaped); diff --git a/nodedb/src/control/server/native/dispatch/sql.rs b/nodedb/src/control/server/native/dispatch/sql.rs index fef03ff46..fc2ec9bd7 100644 --- a/nodedb/src/control/server/native/dispatch/sql.rs +++ b/nodedb/src/control/server/native/dispatch/sql.rs @@ -392,7 +392,7 @@ async fn execute_planned( if let Err(error) = stream.attach_lease_scope(scope) { return resp(error_to_native(seq, &error)); } - return SqlOutcome::Stream(stream); + return SqlOutcome::Stream(Box::new(stream)); } Ok(None) => {} Err(error) => return resp(error_to_native(seq, &error)), diff --git a/nodedb/src/control/server/native/dispatch/sql_loop.rs b/nodedb/src/control/server/native/dispatch/sql_loop.rs index d7c3397d3..9905d199c 100644 --- a/nodedb/src/control/server/native/dispatch/sql_loop.rs +++ b/nodedb/src/control/server/native/dispatch/sql_loop.rs @@ -12,6 +12,7 @@ use nodedb_types::protocol::NativeResponse; use nodedb_types::value::Value; use crate::bridge::envelope::Status; +use crate::control::sequence::SessionSequenceAccess; use crate::control::server::response_shape::compose::{ShapeOutcome, shape_response_materialized}; use crate::control::server::response_shape::redaction::QueryRedaction; use crate::control::server::response_shape::request::MaterializedShapeRequest; @@ -67,6 +68,15 @@ pub(super) async fn run_dispatch_loop( // name) a true no-op on the hot path for every deployment that hasn't // turned it on. let metering_enabled = ctx.state.metering_config.enabled; + // Session-scoped sequence access for the statement's Control-Plane + // computed columns: the registry plus this connection's `currval` map. + let session_sequences = ctx.sessions.sequence_values(ctx.peer_addr); + let sequences = SessionSequenceAccess::for_session( + ctx.state, + session_sequences, + database_id, + ctx.tenant_id(), + ); for task in tasks { if task.tenant_id != ctx.tenant_id() { @@ -204,6 +214,7 @@ pub(super) async fn run_dispatch_loop( database_id, tenant_id: ctx.tenant_id(), redaction: Some(redaction.ctx(&ctx.state.redaction)), + sequences: Some(&sequences), }) { Ok(ShapeOutcome::Rows(mut shaped)) => { if let Some(notice) = shaped.notice.take() { @@ -315,6 +326,7 @@ pub(super) async fn run_dispatch_loop( database_id, tenant_id: ctx.tenant_id(), redaction: Some(redaction.ctx(&ctx.state.redaction)), + sequences: Some(&sequences), }) { Ok(ShapeOutcome::Rows(mut shaped)) => { if let Some(notice) = shaped.notice.take() { diff --git a/nodedb/src/control/server/native/dispatch/streaming.rs b/nodedb/src/control/server/native/dispatch/streaming.rs index cda6e6e18..725a9b77e 100644 --- a/nodedb/src/control/server/native/dispatch/streaming.rs +++ b/nodedb/src/control/server/native/dispatch/streaming.rs @@ -36,7 +36,7 @@ pub(crate) enum SqlOutcome { /// A single materialized response — encoded/chunked by the session loop. Response(Box), /// A lazy row stream to be emitted as multiple frames. - Stream(SqlStream), + Stream(Box), } impl SqlOutcome { @@ -101,6 +101,9 @@ impl SqlStream { /// Eligibility mirrors the pgwire `maybe_stream_select` predicate: /// - single task (`tasks.len() == 1`), /// - `post_set_op == PostSetOp::None`, +/// - no Control-Plane computed column in the projection (`cp_computed` +/// needs per-row session sequence access, which the per-batch shaper +/// does not carry), /// - autocommit (not inside a `BEGIN..COMMIT` block), and /// - the plan is `Query(Exchange(Gather{as_aggregate:false}))` over a /// streamable unordered scan (via [`streamable_gather_child`]). @@ -118,6 +121,7 @@ pub(crate) async fn try_open_sql_stream( return Ok(None); }; if task.post_set_op != PostSetOp::None + || output_schema.is_some_and(|s| !s.cp_computed.is_empty()) || ctx.sessions.transaction_state(ctx.peer_addr) == TransactionState::InBlock { return Ok(None); diff --git a/nodedb/src/control/server/native/session/run.rs b/nodedb/src/control/server/native/session/run.rs index 165f1c655..8f6469a45 100644 --- a/nodedb/src/control/server/native/session/run.rs +++ b/nodedb/src/control/server/native/session/run.rs @@ -311,7 +311,7 @@ impl NativeSession { dispatch::SqlOutcome::Stream(sql_stream) => { super::session_stream::emit_sql_stream( &mut self.stream, - sql_stream, + *sql_stream, format, self.state.as_ref(), ) diff --git a/nodedb/src/control/server/native/session/session_stream.rs b/nodedb/src/control/server/native/session/session_stream.rs index 7dae7528b..a979d21d9 100644 --- a/nodedb/src/control/server/native/session/session_stream.rs +++ b/nodedb/src/control/server/native/session/session_stream.rs @@ -43,7 +43,10 @@ fn decode_batch_to_columns_rows( ) -> crate::Result<(Vec, Vec>)> { match decode_payload_value(payload) { Ok(decoded) => { - let shaped = shape_decoded_rows(decoded, projection, redaction)?; + // A streamed plan never carries Control-Plane computed columns: + // `try_open_sql_stream` declines those, so no session access + // is needed per batch. + let shaped = shape_decoded_rows(decoded, projection, redaction, None)?; Ok(to_native_columns_rows(&shaped)) } Err(_) => Ok(( diff --git a/nodedb/src/control/server/pgwire/handler/prepared/execute.rs b/nodedb/src/control/server/pgwire/handler/prepared/execute.rs index 61327b308..182b0d1db 100644 --- a/nodedb/src/control/server/pgwire/handler/prepared/execute.rs +++ b/nodedb/src/control/server/pgwire/handler/prepared/execute.rs @@ -148,6 +148,9 @@ impl NodeDbPgHandler { }) .collect(), is_star: false, + // The Describe-phase fields carry no expressions; the + // execute path merges the statement's own computed list in. + cp_computed: Vec::new(), }) }; diff --git a/nodedb/src/control/server/pgwire/handler/routing/calvin_response.rs b/nodedb/src/control/server/pgwire/handler/routing/calvin_response.rs index dd305ec63..bbb3fb614 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/calvin_response.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/calvin_response.rs @@ -81,6 +81,9 @@ pub(super) fn calvin_execution_response( database_id, tenant_id, redaction: Some(redaction.ctx(&state.redaction)), + // A RETURNING list names stored columns only, never a + // Control-Plane computed column. + sequences: None, }) { return Ok(CalvinTaskOutcome::Rows(shaped)); diff --git a/nodedb/src/control/server/pgwire/handler/routing/cluster_array.rs b/nodedb/src/control/server/pgwire/handler/routing/cluster_array.rs index 0024ffc96..b77b9a375 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/cluster_array.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/cluster_array.rs @@ -19,7 +19,7 @@ use crate::control::server::response_shape::redaction::QueryRedaction; use crate::control::server::response_shape::schema::OutputSchema; use crate::control::server::shared::session::SessionId; -use super::super::super::types::{error_to_sqlstate, sqlstate_error}; +use super::super::super::types::{error_to_sqlstate, shape_error_to_pg}; use super::super::core::NodeDbPgHandler; use super::super::plan::{PlanKind, payload_to_response}; use super::super::shape_encode; @@ -122,13 +122,16 @@ impl NodeDbPgHandler { }; let redaction = QueryRedaction::for_collections(tenant_id, auth, vec![(String::new(), array_name)]); + // A cluster array plan projects attribute names only, never a + // Control-Plane computed column, so no session sequence access. match compose::shape_payload_no_plan( &payload_bytes, cluster_plan_kind, projection, Some(redaction.ctx(&self.state.redaction)), + None, ) - .map_err(|e| sqlstate_error("XX000", e.message()))? + .map_err(|e| shape_error_to_pg(&e))? { ShapeOutcome::Rows(shaped) => { let (response, notice) = diff --git a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/finish.rs b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/finish.rs new file mode 100644 index 000000000..f8353a117 --- /dev/null +++ b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/finish.rs @@ -0,0 +1,103 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The statement's tail after every task ran: the folded `RETURNING` rows as +//! one result set, then the set-operation merge of the deferred payloads. + +use std::sync::Arc; + +use pgwire::api::results::{FieldFormat, Response}; +use pgwire::error::PgWireResult; + +use nodedb_physical::physical_task::PostSetOp; + +use crate::control::sequence::{SequenceAccess, SessionSequenceAccess, SessionSequenceValues}; +use crate::control::server::response_shape::redaction::QueryRedaction; +use crate::control::server::response_shape::schema::OutputSchema; +use crate::control::server::response_shape::types::ShapedRows; +use crate::control::server::shared::session::SessionId; +use crate::types::{DatabaseId, TenantId}; + +use super::super::super::core::NodeDbPgHandler; +use super::super::super::shape_encode; +use super::super::set_ops; + +/// What the dispatch loop accumulated for the statement's tail. +pub(super) struct StatementTail<'a> { + /// The statement's `RETURNING` rows, folded across every task. + pub(super) returning_rows: Option, + /// Per-branch payloads deferred for a set-operation merge. + pub(super) dedup_payloads: Vec>, + pub(super) dedup_set_op: PostSetOp, + pub(super) projection: Option<&'a OutputSchema>, + pub(super) result_formats: &'a [FieldFormat], + /// Redaction over the union of the branches' sources, present only when + /// the statement carries a set operation. + pub(super) set_op_redaction: Option, + /// This connection's `currval` map, for the projection's Control-Plane + /// computed columns. + pub(super) session_sequences: Option>, + /// The statement's database, from its first task; `None` for an empty + /// task list, which also defers no payload. + pub(super) statement_database_id: Option, + pub(super) tenant_id: TenantId, + pub(super) session_id: SessionId, +} + +impl NodeDbPgHandler { + /// Emit the statement's tail onto `responses`. + pub(super) fn finish_statement( + &self, + responses: &mut Vec, + tail: StatementTail<'_>, + ) -> PgWireResult<()> { + let StatementTail { + returning_rows, + dedup_payloads, + dedup_set_op, + projection, + result_formats, + set_op_redaction, + session_sequences, + statement_database_id, + tenant_id, + session_id, + } = tail; + + // The statement's RETURNING rows, as one result set. + if let Some(shaped) = returning_rows { + let (response, notice) = shape_encode::shaped_query_response(shaped, result_formats); + if let Some(n) = notice { + self.sessions.push_notice(session_id, n); + } + responses.push(response); + } + + // Set operations: merge sub-query payloads. + if !dedup_payloads.is_empty() { + let sequences = statement_database_id.map(|database_id| { + SessionSequenceAccess::for_session( + &self.state, + session_sequences, + database_id, + tenant_id, + ) + }); + let (response, notice) = set_ops::apply_set_ops( + &dedup_payloads, + dedup_set_op, + projection, + result_formats, + set_op_redaction + .as_ref() + .map(|r| r.ctx(&self.state.redaction)), + sequences.as_ref().map(|s| s as &dyn SequenceAccess), + )?; + if let Some(n) = notice { + self.sessions.push_notice(session_id, n); + } + responses.push(response); + } + + Ok(()) + } +} diff --git a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/mod.rs b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/mod.rs new file mode 100644 index 000000000..109d57c5b --- /dev/null +++ b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/mod.rs @@ -0,0 +1,9 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The per-task dispatch loop for non-Calvin pgwire queries. + +mod finish; +mod run; +mod task; + +pub(crate) use run::DispatchTaskContext; diff --git a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop.rs b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs similarity index 79% rename from nodedb/src/control/server/pgwire/handler/routing/dispatch_loop.rs rename to nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs index 01baa406c..5f46b7ef5 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs @@ -1,6 +1,10 @@ // SPDX-License-Identifier: BUSL-1.1 -//! The per-task dispatch loop for non-Calvin pgwire queries. +//! The per-task dispatch loop for non-Calvin pgwire queries: tenant check, +//! in-transaction routing, streaming fast path, pre-dispatch hooks, dispatch, +//! read tracking, AFTER triggers, and metering. Shaping one task's response +//! lives in `task.rs`; the statement's tail (folded RETURNING rows and the +//! set-op merge) lives in `finish.rs`. //! //! Split out of `execute.rs`, which keeps the plan/authorize/admit entry //! points and hands the admitted task list here. @@ -12,9 +16,7 @@ use pgwire::error::{ErrorInfo, PgWireError, PgWireResult}; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::security::request_scope::RequestAuthScope; -use crate::control::server::response_shape::compose::{self, ShapeOutcome}; use crate::control::server::response_shape::redaction::QueryRedaction; -use crate::control::server::response_shape::request::MaterializedShapeRequest; use crate::control::server::response_shape::types::ShapedRows; use crate::control::server::shared::ddl::neutral::maintenance::auto_analyze; use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; @@ -23,26 +25,29 @@ use crate::control::server::shared::session::SessionId; use crate::types::TenantId; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; -use super::super::super::types::{error_to_sqlstate, response_status_to_sqlstate, sqlstate_error}; -use super::super::core::NodeDbPgHandler; -use super::super::plan::{PlanKind, describe_plan, payload_to_response}; -use super::super::shape_encode; -use super::result_shaping::ResultShaping; -use super::set_ops; -use super::streaming::StreamSelectContext; +use super::super::super::super::types::{ + error_to_sqlstate, response_status_to_sqlstate, sqlstate_error, +}; +use super::super::super::core::NodeDbPgHandler; +use super::super::super::plan::{PlanKind, describe_plan}; +use super::super::execute_dml_hooks; +use super::super::result_shaping::ResultShaping; +use super::super::streaming::StreamSelectContext; +use super::finish::StatementTail; +use super::task::ShapeTaskParams; -pub(super) struct DispatchTaskContext<'a> { - pub(super) plan_lease_scope: Arc, - pub(super) tenant_id: TenantId, - pub(super) identity: &'a AuthenticatedIdentity, - pub(super) auth_ctx: &'a crate::control::security::auth_context::AuthContext, - pub(super) session_id: SessionId, - pub(super) shaping: ResultShaping<'a>, +pub(crate) struct DispatchTaskContext<'a> { + pub(crate) plan_lease_scope: Arc, + pub(crate) tenant_id: TenantId, + pub(crate) identity: &'a AuthenticatedIdentity, + pub(crate) auth_ctx: &'a crate::control::security::auth_context::AuthContext, + pub(crate) session_id: SessionId, + pub(crate) shaping: ResultShaping<'a>, } impl NodeDbPgHandler { /// Execute the per-task dispatch loop for non-Calvin queries. - pub(super) async fn dispatch_task_loop( + pub(crate) async fn dispatch_task_loop( &self, tasks: Vec, context: DispatchTaskContext<'_>, @@ -72,6 +77,11 @@ impl NodeDbPgHandler { // an extended-query client reads as several results for one statement. let mut returning_rows: Option = None; let mut responses = Vec::with_capacity(tasks.len()); + // Session-scoped sequence access for the statement's Control-Plane + // computed columns, resolved once: every task of one statement + // targets the same database, so the first task's names it. + let session_sequences = self.sessions.sequence_values(session_id); + let statement_database_id = tasks.first().map(|t| t.database_id); // Checked once rather than per task — metering is disabled by // default, so this keeps the per-task extraction below (which clones // the collection name) a true no-op on the hot path for every @@ -107,10 +117,10 @@ impl NodeDbPgHandler { .route_task_in_txn(session_id, identity, task, Arc::clone(&plan_lease_scope)) .await? { - super::execute_dml_hooks::TxnRouteOutcome::Proceed(routed_task) => { + execute_dml_hooks::TxnRouteOutcome::Proceed(routed_task) => { task = *routed_task; } - super::execute_dml_hooks::TxnRouteOutcome::Handled(resp) => { + execute_dml_hooks::TxnRouteOutcome::Handled(resp) => { if returns_rows { let (severity, code, message) = error_to_sqlstate( &crate::control::server::shared::returning:: @@ -222,7 +232,7 @@ impl NodeDbPgHandler { // under the size limit; behavior is unchanged). let (dml_info, old_row, truncate_restart_collection) = match self .run_pre_dispatch_hooks( - super::execute_dml_hooks::PreDispatchContext { + execute_dml_hooks::PreDispatchContext { identity, auth: auth_ctx, tenant_id, @@ -234,12 +244,12 @@ impl NodeDbPgHandler { ) .await? { - super::execute_dml_hooks::PreDispatchOutcome::Handled(resp) => { + execute_dml_hooks::PreDispatchOutcome::Handled(resp) => { responses.push(resp); continue; } - super::execute_dml_hooks::PreDispatchOutcome::Proceed(proceed) => { - let super::execute_dml_hooks::PreDispatchProceed { + execute_dml_hooks::PreDispatchOutcome::Proceed(proceed) => { + let execute_dml_hooks::PreDispatchProceed { task: proceeding_task, dml_info, old_row, @@ -384,53 +394,30 @@ impl NodeDbPgHandler { // set-op-deferred branch (its rows are only known after the // later cross-task merge) and for `Passthrough` (no row payload // to count); `meter_dispatch` charges one unit for `None`. - let mut task_rows: Option = None; - if needs_set_op && resp_post_set_op != PostSetOp::None { + let task_rows = if needs_set_op && resp_post_set_op != PostSetOp::None { dedup_payloads.push(resp.payload.to_vec()); if dedup_set_op == PostSetOp::None { dedup_set_op = resp_post_set_op; } + None } else { - let redaction = QueryRedaction::for_plan(tenant_id, auth_ctx, &plan_for_response); - match compose::shape_response_materialized(MaterializedShapeRequest { - payload: &resp.payload, - plan: &plan_for_response, - plan_kind, - projection, - state: &self.state, - database_id: task_database_id, - tenant_id, - redaction: Some(redaction.ctx(&self.state.redaction)), - }) - .map_err(|e| sqlstate_error("XX000", e.message()))? - { - ShapeOutcome::Rows(shaped) => { - task_rows = Some(shaped.rows.len() as u64); - if matches!(plan_kind, PlanKind::ReturningRows) { - // Folded, not emitted: the whole statement answers - // with one result set after the loop. - match returning_rows { - Some(ref mut accumulated) => accumulated.append(shaped), - None => returning_rows = Some(shaped), - } - } else { - let (response, notice) = - shape_encode::shaped_query_response(shaped, result_formats); - if let Some(n) = notice { - self.sessions.push_notice(session_id, n); - } - responses.push(response); - } - } - ShapeOutcome::Passthrough => { - let shaped = payload_to_response(&resp.payload, plan_kind)?; - if let Some(notice) = shaped.notice { - self.sessions.push_notice(session_id, notice); - } - responses.push(shaped.response); - } - } - } + self.shape_task_response( + ShapeTaskParams { + response: &resp, + plan: &plan_for_response, + plan_kind, + projection, + result_formats, + session_id, + tenant_id, + database_id: task_database_id, + auth_ctx, + session_sequences: session_sequences.clone(), + }, + &mut responses, + &mut returning_rows, + )? + }; // Metered here, once per successfully dispatched task — every // path reaching this point already passed the @@ -443,31 +430,21 @@ impl NodeDbPgHandler { } } - // The statement's RETURNING rows, as one result set. - if let Some(shaped) = returning_rows { - let (response, notice) = shape_encode::shaped_query_response(shaped, result_formats); - if let Some(n) = notice { - self.sessions.push_notice(session_id, n); - } - responses.push(response); - } - - // Set operations: merge sub-query payloads. - if needs_set_op && !dedup_payloads.is_empty() { - let (response, notice) = set_ops::apply_set_ops( - &dedup_payloads, + self.finish_statement( + &mut responses, + StatementTail { + returning_rows, + dedup_payloads, dedup_set_op, projection, result_formats, - set_op_redaction - .as_ref() - .map(|r| r.ctx(&self.state.redaction)), - )?; - if let Some(n) = notice { - self.sessions.push_notice(session_id, n); - } - responses.push(response); - } + set_op_redaction, + session_sequences, + statement_database_id, + tenant_id, + session_id, + }, + )?; Ok(responses) } diff --git a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/task.rs b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/task.rs new file mode 100644 index 000000000..f639afa0d --- /dev/null +++ b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/task.rs @@ -0,0 +1,115 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Shaping of one dispatched task's Data-Plane response into pgwire output. + +use std::sync::Arc; + +use pgwire::api::results::{FieldFormat, Response}; +use pgwire::error::PgWireResult; + +use crate::bridge::envelope::PhysicalPlan; +use crate::control::sequence::{SessionSequenceAccess, SessionSequenceValues}; +use crate::control::server::response_shape::compose::{self, ShapeOutcome}; +use crate::control::server::response_shape::redaction::QueryRedaction; +use crate::control::server::response_shape::request::MaterializedShapeRequest; +use crate::control::server::response_shape::schema::OutputSchema; +use crate::control::server::response_shape::types::ShapedRows; +use crate::control::server::shared::session::SessionId; +use crate::types::{DatabaseId, TenantId}; + +use super::super::super::super::types::shape_error_to_pg; +use super::super::super::core::NodeDbPgHandler; +use super::super::super::plan::{PlanKind, payload_to_response}; +use super::super::super::shape_encode; + +/// Everything needed to shape one task's response. +pub(super) struct ShapeTaskParams<'a> { + pub(super) response: &'a crate::bridge::envelope::Response, + pub(super) plan: &'a PhysicalPlan, + pub(super) plan_kind: PlanKind, + pub(super) projection: Option<&'a OutputSchema>, + pub(super) result_formats: &'a [FieldFormat], + pub(super) session_id: SessionId, + pub(super) tenant_id: TenantId, + pub(super) database_id: DatabaseId, + pub(super) auth_ctx: &'a crate::control::security::auth_context::AuthContext, + /// This connection's `currval` map, for the projection's Control-Plane + /// computed columns. + pub(super) session_sequences: Option>, +} + +impl NodeDbPgHandler { + /// Shape one task's response: rows are encoded and pushed onto + /// `responses`, except `RETURNING` rows, which fold into + /// `returning_rows` so the statement answers with one result set. + /// + /// Returns the task's own row count for metering, `None` for a + /// passthrough response with no row payload to count. + pub(super) fn shape_task_response( + &self, + params: ShapeTaskParams<'_>, + responses: &mut Vec, + returning_rows: &mut Option, + ) -> PgWireResult> { + let ShapeTaskParams { + response, + plan, + plan_kind, + projection, + result_formats, + session_id, + tenant_id, + database_id, + auth_ctx, + session_sequences, + } = params; + let redaction = QueryRedaction::for_plan(tenant_id, auth_ctx, plan); + let sequences = SessionSequenceAccess::for_session( + &self.state, + session_sequences, + database_id, + tenant_id, + ); + match compose::shape_response_materialized(MaterializedShapeRequest { + payload: &response.payload, + plan, + plan_kind, + projection, + state: &self.state, + database_id, + tenant_id, + redaction: Some(redaction.ctx(&self.state.redaction)), + sequences: Some(&sequences), + }) + .map_err(|e| shape_error_to_pg(&e))? + { + ShapeOutcome::Rows(shaped) => { + let task_rows = shaped.rows.len() as u64; + if matches!(plan_kind, PlanKind::ReturningRows) { + // Folded, not emitted: the whole statement answers with + // one result set after the loop. + match returning_rows { + Some(accumulated) => accumulated.append(shaped), + None => *returning_rows = Some(shaped), + } + } else { + let (encoded, notice) = + shape_encode::shaped_query_response(shaped, result_formats); + if let Some(n) = notice { + self.sessions.push_notice(session_id, n); + } + responses.push(encoded); + } + Ok(Some(task_rows)) + } + ShapeOutcome::Passthrough => { + let shaped = payload_to_response(&response.payload, plan_kind)?; + if let Some(notice) = shaped.notice { + self.sessions.push_notice(session_id, notice); + } + responses.push(shaped.response); + Ok(None) + } + } + } +} diff --git a/nodedb/src/control/server/pgwire/handler/routing/execute.rs b/nodedb/src/control/server/pgwire/handler/routing/execute.rs index c6139687f..f829e435a 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/execute.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/execute.rs @@ -145,9 +145,20 @@ impl NodeDbPgHandler { } // An externally-supplied prepared-statement schema (from the Describe - // phase) wins; otherwise use the planner's fresh output schema for this - // statement. - let effective_schema = shaping.projection.or(Some(&output_schema)); + // phase) names the columns; otherwise the planner's fresh output + // schema for this statement does. The Control-Plane computed list is + // known only to this statement's plan, so it rides along under the + // Describe-phase columns: the shaper evaluates it before those + // columns project, and a computed alias never renders as NULL. + let effective_schema_owned = match shaping.projection { + Some(described) => crate::control::server::response_shape::schema::OutputSchema { + columns: described.columns.clone(), + is_star: described.is_star, + cp_computed: output_schema.cp_computed, + }, + None => output_schema, + }; + let effective_schema = Some(&effective_schema_owned); // Implicit-edge dependent predicates must be preempted onto the // OLLP/Calvin path before gateway forwarding or ordinary dispatch. diff --git a/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs b/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs index 41cbb3a3e..a479362d8 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs @@ -21,7 +21,7 @@ use crate::control::trigger::dml_hook::DmlWriteInfo; use crate::types::TenantId; use nodedb_physical::physical_task::PhysicalTask; -use super::super::super::types::{error_to_sqlstate, sqlstate_error}; +use super::super::super::types::{error_to_sqlstate, shape_error_to_pg}; use super::super::core::NodeDbPgHandler; use super::super::plan::PlanKind; @@ -326,13 +326,16 @@ impl NodeDbPgHandler { // A clone write can carry RETURNING rows, which deliver // stored column values just as a SELECT does. let redaction = QueryRedaction::for_plan(tenant_id, auth, &task.plan); + // A clone write's RETURNING list names stored columns + // only, never a Control-Plane computed column. match shape_payload_no_plan( resp.payload.as_ref(), plan_kind, projection, Some(redaction.ctx(&self.state.redaction)), + None, ) - .map_err(|e| sqlstate_error("XX000", e.message()))? + .map_err(|e| shape_error_to_pg(&e))? { ShapeOutcome::Rows(shaped) => { // Clone write-path DML result (PointUpdate/PointDelete): diff --git a/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs b/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs index b7bdd9960..5fbacb3ff 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs @@ -22,7 +22,7 @@ use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch} use crate::types::{TenantId, TraceId}; use nodedb_physical::physical_task::PhysicalTask; -use super::super::super::types::sqlstate_error; +use super::super::super::types::shape_error_to_pg; use super::super::core::NodeDbPgHandler; use super::super::plan::{PlanKind, multirow_payload_to_response}; use super::super::shape_encode; @@ -69,13 +69,16 @@ fn push_shaped_response( responses.push(Response::Execution(Tag::new("OK"))); return Ok(()); } + // Gateway forwarding carries no session: a projection with Control-Plane + // computed columns is refused by the shaper rather than NULL-filled. match compose::shape_payload_no_plan( payload, PlanKind::MultiRow, projection, Some(redaction.ctx(&state.redaction)), + None, ) - .map_err(|e| sqlstate_error("XX000", e.message()))? + .map_err(|e| shape_error_to_pg(&e))? { ShapeOutcome::Rows(shaped) => { let (response, notice) = shape_encode::shaped_query_response(shaped, result_formats); @@ -227,8 +230,9 @@ impl NodeDbPgHandler { PlanKind::MultiRow, projection, Some(redaction.ctx(&self.state.redaction)), + None, ) - .map_err(|e| sqlstate_error("XX000", e.message()))? + .map_err(|e| shape_error_to_pg(&e))? { ShapeOutcome::Rows(shaped) => { task_rows = Some(task_rows.unwrap_or(0) + shaped.rows.len() as u64); diff --git a/nodedb/src/control/server/pgwire/handler/routing/result_shaping.rs b/nodedb/src/control/server/pgwire/handler/routing/result_shaping.rs index 3274b604a..4e2d2c2b9 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/result_shaping.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/result_shaping.rs @@ -7,7 +7,7 @@ use pgwire::api::results::FieldFormat; use crate::control::server::response_shape::schema::OutputSchema; #[derive(Clone, Copy)] -pub(in crate::control::server::pgwire::handler) struct ResultShaping<'a> { +pub(crate) struct ResultShaping<'a> { pub projection: Option<&'a OutputSchema>, pub formats: &'a [FieldFormat], } diff --git a/nodedb/src/control/server/pgwire/handler/routing/set_ops.rs b/nodedb/src/control/server/pgwire/handler/routing/set_ops.rs index 43d15f976..0d4909f12 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/set_ops.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/set_ops.rs @@ -8,6 +8,7 @@ use pgwire::error::PgWireResult; use nodedb_physical::physical_task::PostSetOp; +use crate::control::sequence::SequenceAccess; use crate::control::server::response_shape::compose::{self, ShapeOutcome}; use crate::control::server::response_shape::redaction::RedactionCtx; use crate::control::server::response_shape::schema::OutputSchema; @@ -15,18 +16,22 @@ use crate::control::server::set_op_merge::{ SetMergeMode, dedup_union_payloads, merge_set_op_payloads, }; -use super::super::super::types::sqlstate_error; +use super::super::super::types::shape_error_to_pg; use super::super::plan::{PlanKind, multirow_payload_to_response}; use super::super::shape_encode; /// Apply set operation merging to collected sub-query payloads, then shape and /// project the merged result into an already-encoded pgwire response. +/// +/// `sequences` resolves the projection's Control-Plane computed columns +/// over the merged rows. pub(super) fn apply_set_ops( dedup_payloads: &[Vec], dedup_set_op: PostSetOp, projection: Option<&OutputSchema>, result_formats: &[FieldFormat], redaction: Option>, + sequences: Option<&dyn SequenceAccess>, ) -> PgWireResult<(Response, Option)> { let merged = match dedup_set_op { PostSetOp::Intersect | PostSetOp::IntersectAll => { @@ -38,8 +43,14 @@ pub(super) fn apply_set_ops( _ => dedup_union_payloads(dedup_payloads), }; Ok( - match compose::shape_payload_no_plan(&merged, PlanKind::MultiRow, projection, redaction) - .map_err(|e| sqlstate_error("XX000", e.message()))? + match compose::shape_payload_no_plan( + &merged, + PlanKind::MultiRow, + projection, + redaction, + sequences, + ) + .map_err(|e| shape_error_to_pg(&e))? { ShapeOutcome::Rows(shaped) => { shape_encode::shaped_query_response(shaped, result_formats) diff --git a/nodedb/src/control/server/pgwire/handler/routing/streaming.rs b/nodedb/src/control/server/pgwire/handler/routing/streaming.rs index 01cfdc0dc..c52b5b697 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/streaming.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/streaming.rs @@ -22,7 +22,7 @@ use crate::control::server::response_shape::redaction::QueryRedaction; use crate::control::server::shared::session::SessionId; use super::super::super::types::error_to_sqlstate; -use super::super::super::types::sqlstate_error; +use super::super::super::types::shape_error_to_pg; use super::super::core::NodeDbPgHandler; use super::super::plan::{PlanKind, multirow_payload_to_response}; use super::super::stream_response; @@ -47,6 +47,9 @@ impl NodeDbPgHandler { /// /// Eligibility (all required): /// - no post-set-op (UNION/INTERSECT/EXCEPT need the full sets), + /// - no Control-Plane computed column in the projection + /// (`cp_computed` needs per-row session sequence access, which the + /// per-batch shaper does not carry), /// - `PlanKind::MultiRow` (the streamed shape is one TEXT column), /// - autocommit (not inside a BEGIN..COMMIT block — in-block reads /// participate in snapshot-isolation read tracking on the normal path), @@ -69,6 +72,9 @@ impl NodeDbPgHandler { lease_scope, } = context; if task.post_set_op != PostSetOp::None + || shaping + .projection + .is_some_and(|s| !s.cp_computed.is_empty()) || !matches!(plan_kind, PlanKind::MultiRow) || self.sessions.transaction_state(session_id) == crate::control::server::shared::session::TransactionState::InBlock @@ -135,13 +141,16 @@ impl NodeDbPgHandler { Response::Execution(Tag::new("OK")) } else { let redaction = QueryRedaction::for_plan(task.tenant_id, auth, &child_plan); + // The gate above admits no Control-Plane computed + // column, so no session sequence access is needed. match compose::shape_payload_no_plan( payload, plan_kind, shaping.projection, Some(redaction.ctx(&state.redaction)), + None, ) - .map_err(|e| sqlstate_error("XX000", e.message()))? + .map_err(|e| shape_error_to_pg(&e))? { ShapeOutcome::Rows(shaped) => { let (response, _notice) = crate::control::server::pgwire::handler::shape_encode::shaped_query_response( diff --git a/nodedb/src/control/server/pgwire/handler/shape_encode/cell.rs b/nodedb/src/control/server/pgwire/handler/shape_encode/cell.rs index 19bb7429d..fb5a2004b 100644 --- a/nodedb/src/control/server/pgwire/handler/shape_encode/cell.rs +++ b/nodedb/src/control/server/pgwire/handler/shape_encode/cell.rs @@ -25,7 +25,7 @@ use nodedb_types::error::NodeDbError; use nodedb_types::{NdbDateTime, Value}; use crate::control::server::pgwire::numeric_narrow::{checked_narrow, checked_narrow_f32}; -use crate::control::server::pgwire::types::error_map::{numeric_code_to_sqlstate, sqlstate_error}; +use crate::control::server::pgwire::types::error_map::shape_error_to_pg; use crate::control::server::response_shape::cell::{cell_text, instant_of, shape_mismatch}; use crate::control::server::response_shape::types::DdlColType; @@ -169,7 +169,7 @@ fn float_of(column: &str, v: &Value) -> PgWireResult { /// Map a cell error to the pgwire error the client reads, with the SQLSTATE /// its numeric code maps to. fn to_pg_error(e: NodeDbError) -> PgWireError { - sqlstate_error(numeric_code_to_sqlstate(e.code()), e.message()) + shape_error_to_pg(&e) } /// An instant as pgwire encodes it under a `timestamp`/`timestamptz` diff --git a/nodedb/src/control/server/pgwire/handler/stream_response.rs b/nodedb/src/control/server/pgwire/handler/stream_response.rs index 847242dfb..a63f7c5ff 100644 --- a/nodedb/src/control/server/pgwire/handler/stream_response.rs +++ b/nodedb/src/control/server/pgwire/handler/stream_response.rs @@ -223,10 +223,14 @@ pub(crate) fn streaming_shaped_response( })?; // Resolved once before the first batch was pulled; this only // re-borrows it, so no batch can slip out ahead of the policy. + // A streamed plan never carries Control-Plane computed columns: + // `maybe_stream_select` declines those, so no session access is + // needed per batch. let shaped = shape_decoded_rows( value, Some(&schema_out), redaction.as_ref().map(|r| r.ctx(&state.redaction)), + None, ) .map_err(|e| { PgWireError::UserError(Box::new(ErrorInfo::new( @@ -341,6 +345,7 @@ pub(crate) async fn streaming_star_response( nodedb_types::Value::Array(values), None, redaction.as_ref().map(|r| r.ctx(&state.redaction)), + None, ) { Ok(s) => s, Err(e) => { diff --git a/nodedb/src/control/server/pgwire/types/error_map.rs b/nodedb/src/control/server/pgwire/types/error_map.rs index 976634640..8beed245b 100644 --- a/nodedb/src/control/server/pgwire/types/error_map.rs +++ b/nodedb/src/control/server/pgwire/types/error_map.rs @@ -17,6 +17,14 @@ pub fn sqlstate_error(code: &str, message: &str) -> PgWireError { ))) } +/// Map an error raised while shaping a response to the pgwire error the +/// client reads, with the SQLSTATE its numeric code maps to. A per-row +/// sequence accessor refusal (`42704`, `55000`) or a division by zero +/// (`22012`) keeps its class instead of collapsing to `XX000`. +pub fn shape_error_to_pg(e: &nodedb_types::NodeDbError) -> PgWireError { + sqlstate_error(numeric_code_to_sqlstate(e.code()), e.message()) +} + /// Map a NodeDB `Error` to a PostgreSQL SQLSTATE code + message. pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, String) { match err { @@ -215,6 +223,10 @@ pub(crate) fn numeric_code_to_sqlstate(code: nodedb_types::error::ErrorCode) -> Ec::BAD_REQUEST | Ec::PLAN_ERROR => sqlstate::SYNTAX_ERROR, // Mirrors the `UndefinedFunction` arm. Ec::UNDEFINED_FUNCTION => sqlstate::UNDEFINED_FUNCTION, + // Mirrors the `UndefinedObject` arm. + Ec::UNDEFINED_OBJECT => sqlstate::UNDEFINED_OBJECT, + // Mirrors the `ObjectNotInPrerequisiteState` arm. + Ec::OBJECT_NOT_READY => sqlstate::OBJECT_NOT_IN_PREREQUISITE_STATE, // Mirrors the `UndefinedColumn` arm. Ec::UNDEFINED_COLUMN => sqlstate::UNDEFINED_COLUMN, // Mirrors the `AmbiguousColumn` arm. diff --git a/nodedb/src/control/server/pgwire/types/mod.rs b/nodedb/src/control/server/pgwire/types/mod.rs index e4796724b..e527819fd 100644 --- a/nodedb/src/control/server/pgwire/types/mod.rs +++ b/nodedb/src/control/server/pgwire/types/mod.rs @@ -11,7 +11,8 @@ pub mod parse; pub mod privilege; pub use error_map::{ - error_to_sqlstate, notice_warning, response_status_to_sqlstate, sqlstate_error, + error_to_sqlstate, notice_warning, response_status_to_sqlstate, shape_error_to_pg, + sqlstate_error, }; pub use field::{ bool_field, bytea_field, float4_array_field, float4_field, float8_array_field, float8_field, diff --git a/nodedb/src/control/server/response_shape/compose/array_slice.rs b/nodedb/src/control/server/response_shape/compose/array_slice.rs index e9d921ea6..685ddc7ce 100644 --- a/nodedb/src/control/server/response_shape/compose/array_slice.rs +++ b/nodedb/src/control/server/response_shape/compose/array_slice.rs @@ -20,9 +20,9 @@ const TRUNCATED_BEFORE_HORIZON_NOTICE: &str = "AS OF SYSTEM TIME cutoff is older /// notice. /// /// Array slices never carry a SELECT-list projection, so `shape_decoded_rows` -/// is always called with a `None` projection here — but redaction still -/// applies to the cells. A payload that decodes to no value shapes as an -/// empty result set. +/// is always called with a `None` projection and no sequence access here — +/// but redaction still applies to the cells. A payload that decodes to no +/// value shapes as an empty result set. pub(super) fn shape_array_slice( payload: &[u8], redaction: Option>, @@ -41,7 +41,7 @@ pub(super) fn shape_array_slice( let notice = truncated.then(|| TRUNCATED_BEFORE_HORIZON_NOTICE.to_string()); let mut shaped = match rows { - Ok(value) => shape_decoded_rows(value, None, redaction)?, + Ok(value) => shape_decoded_rows(value, None, redaction, None)?, Err(_) => empty_shaped(), }; shaped.notice = notice; diff --git a/nodedb/src/control/server/response_shape/compose/kernel.rs b/nodedb/src/control/server/response_shape/compose/kernel.rs index d0c0942e3..1853049c8 100644 --- a/nodedb/src/control/server/response_shape/compose/kernel.rs +++ b/nodedb/src/control/server/response_shape/compose/kernel.rs @@ -12,9 +12,12 @@ use std::collections::HashSet; use nodedb_types::Value; use nodedb_types::columnar::schema::is_reserved_bitemporal_column; +use crate::control::sequence::SequenceAccess; + use super::super::project::push_flat_rows; use super::super::redaction::RedactionCtx; use super::super::schema::OutputSchema; +use super::super::stamp::stamp_rows; use super::super::types::{DdlColType, ShapedRow, ShapedRows}; /// Pure shaping core: given an already-decoded Data-Plane value, unwrap the @@ -27,10 +30,15 @@ use super::super::types::{DdlColType, ShapedRow, ShapedRows}; /// function does none of that. A streamed scan batch has no plan to KV-wrap /// or vector-translate but still needs the same envelope-unwrap + projection /// logic applied per batch, so streaming callers call this directly. +/// +/// `sequences` resolves the projection's Control-Plane computed columns +/// (`cp_computed`). A projection that carries any and a caller that passes +/// `None` is an error: the alias would otherwise project as NULL. pub fn shape_decoded_rows( decoded: Value, projection: Option<&OutputSchema>, redaction: Option>, + sequences: Option<&dyn SequenceAccess>, ) -> crate::Result { let mut rows = Vec::new(); push_flat_rows(decoded, &mut rows)?; @@ -45,6 +53,12 @@ pub fn shape_decoded_rows( // says to withhold, so the hook belongs exactly here. redact_rows(redaction.as_ref(), &mut rows); + // Control-Plane computed columns are stamped onto the flat rows AFTER + // redaction (an expression reads the values the policy lets through) + // and BEFORE projection (the projection drops the pass-through base + // columns the expression reads and keeps only the alias). + stamp_computed_columns(projection, &mut rows, sequences)?; + match projection { Some(s) if !s.is_star && !s.columns.is_empty() => { let lookup_keys: Vec = s.columns.iter().map(|c| c.lookup_key.clone()).collect(); @@ -83,6 +97,29 @@ pub fn shape_decoded_rows( } } +/// Stamp the projection's Control-Plane computed columns onto the flat rows. +/// +/// A projection with no computed column is a no-op whatever `sequences` is. +/// One that carries any needs session sequence access: a caller with none +/// (a per-batch stream, gateway forwarding, a clone merge) cannot answer +/// the statement, and says so rather than shipping NULL under the alias. +pub(in crate::control::server::response_shape) fn stamp_computed_columns( + projection: Option<&OutputSchema>, + rows: &mut [ShapedRow], + sequences: Option<&dyn SequenceAccess>, +) -> crate::Result<()> { + let Some(schema) = projection.filter(|s| !s.cp_computed.is_empty()) else { + return Ok(()); + }; + let Some(access) = sequences else { + return Err(crate::Error::FeatureNotSupported { + detail: "Control-Plane computed columns need session sequence access on this path" + .to_string(), + }); + }; + stamp_rows(rows, &schema.cp_computed, access) +} + /// Apply the statement's column-level redaction policy to every flat row. /// /// A `None` context means the producer has no requester identity in scope and @@ -281,6 +318,7 @@ mod tests { }) .collect(), is_star: false, + cp_computed: Vec::new(), } } @@ -306,7 +344,7 @@ mod tests { let sources = vec![(String::new(), "users".to_string())]; let decoded = one_row(&[("email", text("a@b.c")), ("name", text("Alice"))]); - let shaped = shape_decoded_rows(decoded, None, Some(ctx(&store, &roles, &sources))) + let shaped = shape_decoded_rows(decoded, None, Some(ctx(&store, &roles, &sources)), None) .expect("shape rows"); assert_eq!(shaped.rows[0]["email"], text("***")); assert_eq!(shaped.rows[0]["name"], text("Alice")); @@ -326,8 +364,8 @@ mod tests { let sources = vec![(String::new(), "users".to_string())]; let decoded = one_row(&[("email", text("a@b.c")), ("name", text("Alice"))]); - let baseline = shape_decoded_rows(decoded.clone(), None, None).expect("shape rows"); - let shaped = shape_decoded_rows(decoded, None, Some(ctx(&store, &roles, &sources))) + let baseline = shape_decoded_rows(decoded.clone(), None, None, None).expect("shape rows"); + let shaped = shape_decoded_rows(decoded, None, Some(ctx(&store, &roles, &sources)), None) .expect("shape rows"); assert_eq!(shaped.rows, baseline.rows); assert_eq!(shaped.columns, baseline.columns); @@ -352,6 +390,7 @@ mod tests { decoded, Some(&projection), Some(ctx(&store, &roles, &sources)), + None, ) .expect("shape rows"); assert_eq!(shaped.columns, vec!["contact".to_string()]); @@ -380,6 +419,7 @@ mod tests { decoded, Some(&projection), Some(ctx(&store, &roles, &sources)), + None, ) .expect("shape rows"); // `cell_keys` suffixes the duplicate display name. @@ -401,7 +441,7 @@ mod tests { let sources = vec![(String::new(), "users".to_string())]; let decoded = one_row(&[("id", text("u1")), ("email", text("a@b.c"))]); - let shaped = shape_decoded_rows(decoded, None, Some(ctx(&store, &roles, &sources))) + let shaped = shape_decoded_rows(decoded, None, Some(ctx(&store, &roles, &sources)), None) .expect("shape rows"); assert!( shaped.columns.contains(&"email".to_string()), @@ -434,7 +474,8 @@ mod tests { let decoded = one_row(&[("at", Value::NaiveDateTime(at)), ("id", text("r1"))]); let projection = named_projection(&[("at", "at")]); - let shaped = shape_decoded_rows(decoded, Some(&projection), None).expect("shape rows"); + let shaped = + shape_decoded_rows(decoded, Some(&projection), None, None).expect("shape rows"); assert_eq!(shaped.rows[0]["at"], Value::NaiveDateTime(at)); assert_eq!( crate::control::server::response_shape::cell::value_to_wire_json(&shaped.rows[0]["at"]), @@ -489,4 +530,59 @@ mod tests { assert_eq!(derive_columns(&rows), vec!["id", "name"]); } + + // ── Control-Plane computed columns ────────────────────────────────── + + /// Answers every accessor with a fixed value. + struct FixedAccess(i64); + + impl SequenceAccess for FixedAccess { + fn nextval(&self, _name: &str) -> crate::Result { + Ok(self.0) + } + fn currval(&self, _name: &str) -> crate::Result { + Ok(self.0) + } + fn setval(&self, _name: &str, value: i64) -> crate::Result { + Ok(value) + } + } + + /// `SELECT id, nextval('s') AS n FROM t`: the projection announces `n`, + /// the row carries only `id`, and the stamped value projects under `n`. + fn cp_computed_projection() -> OutputSchema { + let mut projection = named_projection(&[("id", "id"), ("n", "n")]); + projection.cp_computed = vec![ + crate::control::server::response_shape::schema::CpComputedColumn { + alias: "n".to_string(), + expr: crate::bridge::expr_eval::SqlExpr::Function { + name: "nextval".to_string(), + args: vec![crate::bridge::expr_eval::SqlExpr::Literal(text("s"))], + }, + }, + ]; + projection + } + + #[test] + fn a_computed_column_is_stamped_before_projection() { + let decoded = one_row(&[("id", text("r1"))]); + let projection = cp_computed_projection(); + let access = FixedAccess(41); + + let shaped = shape_decoded_rows(decoded, Some(&projection), None, Some(&access)) + .expect("shape rows"); + assert_eq!(shaped.columns, vec!["id".to_string(), "n".to_string()]); + assert_eq!(shaped.rows[0]["n"], Value::Integer(41)); + } + + #[test] + fn a_computed_column_without_sequence_access_is_an_error_not_a_null() { + let decoded = one_row(&[("id", text("r1"))]); + let projection = cp_computed_projection(); + + let err = + shape_decoded_rows(decoded, Some(&projection), None, None).expect_err("must refuse"); + assert!(matches!(err, crate::Error::FeatureNotSupported { .. })); + } } diff --git a/nodedb/src/control/server/response_shape/compose/materialized.rs b/nodedb/src/control/server/response_shape/compose/materialized.rs index 6c5b30142..b454ed801 100644 --- a/nodedb/src/control/server/response_shape/compose/materialized.rs +++ b/nodedb/src/control/server/response_shape/compose/materialized.rs @@ -19,6 +19,7 @@ //! is shared with per-batch lazy streaming callers, which have an //! already-decoded batch and only need the envelope-unwrap + projection logic. +use crate::control::sequence::SequenceAccess; use crate::control::server::response_translate::dispatch::translate_search_response; use crate::data::executor::response_codec::decode_payload_value; use nodedb_types::NodeDbError; @@ -60,6 +61,7 @@ pub fn shape_response_materialized( database_id, tenant_id, redaction, + sequences, } = request; match plan_kind { @@ -79,9 +81,11 @@ pub fn shape_response_materialized( PlanKind::ArraySlice => shape_array_slice(&translated, redaction)?, // `RETURNING` rows are held to the columns already announced to the // client, when any were — see `super::returning`. - PlanKind::ReturningRows => shape_returning_rows(&translated, projection, redaction)?, + PlanKind::ReturningRows => { + shape_returning_rows(&translated, projection, redaction, sequences)? + } PlanKind::SingleDocument | PlanKind::MultiRow => { - shape_generic_rows(&translated, projection, redaction)? + shape_generic_rows(&translated, projection, redaction, sequences)? } // Handled by the early return above; kept exhaustive (no catch-all, // no panic) so a future PlanKind desync degrades to passthrough @@ -99,21 +103,26 @@ pub fn shape_response_materialized( /// scan-envelope unwrap + optional SELECT-list projection steps, skipping the /// plan-dependent `apply_kv_wrap` / `translate_search_response` transforms those /// callers never ran. +/// +/// `sequences` resolves the projection's Control-Plane computed columns; a +/// caller with no session in scope passes `None`, and a projection that +/// carries computed columns then fails rather than shipping NULL. pub fn shape_payload_no_plan( payload: &[u8], plan_kind: PlanKind, projection: Option<&OutputSchema>, redaction: Option>, + sequences: Option<&dyn SequenceAccess>, ) -> Result { Ok(match plan_kind { PlanKind::Execution | PlanKind::DmlResult(_) => ShapeOutcome::Passthrough, PlanKind::ArraySlice => ShapeOutcome::Rows(shape_array_slice(payload, redaction)?), - PlanKind::ReturningRows => { - ShapeOutcome::Rows(shape_returning_rows(payload, projection, redaction)?) - } - PlanKind::SingleDocument | PlanKind::MultiRow => { - ShapeOutcome::Rows(shape_generic_rows(payload, projection, redaction)?) - } + PlanKind::ReturningRows => ShapeOutcome::Rows(shape_returning_rows( + payload, projection, redaction, sequences, + )?), + PlanKind::SingleDocument | PlanKind::MultiRow => ShapeOutcome::Rows(shape_generic_rows( + payload, projection, redaction, sequences, + )?), }) } @@ -127,12 +136,13 @@ fn shape_generic_rows( payload: &[u8], projection: Option<&OutputSchema>, redaction: Option>, + sequences: Option<&dyn SequenceAccess>, ) -> crate::Result { if payload.is_empty() { return Ok(empty_shaped()); } match decode_payload_value(payload) { - Ok(value) => shape_decoded_rows(value, projection, redaction), + Ok(value) => shape_decoded_rows(value, projection, redaction, sequences), Err(_) => Ok(single_result_row( String::from_utf8_lossy(payload).into_owned(), )), @@ -203,6 +213,7 @@ mod tests { database_id: DatabaseId::new(1), tenant_id: TenantId::new(1), redaction: None, + sequences: None, }) .expect("execution plan passthrough"); assert!(matches!(materialized, ShapeOutcome::Passthrough)); @@ -216,7 +227,7 @@ mod tests { expected ); - let no_plan = shape_payload_no_plan(&payload, kind, None, None); + let no_plan = shape_payload_no_plan(&payload, kind, None, None, None); assert!(matches!(no_plan, Ok(ShapeOutcome::Passthrough))); assert_eq!( payload, original_payload, @@ -244,7 +255,7 @@ mod tests { .expect("encode"); let ShapeOutcome::Rows(shaped) = - shape_payload_no_plan(&payload, PlanKind::MultiRow, None, None).expect("shape") + shape_payload_no_plan(&payload, PlanKind::MultiRow, None, None, None).expect("shape") else { panic!("multi-row plan must yield rows"); }; diff --git a/nodedb/src/control/server/response_shape/mod.rs b/nodedb/src/control/server/response_shape/mod.rs index 01f1b6b08..89735375b 100644 --- a/nodedb/src/control/server/response_shape/mod.rs +++ b/nodedb/src/control/server/response_shape/mod.rs @@ -10,6 +10,7 @@ pub mod redaction; pub mod request; pub mod returning; pub mod schema; +pub mod stamp; pub mod types; pub use cell::{row_to_wire_json, value_to_wire_json}; diff --git a/nodedb/src/control/server/response_shape/request.rs b/nodedb/src/control/server/response_shape/request.rs index bcccc0c28..9e66c00c0 100644 --- a/nodedb/src/control/server/response_shape/request.rs +++ b/nodedb/src/control/server/response_shape/request.rs @@ -8,6 +8,7 @@ //! both unreadable and easy to transpose at a call site. use crate::bridge::envelope::PhysicalPlan; +use crate::control::sequence::SequenceAccess; use crate::control::state::SharedState; use nodedb_types::{DatabaseId, TenantId}; @@ -31,4 +32,9 @@ pub struct MaterializedShapeRequest<'a> { /// Column-level redaction for this statement, resolved once per query. /// `None` only where the producer has no requester identity at all. pub redaction: Option>, + /// Session-scoped sequence access for the projection's Control-Plane + /// computed columns. `None` only where the producer has no session in + /// scope; a projection that carries computed columns then fails rather + /// than shipping NULL under the alias. + pub sequences: Option<&'a dyn SequenceAccess>, } diff --git a/nodedb/src/control/server/response_shape/returning.rs b/nodedb/src/control/server/response_shape/returning.rs index e9450ebed..4483fc880 100644 --- a/nodedb/src/control/server/response_shape/returning.rs +++ b/nodedb/src/control/server/response_shape/returning.rs @@ -38,7 +38,9 @@ use nodedb_types::{NativeCell, NdbDateTime, NodeDbError, Value}; use crate::data::executor::response_codec::{RowsPayload, decode_payload_to_json}; -use super::compose::kernel::{project_row, redact_rows, single_result_row}; +use crate::control::sequence::SequenceAccess; + +use super::compose::kernel::{project_row, redact_rows, single_result_row, stamp_computed_columns}; use super::project::cell_keys; use super::redaction::RedactionCtx; use super::schema::OutputSchema; @@ -48,11 +50,14 @@ use super::types::{DdlColType, ShapedRow, ShapedRows}; /// /// `projection` carries the columns already announced to the client for this /// statement, when any were. See the module docs for why they win over the -/// payload's own list. +/// payload's own list. `sequences` resolves the projection's Control-Plane +/// computed columns; a projection that carries any and no access is an +/// error, never a NULL cell. pub fn shape_returning_rows( payload: &[u8], projection: Option<&OutputSchema>, redaction: Option>, + sequences: Option<&dyn SequenceAccess>, ) -> Result { let announced = announced_columns(projection); @@ -95,6 +100,9 @@ pub fn shape_returning_rows( // does, so the same redaction applies — and it runs on the payload's own // names, before any projection renames or drops them. redact_rows(redaction.as_ref(), &mut rows); + // Stamped after redaction and before the announced projection drops the + // pass-through columns an expression reads. + stamp_computed_columns(projection, &mut rows, sequences)?; let Some(schema) = announced else { let column_types = ShapedRows::text_types(columns.len()); @@ -286,6 +294,7 @@ mod tests { }) .collect(), is_star: false, + cp_computed: Vec::new(), } } @@ -298,7 +307,7 @@ mod tests { &["id", "name", "score"], &[&[Some("a"), Some("x"), Some("1")]], ); - let shaped = shape_returning_rows(&bytes, None, None).expect("shape"); + let shaped = shape_returning_rows(&bytes, None, None, None).expect("shape"); assert_eq!(shaped.columns, ["id", "name", "score"]); assert_eq!(shaped.rows.len(), 1); } @@ -312,7 +321,7 @@ mod tests { &[&[Some("a"), Some("x"), Some("surprise")]], ); let schema = announced(&[("id", DdlColType::Text), ("name", DdlColType::Text)]); - let shaped = shape_returning_rows(&bytes, Some(&schema), None).expect("shape"); + let shaped = shape_returning_rows(&bytes, Some(&schema), None, None).expect("shape"); assert_eq!(shaped.columns, ["id", "name"]); assert_eq!(shaped.column_types.len(), shaped.columns.len()); let keys = shaped.cell_keys(); @@ -331,7 +340,7 @@ mod tests { ("name", DdlColType::Text), ("score", DdlColType::Int8), ]); - let shaped = shape_returning_rows(&bytes, Some(&schema), None).expect("shape"); + let shaped = shape_returning_rows(&bytes, Some(&schema), None, None).expect("shape"); assert_eq!(shaped.columns, ["id", "name", "score"]); assert_eq!(shaped.rows[0]["id"], text("a")); assert_eq!(shaped.rows[0]["name"], Value::Null); @@ -352,7 +361,7 @@ mod tests { ("b", DdlColType::Bool), ("s", DdlColType::Text), ]); - let shaped = shape_returning_rows(&bytes, Some(&schema), None).expect("shape"); + let shaped = shape_returning_rows(&bytes, Some(&schema), None, None).expect("shape"); assert_eq!(shaped.rows[0]["i"], Value::Integer(42)); assert_eq!(shaped.rows[0]["f"], Value::Float(1.5)); assert_eq!(shaped.rows[0]["b"], Value::Bool(true)); @@ -379,7 +388,7 @@ mod tests { ("tstz", DdlColType::Timestamptz), ("digits", DdlColType::Timestamp), ]); - let shaped = shape_returning_rows(&bytes, Some(&schema), None).expect("shape"); + let shaped = shape_returning_rows(&bytes, Some(&schema), None, None).expect("shape"); assert_eq!(shaped.rows[0]["ts"], Value::NaiveDateTime(at)); assert_eq!(shaped.rows[0]["tstz"], Value::DateTime(at)); assert_eq!(shaped.rows[0]["digits"], text("1583402400000000")); @@ -407,7 +416,7 @@ mod tests { ("f", DdlColType::Float8), ("gone", DdlColType::Text), ]); - let shaped = shape_returning_rows(&bytes, Some(&schema), None).expect("shape"); + let shaped = shape_returning_rows(&bytes, Some(&schema), None, None).expect("shape"); assert_eq!(shaped.rows[0]["n"], Value::Integer(7)); assert_eq!(shaped.rows[0]["at"], Value::NaiveDateTime(at)); assert_eq!( @@ -425,7 +434,7 @@ mod tests { fn an_instant_cell_renders_iso8601_without_a_projection() { let at = NdbDateTime::from_micros(1_583_402_400_000_000); let bytes = typed_payload(&["at"], vec![vec![Value::DateTime(at)]]); - let shaped = shape_returning_rows(&bytes, None, None).expect("shape"); + let shaped = shape_returning_rows(&bytes, None, None, None).expect("shape"); assert_eq!(shaped.rows[0]["at"], Value::DateTime(at)); assert_eq!( value_to_wire_json(&shaped.rows[0]["at"]), @@ -439,7 +448,7 @@ mod tests { fn an_unparseable_cell_stays_text() { let bytes = payload(&["i"], &[&[Some("not a number")]]); let schema = announced(&[("i", DdlColType::Int8)]); - let shaped = shape_returning_rows(&bytes, Some(&schema), None).expect("shape"); + let shaped = shape_returning_rows(&bytes, Some(&schema), None, None).expect("shape"); assert_eq!(shaped.rows[0]["i"], text("not a number")); } @@ -448,7 +457,7 @@ mod tests { #[test] fn an_empty_payload_keeps_the_announced_columns() { let schema = announced(&[("id", DdlColType::Text), ("score", DdlColType::Int8)]); - let shaped = shape_returning_rows(&[], Some(&schema), None).expect("shape"); + let shaped = shape_returning_rows(&[], Some(&schema), None, None).expect("shape"); assert_eq!(shaped.columns, ["id", "score"]); assert!(shaped.rows.is_empty()); } @@ -459,7 +468,7 @@ mod tests { fn a_rowless_payload_keeps_the_announced_columns() { let bytes = payload(&[], &[]); let schema = announced(&[("id", DdlColType::Text)]); - let shaped = shape_returning_rows(&bytes, Some(&schema), None).expect("shape"); + let shaped = shape_returning_rows(&bytes, Some(&schema), None, None).expect("shape"); assert_eq!(shaped.columns, ["id"]); assert!(shaped.rows.is_empty()); } @@ -470,14 +479,14 @@ mod tests { #[test] fn a_malformed_payload_fails_when_columns_were_announced() { let schema = announced(&[("id", DdlColType::Text)]); - assert!(shape_returning_rows(&[0xFF, 0xFE], Some(&schema), None).is_err()); + assert!(shape_returning_rows(&[0xFF, 0xFE], Some(&schema), None, None).is_err()); } /// With nothing announced there is no contract to violate, so the legacy /// single-column fallback still applies. #[test] fn a_malformed_payload_falls_back_when_nothing_was_announced() { - let shaped = shape_returning_rows(&[0xFF, 0xFE], None, None).expect("fallback"); + let shaped = shape_returning_rows(&[0xFF, 0xFE], None, None, None).expect("fallback"); assert_eq!(shaped.columns, ["result"]); } } diff --git a/nodedb/src/control/server/response_shape/schema.rs b/nodedb/src/control/server/response_shape/schema.rs index 63a5b6784..62e6bdb8c 100644 --- a/nodedb/src/control/server/response_shape/schema.rs +++ b/nodedb/src/control/server/response_shape/schema.rs @@ -21,15 +21,30 @@ pub struct OutputColumn { pub ty: super::types::DdlColType, } +/// A SELECT-list column the Control Plane computes per output row after +/// the Data Plane returns the rows. `expr` is the bridge form; sequence +/// accessor calls inside it are resolved against the session's registry. +#[derive(Debug, Clone)] +pub struct CpComputedColumn { + /// The output name the evaluated value is written under. + pub alias: String, + /// The expression to evaluate against the flat row. + pub expr: crate::bridge::expr_eval::SqlExpr, +} + /// The authoritative output schema of a query, resolved by the planner. /// /// `columns` is the ordered projected column list. `is_star` marks a /// `SELECT *` whose concrete columns are only known from the returned rows -/// (id-first union derivation still applies for that case). +/// (id-first union derivation still applies for that case). `cp_computed` +/// lists the columns the response shaper evaluates on the Control Plane +/// before the projection runs; each one also appears in `columns` under its +/// alias. #[derive(Clone, Debug, Default)] pub struct OutputSchema { pub columns: Vec, pub is_star: bool, + pub cp_computed: Vec, } /// Maps the planner's resolved SQL column type to the response shaper's diff --git a/nodedb/src/control/server/response_shape/stamp/evaluate.rs b/nodedb/src/control/server/response_shape/stamp/evaluate.rs new file mode 100644 index 000000000..403aa28d0 --- /dev/null +++ b/nodedb/src/control/server/response_shape/stamp/evaluate.rs @@ -0,0 +1,238 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Evaluates the Control-Plane computed columns of a result set, row by row, +//! and writes each value into the flat row under its alias. +//! +//! Rows run in slice order and columns in SELECT-list order. The row document +//! an expression sees is built once per row and updated with each stamped +//! column's alias, so a later column reads an earlier alias exactly as SQL +//! left-to-right evaluation does. + +use std::collections::HashMap; + +use nodedb_types::Value; + +use crate::control::sequence::SequenceAccess; +use crate::control::server::response_shape::schema::CpComputedColumn; +use crate::control::server::response_shape::types::ShapedRow; + +use super::substitute::resolve_accessors; + +/// Stamp every computed column onto every row. +/// +/// A sequence accessor error keeps its own class (`42704` for an unknown +/// sequence, `55000` for a prerequisite-state refusal), and an evaluation +/// error such as division by zero keeps `22012`. +pub fn stamp_rows( + rows: &mut [ShapedRow], + columns: &[CpComputedColumn], + access: &dyn SequenceAccess, +) -> crate::Result<()> { + if columns.is_empty() { + return Ok(()); + } + for row in rows.iter_mut() { + let mut doc = row_document(row); + for column in columns { + let resolved = resolve_accessors(&column.expr, &doc, access)?; + let value = resolved.eval(&doc)?; + if let Some(map) = doc.as_object_mut() { + map.insert(column.alias.clone(), value.clone()); + } + row.insert(column.alias.clone(), value); + } + } + Ok(()) +} + +/// The flat row as the document an expression evaluates against. +fn row_document(row: &ShapedRow) -> Value { + Value::Object( + row.iter() + .map(|(key, value)| (key.clone(), value.clone())) + .collect::>(), + ) +} + +#[cfg(test)] +mod tests { + use std::cell::RefCell; + + use super::*; + use crate::bridge::expr_eval::{BinaryOp, SqlExpr}; + + /// Hands out 1, 2, 3, ... and remembers the last value per sequence. + struct FakeAccess { + next: RefCell, + last: RefCell>, + } + + impl FakeAccess { + fn new() -> Self { + Self { + next: RefCell::new(0), + last: RefCell::new(HashMap::new()), + } + } + } + + impl SequenceAccess for FakeAccess { + fn nextval(&self, name: &str) -> crate::Result { + let mut next = self.next.borrow_mut(); + *next += 1; + self.last.borrow_mut().insert(name.to_string(), *next); + Ok(*next) + } + fn currval(&self, name: &str) -> crate::Result { + self.last.borrow().get(name).copied().ok_or_else(|| { + crate::Error::ObjectNotInPrerequisiteState { + object: name.to_string(), + detail: format!("currval of sequence \"{name}\" is not yet defined"), + } + }) + } + fn setval(&self, name: &str, value: i64) -> crate::Result { + *self.next.borrow_mut() = value; + self.last.borrow_mut().insert(name.to_string(), value); + Ok(value) + } + } + + fn accessor(name: &str) -> SqlExpr { + SqlExpr::Function { + name: name.to_string(), + args: vec![SqlExpr::Literal(Value::String("s".to_string()))], + } + } + + fn column(alias: &str, expr: SqlExpr) -> CpComputedColumn { + CpComputedColumn { + alias: alias.to_string(), + expr, + } + } + + fn rows(n: usize) -> Vec { + (0..n) + .map(|i| { + let mut row = ShapedRow::new(); + row.insert("id".to_string(), Value::Integer(i as i64)); + row + }) + .collect() + } + + #[test] + fn each_row_gets_its_own_nextval_in_row_order() { + let access = FakeAccess::new(); + let mut rows = rows(3); + stamp_rows(&mut rows, &[column("n", accessor("nextval"))], &access).expect("stamp"); + let stamped: Vec<&Value> = rows.iter().map(|r| &r["n"]).collect(); + assert_eq!( + stamped, + [&Value::Integer(1), &Value::Integer(2), &Value::Integer(3)] + ); + } + + #[test] + fn currval_after_nextval_in_the_same_row_reads_that_rows_value() { + let access = FakeAccess::new(); + let mut rows = rows(2); + stamp_rows( + &mut rows, + &[ + column("n", accessor("nextval")), + column("c", accessor("currval")), + ], + &access, + ) + .expect("stamp"); + assert_eq!(rows[0]["n"], rows[0]["c"]); + assert_eq!(rows[1]["n"], rows[1]["c"]); + assert_eq!(rows[1]["n"], Value::Integer(2)); + } + + #[test] + fn a_later_column_can_reference_an_earlier_alias() { + let access = FakeAccess::new(); + let mut rows = rows(1); + stamp_rows( + &mut rows, + &[ + column("n", accessor("nextval")), + column( + "doubled", + SqlExpr::BinaryOp { + left: Box::new(SqlExpr::Column("n".to_string())), + op: BinaryOp::Mul, + right: Box::new(SqlExpr::Literal(Value::Integer(2))), + }, + ), + ], + &access, + ) + .expect("stamp"); + assert_eq!(rows[0]["doubled"], Value::Integer(2)); + } + + #[test] + fn an_expression_reads_the_rows_own_columns() { + let access = FakeAccess::new(); + let mut rows = rows(2); + stamp_rows( + &mut rows, + &[column( + "tagged", + SqlExpr::BinaryOp { + left: Box::new(accessor("nextval")), + op: BinaryOp::Add, + right: Box::new(SqlExpr::Column("id".to_string())), + }, + )], + &access, + ) + .expect("stamp"); + assert_eq!(rows[0]["tagged"], Value::Integer(1)); + assert_eq!(rows[1]["tagged"], Value::Integer(3)); + } + + #[test] + fn currval_before_nextval_fails_with_its_own_class() { + let access = FakeAccess::new(); + let mut rows = rows(1); + let err = stamp_rows(&mut rows, &[column("c", accessor("currval"))], &access) + .expect_err("must fail"); + assert!(matches!( + err, + crate::Error::ObjectNotInPrerequisiteState { .. } + )); + } + + #[test] + fn division_by_zero_keeps_its_class() { + let access = FakeAccess::new(); + let mut rows = rows(1); + let err = stamp_rows( + &mut rows, + &[column( + "bad", + SqlExpr::BinaryOp { + left: Box::new(accessor("nextval")), + op: BinaryOp::Div, + right: Box::new(SqlExpr::Literal(Value::Integer(0))), + }, + )], + &access, + ) + .expect_err("must fail"); + assert!(matches!(err, crate::Error::DivisionByZero)); + } + + #[test] + fn no_columns_leaves_rows_untouched() { + let access = FakeAccess::new(); + let mut rows = rows(2); + stamp_rows(&mut rows, &[], &access).expect("stamp"); + assert!(rows.iter().all(|r| r.len() == 1)); + } +} diff --git a/nodedb/src/control/server/response_shape/stamp/mod.rs b/nodedb/src/control/server/response_shape/stamp/mod.rs new file mode 100644 index 000000000..894b2f929 --- /dev/null +++ b/nodedb/src/control/server/response_shape/stamp/mod.rs @@ -0,0 +1,8 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Control-Plane evaluation of per-row computed SELECT-list columns. + +pub mod evaluate; +mod substitute; + +pub use evaluate::stamp_rows; diff --git a/nodedb/src/control/server/response_shape/stamp/substitute.rs b/nodedb/src/control/server/response_shape/stamp/substitute.rs new file mode 100644 index 000000000..5d2a0f75e --- /dev/null +++ b/nodedb/src/control/server/response_shape/stamp/substitute.rs @@ -0,0 +1,367 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Replaces every sequence accessor call in an expression with the value the +//! session's registry hands back for the current row. +//! +//! The walk is exhaustive over [`SqlExpr`] and resolves arguments before the +//! call that holds them, so an accessor nested inside another accessor's +//! argument runs first. Sibling calls run left to right, which is the order +//! `nextval` then `currval` in one SELECT list relies on. + +use nodedb_types::Value; + +use crate::bridge::expr_eval::SqlExpr; +use crate::control::sequence::SequenceAccess; + +/// Resolve every sequence accessor call in `expr` against `row`, returning +/// an expression with each call replaced by its integer result. +pub(super) fn resolve_accessors( + expr: &SqlExpr, + row: &Value, + access: &dyn SequenceAccess, +) -> crate::Result { + match expr { + SqlExpr::Column(_) + | SqlExpr::Literal(_) + | SqlExpr::OldColumn(_) + | SqlExpr::ExcludedColumn(_) => Ok(expr.clone()), + SqlExpr::BinaryOp { left, op, right } => Ok(SqlExpr::BinaryOp { + left: Box::new(resolve_accessors(left, row, access)?), + op: *op, + right: Box::new(resolve_accessors(right, row, access)?), + }), + SqlExpr::Negate(inner) => Ok(SqlExpr::Negate(Box::new(resolve_accessors( + inner, row, access, + )?))), + SqlExpr::Cast { expr, to_type } => Ok(SqlExpr::Cast { + expr: Box::new(resolve_accessors(expr, row, access)?), + to_type: to_type.clone(), + }), + SqlExpr::Case { + operand, + when_thens, + else_expr, + } => Ok(SqlExpr::Case { + operand: resolve_boxed(operand.as_deref(), row, access)?, + when_thens: when_thens + .iter() + .map(|(when, then)| { + Ok(( + resolve_accessors(when, row, access)?, + resolve_accessors(then, row, access)?, + )) + }) + .collect::>>()?, + else_expr: resolve_boxed(else_expr.as_deref(), row, access)?, + }), + SqlExpr::Coalesce(items) => Ok(SqlExpr::Coalesce(resolve_list(items, row, access)?)), + SqlExpr::NullIf(left, right) => Ok(SqlExpr::NullIf( + Box::new(resolve_accessors(left, row, access)?), + Box::new(resolve_accessors(right, row, access)?), + )), + SqlExpr::IsNull { expr, negated } => Ok(SqlExpr::IsNull { + expr: Box::new(resolve_accessors(expr, row, access)?), + negated: *negated, + }), + SqlExpr::Function { name, args } => { + let args = resolve_list(args, row, access)?; + let lowered = name.to_ascii_lowercase(); + match lowered.as_str() { + "nextval" => { + let sequence = sequence_name(&lowered, &args, row)?; + Ok(SqlExpr::Literal(Value::Integer(access.nextval(&sequence)?))) + } + "currval" => { + let sequence = sequence_name(&lowered, &args, row)?; + Ok(SqlExpr::Literal(Value::Integer(access.currval(&sequence)?))) + } + "setval" => { + let (sequence, value) = setval_args(&args, row)?; + Ok(SqlExpr::Literal(Value::Integer( + access.setval(&sequence, value)?, + ))) + } + _ => Ok(SqlExpr::Function { + name: name.clone(), + args, + }), + } + } + } +} + +fn resolve_boxed( + expr: Option<&SqlExpr>, + row: &Value, + access: &dyn SequenceAccess, +) -> crate::Result>> { + expr.map(|e| resolve_accessors(e, row, access).map(Box::new)) + .transpose() +} + +fn resolve_list( + items: &[SqlExpr], + row: &Value, + access: &dyn SequenceAccess, +) -> crate::Result> { + items + .iter() + .map(|item| resolve_accessors(item, row, access)) + .collect() +} + +/// The single sequence-name argument of `nextval` / `currval`, evaluated +/// against the row. +fn sequence_name(function: &str, args: &[SqlExpr], row: &Value) -> crate::Result { + let [arg] = args else { + return Err(crate::Error::PlanError { + detail: format!( + "{function}() takes exactly one argument, the sequence name; got {}", + args.len() + ), + }); + }; + match arg.eval(row)? { + Value::String(name) => Ok(name), + other => Err(crate::Error::PlanError { + detail: format!( + "{function}() argument must evaluate to a sequence name (text); got {other:?}" + ), + }), + } +} + +/// The `(name, value)` arguments of `setval`, both evaluated against the row. +fn setval_args(args: &[SqlExpr], row: &Value) -> crate::Result<(String, i64)> { + let [name_arg, value_arg] = args else { + return Err(crate::Error::PlanError { + detail: format!( + "setval() takes exactly two arguments, the sequence name and the value; got {}", + args.len() + ), + }); + }; + let name = match name_arg.eval(row)? { + Value::String(name) => name, + other => { + return Err(crate::Error::PlanError { + detail: format!( + "setval() first argument must evaluate to a sequence name (text); got {other:?}" + ), + }); + } + }; + let value = coerce_i64(&value_arg.eval(row)?).ok_or_else(|| crate::Error::PlanError { + detail: "setval() second argument must evaluate to a bigint".to_string(), + })?; + Ok((name, value)) +} + +/// An integer from an evaluated cell: an integer as is, a float with no +/// fraction inside `i64`, or text that parses as one. +fn coerce_i64(value: &Value) -> Option { + match value { + Value::Integer(i) => Some(*i), + Value::Float(f) if f.fract() == 0.0 && f.abs() < i64::MAX as f64 => Some(*f as i64), + Value::String(s) => s.trim().parse::().ok(), + _ => None, + } +} + +#[cfg(test)] +mod tests { + use std::cell::RefCell; + use std::collections::HashMap; + + use super::*; + use crate::bridge::expr_eval::BinaryOp; + + /// Records every call in order and answers from a counter. + struct FakeAccess { + calls: RefCell>, + next: RefCell, + last: RefCell>, + } + + impl FakeAccess { + fn new() -> Self { + Self { + calls: RefCell::new(Vec::new()), + next: RefCell::new(0), + last: RefCell::new(None), + } + } + } + + impl SequenceAccess for FakeAccess { + fn nextval(&self, name: &str) -> crate::Result { + self.calls.borrow_mut().push(format!("nextval:{name}")); + let mut next = self.next.borrow_mut(); + *next += 1; + *self.last.borrow_mut() = Some(*next); + Ok(*next) + } + fn currval(&self, name: &str) -> crate::Result { + self.calls.borrow_mut().push(format!("currval:{name}")); + self.last + .borrow() + .ok_or_else(|| crate::Error::ObjectNotInPrerequisiteState { + object: name.to_string(), + detail: "not yet called".into(), + }) + } + fn setval(&self, name: &str, value: i64) -> crate::Result { + self.calls + .borrow_mut() + .push(format!("setval:{name}={value}")); + *self.next.borrow_mut() = value; + Ok(value) + } + } + + fn call(name: &str, args: Vec) -> SqlExpr { + SqlExpr::Function { + name: name.to_string(), + args, + } + } + + fn text(s: &str) -> SqlExpr { + SqlExpr::Literal(Value::String(s.to_string())) + } + + fn row() -> Value { + Value::Object(HashMap::new()) + } + + #[test] + fn nextval_becomes_an_integer_literal() { + let access = FakeAccess::new(); + let out = + resolve_accessors(&call("NEXTVAL", vec![text("s")]), &row(), &access).expect("resolve"); + assert_eq!(out, SqlExpr::Literal(Value::Integer(1))); + assert_eq!(access.calls.borrow().as_slice(), ["nextval:s"]); + } + + #[test] + fn siblings_resolve_left_to_right_and_nested_inner_first() { + let access = FakeAccess::new(); + // nextval('s') + setval('s', nextval('s') + 10) + let expr = SqlExpr::BinaryOp { + left: Box::new(call("nextval", vec![text("s")])), + op: BinaryOp::Add, + right: Box::new(call( + "setval", + vec![ + text("s"), + SqlExpr::BinaryOp { + left: Box::new(call("nextval", vec![text("s")])), + op: BinaryOp::Add, + right: Box::new(SqlExpr::Literal(Value::Integer(10))), + }, + ], + )), + }; + let out = resolve_accessors(&expr, &row(), &access).expect("resolve"); + assert_eq!( + access.calls.borrow().as_slice(), + ["nextval:s", "nextval:s", "setval:s=12"] + ); + assert_eq!(out.eval(&row()).expect("eval"), Value::Integer(13)); + } + + #[test] + fn sequence_name_can_come_from_the_row() { + let access = FakeAccess::new(); + let mut fields = HashMap::new(); + fields.insert("seq".to_string(), Value::String("orders".to_string())); + let row = Value::Object(fields); + resolve_accessors( + &call("nextval", vec![SqlExpr::Column("seq".to_string())]), + &row, + &access, + ) + .expect("resolve"); + assert_eq!(access.calls.borrow().as_slice(), ["nextval:orders"]); + } + + #[test] + fn non_text_sequence_name_is_a_plan_error_naming_the_function() { + let access = FakeAccess::new(); + let err = resolve_accessors( + &call("currval", vec![SqlExpr::Literal(Value::Integer(3))]), + &row(), + &access, + ) + .expect_err("must fail"); + assert!( + matches!(err, crate::Error::PlanError { ref detail } if detail.contains("currval()")) + ); + assert!(access.calls.borrow().is_empty()); + } + + #[test] + fn wrong_arity_is_a_plan_error() { + let access = FakeAccess::new(); + assert!(matches!( + resolve_accessors(&call("nextval", vec![]), &row(), &access), + Err(crate::Error::PlanError { .. }) + )); + assert!(matches!( + resolve_accessors(&call("setval", vec![text("s")]), &row(), &access), + Err(crate::Error::PlanError { .. }) + )); + } + + #[test] + fn setval_value_coerces_from_text_and_integral_float() { + let access = FakeAccess::new(); + resolve_accessors( + &call("setval", vec![text("s"), text(" 7 ")]), + &row(), + &access, + ) + .expect("text"); + resolve_accessors( + &call( + "setval", + vec![text("s"), SqlExpr::Literal(Value::Float(9.0))], + ), + &row(), + &access, + ) + .expect("float"); + assert!(matches!( + resolve_accessors( + &call( + "setval", + vec![text("s"), SqlExpr::Literal(Value::Float(9.5))] + ), + &row(), + &access, + ), + Err(crate::Error::PlanError { .. }) + )); + assert_eq!( + access.calls.borrow().as_slice(), + ["setval:s=7", "setval:s=9"] + ); + } + + #[test] + fn other_functions_and_leaves_pass_through_with_resolved_arguments() { + let access = FakeAccess::new(); + let expr = SqlExpr::Coalesce(vec![ + SqlExpr::Column("x".to_string()), + call("abs", vec![call("nextval", vec![text("s")])]), + ]); + let out = resolve_accessors(&expr, &row(), &access).expect("resolve"); + assert_eq!( + out, + SqlExpr::Coalesce(vec![ + SqlExpr::Column("x".to_string()), + call("abs", vec![SqlExpr::Literal(Value::Integer(1))]), + ]) + ); + } +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs index 5859703b5..194e07970 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs @@ -440,6 +440,9 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a database_id, tenant_id, redaction: Some(redaction.ctx(&state.redaction)), + // A RETURNING list names stored columns only, never a + // Control-Plane computed column. + sequences: None, }) .map_err(|error| ddl_err("XX000", error.message().to_string()))?; // Folded rather than pushed: a statement is ONE result set, however From 759f31d238e820cb7f79c81ff47b655e0e9dc64f Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 18 Sep 2026 11:01:02 +0800 Subject: [PATCH 11/15] feat(sql): resolve RETURNING items as full expressions, not just columns Previously RETURNING only accepted a bare column list or star. The clause now resolves through the same path as a top-level SELECT list, via the new resolve_returning_items in nodedb-sql, so computed expressions and sequence accessors work in RETURNING and are evaluated per row by the Control Plane. Split the monolithic shared/returning.rs into a directory module (clause, strip, inject) grouped by concern, kept protocol-neutral across the pgwire planner, the neutral DDL UPSERT path, and the prepared-statement Describe path. --- docs/query-language.md | 11 +- .../src/physical_plan/document/types.rs | 8 +- nodedb-sql/src/lib.rs | 1 + nodedb-sql/src/planner/mod.rs | 2 + nodedb-sql/src/planner/returning.rs | 253 ++++++ .../control/planner/context/query/planning.rs | 108 ++- .../sql_plan_convert/output_schema/build.rs | 10 +- .../output_schema/returning.rs | 129 +-- .../server/pgwire/handler/prepared/parser.rs | 46 +- .../server/pgwire/handler/routing/planning.rs | 28 +- .../neutral/collection/dml/parse/dispatch.rs | 49 +- nodedb/src/control/server/shared/returning.rs | 735 ------------------ .../control/server/shared/returning/clause.rs | 298 +++++++ .../control/server/shared/returning/inject.rs | 302 +++++++ .../control/server/shared/returning/mod.rs | 24 + .../control/server/shared/returning/strip.rs | 199 +++++ .../tests/wire/cases/pgwire_returning_dml.rs | 55 +- 17 files changed, 1353 insertions(+), 905 deletions(-) create mode 100644 nodedb-sql/src/planner/returning.rs delete mode 100644 nodedb/src/control/server/shared/returning.rs create mode 100644 nodedb/src/control/server/shared/returning/clause.rs create mode 100644 nodedb/src/control/server/shared/returning/inject.rs create mode 100644 nodedb/src/control/server/shared/returning/mod.rs create mode 100644 nodedb/src/control/server/shared/returning/strip.rs diff --git a/docs/query-language.md b/docs/query-language.md index 85c2c44dc..1879df585 100644 --- a/docs/query-language.md +++ b/docs/query-language.md @@ -239,9 +239,12 @@ TRUNCATE users; ### RETURNING -Both `UPDATE` and `DELETE` support a `RETURNING` clause to read back affected rows in the same statement: +`INSERT`, `UPSERT`, `UPDATE`, `DELETE`, and `MERGE` accept a `RETURNING` clause that reads back the affected rows in the same statement: ```sql +-- INSERT RETURNING: returns the stored row +INSERT INTO users (id, name) VALUES ('u2', 'Bo') RETURNING id, name; + -- UPDATE RETURNING: returns the post-update image UPDATE users SET role = 'admin' WHERE id = 'u1' RETURNING id, role; UPDATE orders SET status = 'shipped' WHERE id = 'o1' RETURNING *; @@ -249,9 +252,13 @@ UPDATE orders SET status = 'shipped' WHERE id = 'o1' RETURNING *; -- DELETE RETURNING: returns the pre-delete image DELETE FROM users WHERE id = 'u1' RETURNING id, name; DELETE FROM orders WHERE status = 'cancelled' RETURNING *; + +-- Expressions: evaluated per returned row against the stored image +UPDATE orders SET qty = qty + 1 WHERE id = 'o1' RETURNING id, qty * price AS total; +INSERT INTO events (id, kind) VALUES ('e1', 'click') RETURNING id, nextval('event_seq') AS n; ``` -`RETURNING *` expands to all columns. Named columns are returned as bare values — arithmetic expressions in `RETURNING` are not supported. Works in both simple-query and extended-query (prepared statement) protocols. +`RETURNING *` expands to all columns. Every item is a scalar expression over the target collection: a bare column, a column under an alias (`col AS name`), arithmetic, a function call, or a sequence accessor (`nextval`, `currval`, `setval`). The Data Plane returns the base columns an expression reads, and the Control Plane evaluates the expression once per returned row. A sequence accessor advances once per row, in row order. The clause works in both the simple-query and extended-query (prepared statement) protocols, and `Describe` announces an expression under its alias. ## DDL diff --git a/nodedb-physical/src/physical_plan/document/types.rs b/nodedb-physical/src/physical_plan/document/types.rs index 9b8f7df7a..d32109d2c 100644 --- a/nodedb-physical/src/physical_plan/document/types.rs +++ b/nodedb-physical/src/physical_plan/document/types.rs @@ -237,10 +237,12 @@ pub enum ReturningColumns { Named(Vec), } -/// Parsed representation of a RETURNING clause carried through the bridge. +/// The Data-Plane projection of a RETURNING clause carried through the bridge. /// -/// Produced by the Control Plane's `strip_returning()` and injected into -/// `PointUpdate`, `BulkUpdate`, `PointDelete`, and `BulkDelete` variants +/// Derived on the Control Plane from the resolved clause: every stored column +/// the clause names or an expression in it reads, by bare name. The Control +/// Plane evaluates expressions and applies display names after the rows +/// return. Injected into the DML plan variants that carry a `returning` slot /// before crossing the SPSC bridge. #[derive( Debug, diff --git a/nodedb-sql/src/lib.rs b/nodedb-sql/src/lib.rs index 377e74832..8f8f7654a 100644 --- a/nodedb-sql/src/lib.rs +++ b/nodedb-sql/src/lib.rs @@ -41,6 +41,7 @@ pub use catalog::{SqlCatalog, SqlCatalogError}; pub use error::{Result, SqlError}; pub use params::ParamValue; pub use placeholder_types::{InferredParamType, infer_placeholder_types}; +pub use planner::returning::resolve_returning_items; pub use types::*; /// Parse a standalone SQL expression string into an `SqlExpr`. diff --git a/nodedb-sql/src/planner/mod.rs b/nodedb-sql/src/planner/mod.rs index 7d9588428..ae80ed820 100644 --- a/nodedb-sql/src/planner/mod.rs +++ b/nodedb-sql/src/planner/mod.rs @@ -30,8 +30,10 @@ pub mod join; pub mod lateral; pub mod merge; pub mod predicate_coerce; +pub mod returning; pub mod select; +pub use returning::resolve_returning_items; pub use select::qualified_name; pub mod sort; pub mod subquery; diff --git a/nodedb-sql/src/planner/returning.rs b/nodedb-sql/src/planner/returning.rs new file mode 100644 index 000000000..df6a5cbab --- /dev/null +++ b/nodedb-sql/src/planner/returning.rs @@ -0,0 +1,253 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Resolution of a DML `RETURNING` item list against the write's target. +//! +//! The clause is a projection over the target collection, so it converts +//! through the same path a top-level SELECT list does. A bare column and a +//! star pass through. Every other item resolves to +//! [`Projection::CpComputed`]: the Data Plane returns the base columns the +//! expression reads and the Control Plane evaluates it once per returned +//! row. A sequence accessor is allowed here exactly as in a top-level SELECT +//! list. + +use sqlparser::ast; + +use nodedb_types::DatabaseId; + +use crate::error::{Result, SqlError}; +use crate::parser::statement::parse_sql; +use crate::planner::select::helpers::convert_projection; +use crate::resolver::columns::{ResolvedTable, TableScope}; +use crate::types::{Projection, SqlCatalog, SqlExpr}; + +/// Resolve a RETURNING item list against the DML target collection. +/// +/// `items_sql` is the text after the `RETURNING` keyword. Every non-column +/// item resolves to `Projection::CpComputed`: the Data Plane returns the base +/// columns the expression reads and the Control Plane evaluates it per +/// returned row. +pub fn resolve_returning_items( + items_sql: &str, + target: &str, + catalog: &dyn SqlCatalog, +) -> Result> { + let items = parse_items(items_sql)?; + let info = catalog + .get_collection(DatabaseId::DEFAULT, target)? + .ok_or_else(|| SqlError::UnknownTable { + name: target.to_string(), + })?; + let scope = TableScope::single(ResolvedTable { + name: target.to_string(), + alias: None, + info, + })? + .as_statement_output() + .allowing_cp_functions(); + let projection = convert_projection(&items, &scope)?; + Ok(projection.into_iter().map(to_cp_computed).collect()) +} + +/// The SELECT items of `SELECT `. +/// +/// The wrapped statement must be exactly one SELECT with nothing but a +/// projection list: a FROM, WHERE, GROUP BY, ORDER BY, LIMIT, or WITH clause +/// after the keyword is refused, never silently ignored. +fn parse_items(items_sql: &str) -> Result> { + let refuse = || SqlError::Parse { + detail: format!("invalid RETURNING list: '{}'", items_sql.trim()), + }; + let mut statements = parse_sql(&format!("SELECT {items_sql}"))?; + if statements.len() != 1 { + return Err(refuse()); + } + let Some(ast::Statement::Query(query)) = statements.pop() else { + return Err(refuse()); + }; + if query.with.is_some() + || query.order_by.is_some() + || query.limit_clause.is_some() + || query.fetch.is_some() + { + return Err(refuse()); + } + let ast::SetExpr::Select(select) = *query.body else { + return Err(refuse()); + }; + let grouped = match &select.group_by { + ast::GroupByExpr::All(_) => true, + ast::GroupByExpr::Expressions(exprs, _) => !exprs.is_empty(), + }; + if !select.from.is_empty() || select.selection.is_some() || grouped || select.having.is_some() { + return Err(refuse()); + } + Ok(select.projection) +} + +/// A computed item becomes Control-Plane computed. A computed item that is a +/// bare column under an alias stays `Computed`: the value is the stored +/// column, looked up under the source name and displayed under the alias, so +/// nothing is evaluated. +fn to_cp_computed(projection: Projection) -> Projection { + match projection { + Projection::Computed { expr, alias } => match expr { + SqlExpr::Column { .. } => Projection::Computed { expr, alias }, + SqlExpr::Function { .. } + | SqlExpr::Literal(_) + | SqlExpr::BinaryOp { .. } + | SqlExpr::UnaryOp { .. } + | SqlExpr::Case { .. } + | SqlExpr::Cast { .. } + | SqlExpr::Subquery(_) + | SqlExpr::Wildcard + | SqlExpr::IsNull { .. } + | SqlExpr::InList { .. } + | SqlExpr::Between { .. } + | SqlExpr::Like { .. } + | SqlExpr::ArrayLiteral(_) => Projection::CpComputed { expr, alias }, + }, + Projection::Column(_) + | Projection::Star + | Projection::QualifiedStar(_) + | Projection::CpComputed { .. } => projection, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::catalog::SqlCatalogError; + use crate::types::{CollectionInfo, ColumnInfo, EngineType, SqlDataType}; + + /// One strict `items` collection: `id TEXT`, `score BIGINT`. + struct ItemsCatalog; + + impl SqlCatalog for ItemsCatalog { + fn get_collection( + &self, + _database_id: DatabaseId, + name: &str, + ) -> std::result::Result, SqlCatalogError> { + if name != "items" { + return Ok(None); + } + let col = |n: &str, t: SqlDataType| ColumnInfo { + name: n.to_string(), + data_type: t, + nullable: true, + is_primary_key: false, + default: None, + raw_type: None, + int_width: None, + float_width: None, + }; + Ok(Some(CollectionInfo { + name: "items".to_string(), + engine: EngineType::DocumentStrict, + columns: vec![ + col("id", SqlDataType::String), + col("score", SqlDataType::Int64), + ], + primary_key: None, + has_auto_tier: false, + indexes: Vec::new(), + bitemporal: false, + primary: nodedb_types::PrimaryEngine::Document, + vector_primary: None, + partition_strategy: nodedb_types::PartitionStrategy::CollectionHomed, + open_schema: CollectionInfo::open_schema_for(EngineType::DocumentStrict), + })) + } + } + + fn resolve(items: &str) -> Result> { + resolve_returning_items(items, "items", &ItemsCatalog) + } + + #[test] + fn a_bare_column_stays_a_column() { + let projection = resolve("id, score").unwrap(); + assert_eq!(projection.len(), 2); + assert!(matches!(&projection[0], Projection::Column(c) if c == "id")); + assert!(matches!(&projection[1], Projection::Column(c) if c == "score")); + } + + #[test] + fn an_aliased_column_stays_computed_over_the_column() { + let projection = resolve("id AS a").unwrap(); + assert_eq!(projection.len(), 1); + match &projection[0] { + Projection::Computed { expr, alias } => { + assert_eq!(alias, "a"); + assert!(matches!(expr, SqlExpr::Column { name, .. } if name == "id")); + } + other => panic!("expected Computed, got {other:?}"), + } + } + + #[test] + fn an_arithmetic_item_is_cp_computed() { + let projection = resolve("score * 2 AS d").unwrap(); + assert_eq!(projection.len(), 1); + match &projection[0] { + Projection::CpComputed { expr, alias } => { + assert_eq!(alias, "d"); + assert!(matches!(expr, SqlExpr::BinaryOp { .. })); + } + other => panic!("expected CpComputed, got {other:?}"), + } + } + + #[test] + fn an_unaliased_expression_is_named_by_its_text() { + let projection = resolve("score + 1").unwrap(); + match &projection[0] { + Projection::CpComputed { alias, .. } => assert_eq!(alias, "score + 1"), + other => panic!("expected CpComputed, got {other:?}"), + } + } + + #[test] + fn a_sequence_accessor_is_cp_computed() { + let projection = resolve("id, nextval('s') AS n").unwrap(); + assert_eq!(projection.len(), 2); + match &projection[1] { + Projection::CpComputed { expr, alias } => { + assert_eq!(alias, "n"); + assert!(matches!(expr, SqlExpr::Function { name, .. } if name == "nextval")); + } + other => panic!("expected CpComputed, got {other:?}"), + } + } + + #[test] + fn a_star_passes_through() { + let projection = resolve("*").unwrap(); + assert_eq!(projection.len(), 1); + assert!(matches!(projection[0], Projection::Star)); + } + + #[test] + fn an_unknown_column_is_a_resolve_error() { + let err = resolve("ghost").unwrap_err(); + assert!( + matches!(err, SqlError::UnknownColumn { ref column, .. } if column == "ghost"), + "expected UnknownColumn, got {err:?}" + ); + } + + #[test] + fn an_unknown_target_is_an_unknown_table() { + let err = resolve_returning_items("id", "ghost", &ItemsCatalog).unwrap_err(); + assert!(matches!(err, SqlError::UnknownTable { ref name } if name == "ghost")); + } + + #[test] + fn a_trailing_clause_is_refused() { + assert!(resolve("id FROM other").is_err()); + assert!(resolve("id WHERE score > 1").is_err()); + assert!(resolve("id ORDER BY id").is_err()); + assert!(resolve("id; SELECT 1").is_err()); + assert!(resolve("").is_err()); + } +} diff --git a/nodedb/src/control/planner/context/query/planning.rs b/nodedb/src/control/planner/context/query/planning.rs index 3a600b923..0078b683a 100644 --- a/nodedb/src/control/planner/context/query/planning.rs +++ b/nodedb/src/control/planner/context/query/planning.rs @@ -14,7 +14,36 @@ use crate::control::planner::context::security::PlanSecurityContext; use crate::control::planner::plan_error_map::map_plan_error; use crate::control::planner::sql_plan_convert::PlanningPurpose; use crate::control::server::response_shape::schema::OutputSchema; -use nodedb_physical::physical_plan::ReturningSpec; +use crate::control::server::shared::returning::{ + ReturningClause, attach_returning_spec, resolve_returning_for_plans, +}; + +/// Resolve a DML `RETURNING` clause against `plans` and build the announced +/// output schema from it in one step. +/// +/// Both the fresh-adapter path ([`QueryContext::plan_with_nodedb_sql_for_purpose`]) +/// and the parameterized path ([`QueryContext::plan_sql_with_params_and_rls_and_versions`]) +/// resolve the clause before conversion and need the same output schema built +/// from it, so the pairing lives here once. +fn resolve_returning_and_output_schema( + plans: &[nodedb_sql::types::SqlPlan], + returning_items: Option<&str>, + catalog: &C, + database_id: crate::types::DatabaseId, + tenant_id: crate::types::TenantId, +) -> crate::Result<(OutputSchema, Option)> { + let returning = resolve_returning_for_plans(plans, returning_items, catalog, tenant_id)?; + let output_schema = + crate::control::planner::sql_plan_convert::output_schema::build_output_schema( + plans, + catalog, + database_id, + returning + .as_ref() + .map(|clause| clause.projection.as_slice()), + ); + Ok((output_schema, returning)) +} /// Bundled arguments for [`QueryContext::plan_sql_with_rls`]. pub struct PlanSqlWithRlsParams<'a> { @@ -44,13 +73,17 @@ impl QueryContext { /// used as the plan-cache key AND as the input to /// `SharedState::acquire_plan_lease_scope` so cache hits /// and fresh plans share the same lease-acquisition path. + /// + /// `returning_items` is the raw text after a DML `RETURNING` keyword. It + /// resolves here against the planned target: the announced output schema + /// carries the projection, and every task carries the Data-Plane spec. fn plan_with_nodedb_sql_for_purpose( &self, sql: &str, tenant_id: crate::types::TenantId, database_id: crate::types::DatabaseId, purpose: PlanningPurpose, - returning: Option<&ReturningSpec>, + returning_items: Option<&str>, ) -> crate::Result<( Vec, OutputSchema, @@ -142,16 +175,22 @@ impl QueryContext { database_id, tenant_id, }; - let output_schema = - crate::control::planner::sql_plan_convert::output_schema::build_output_schema( - &plans, - catalog.as_ref(), - database_id, - returning, - ); + let (output_schema, returning) = resolve_returning_and_output_schema( + &plans, + returning_items, + catalog.as_ref(), + database_id, + tenant_id, + )?; let cache_eligibility = crate::control::planner::sql_plan_convert::batch_cache_eligibility(&plans); - let tasks = crate::control::planner::sql_plan_convert::convert(&plans, tenant_id, &ctx)?; + let mut tasks = + crate::control::planner::sql_plan_convert::convert(&plans, tenant_id, &ctx)?; + // Attached before the tasks leave: RLS injection, caching, expansion, + // and dispatch all read the plan after this point. + if let Some(clause) = &returning { + attach_returning_spec(&mut tasks, &clause.spec)?; + } Ok((tasks, output_schema, version_set, cache_eligibility)) } @@ -177,21 +216,21 @@ impl QueryContext { /// Plan SQL with RLS injection, announcing a DML `RETURNING` clause. /// - /// `returning` is the spec `strip_returning` parsed off the statement, so - /// the write announces the columns it projects. `None` for a statement - /// that carries no clause. + /// `returning_items` is the item text `strip_returning` split off the + /// statement, so the write announces the columns it projects. `None` for + /// a statement that carries no clause. pub async fn plan_sql_with_rls_returning( &self, sql: &str, tenant_id: crate::types::TenantId, database_id: crate::types::DatabaseId, sec: &PlanSecurityContext<'_>, - returning: Option<&ReturningSpec>, + returning_items: Option<&str>, ) -> crate::Result<( Vec, OutputSchema, )> { - self.plan_sql_with_rls_and_versions(sql, tenant_id, database_id, sec, returning) + self.plan_sql_with_rls_and_versions(sql, tenant_id, database_id, sec, returning_items) .await .map(|(tasks, schema, _, _)| (tasks, schema)) } @@ -208,7 +247,7 @@ impl QueryContext { tenant_id: crate::types::TenantId, database_id: crate::types::DatabaseId, sec: &PlanSecurityContext<'_>, - returning: Option<&ReturningSpec>, + returning_items: Option<&str>, ) -> crate::Result<( Vec, OutputSchema, @@ -220,7 +259,7 @@ impl QueryContext { tenant_id, database_id, sec, - returning, + returning_items, PlanningPurpose::Execute, ) .await @@ -260,7 +299,7 @@ impl QueryContext { tenant_id: crate::types::TenantId, database_id: crate::types::DatabaseId, sec: &PlanSecurityContext<'_>, - returning: Option<&ReturningSpec>, + returning_items: Option<&str>, purpose: PlanningPurpose, ) -> crate::Result<( Vec, @@ -268,8 +307,14 @@ impl QueryContext { crate::control::planner::descriptor_set::DescriptorVersionSet, nodedb_sql::types::PlanCacheEligibility, )> { - let (mut tasks, output_schema, mut version_set, cache_eligibility) = - self.plan_with_nodedb_sql_for_purpose(sql, tenant_id, database_id, purpose, returning)?; + let (mut tasks, output_schema, mut version_set, cache_eligibility) = self + .plan_with_nodedb_sql_for_purpose( + sql, + tenant_id, + database_id, + purpose, + returning_items, + )?; // Versions read BEFORE injection, never after: injection reads live // policy/grant state under its own lock, and a mutation racing in @@ -319,7 +364,7 @@ impl QueryContext { tenant_id: crate::types::TenantId, database_id: crate::types::DatabaseId, sec: &PlanSecurityContext<'_>, - returning: Option<&ReturningSpec>, + returning_items: Option<&str>, ) -> crate::Result<( Vec, OutputSchema, @@ -330,7 +375,7 @@ impl QueryContext { tenant_id, database_id, sec, - returning, + returning_items, ) .await .map(|(tasks, schema, _)| (tasks, schema)) @@ -346,7 +391,7 @@ impl QueryContext { tenant_id: crate::types::TenantId, database_id: crate::types::DatabaseId, sec: &PlanSecurityContext<'_>, - returning: Option<&ReturningSpec>, + returning_items: Option<&str>, ) -> crate::Result<( Vec, OutputSchema, @@ -423,15 +468,18 @@ impl QueryContext { database_id, tenant_id, }; - let output_schema = - crate::control::planner::sql_plan_convert::output_schema::build_output_schema( - &plans, - catalog.as_ref(), - database_id, - returning, - ); + let (output_schema, returning) = resolve_returning_and_output_schema( + &plans, + returning_items, + catalog.as_ref(), + database_id, + tenant_id, + )?; let mut tasks = crate::control::planner::sql_plan_convert::convert(&plans, tenant_id, &ctx)?; + if let Some(clause) = &returning { + attach_returning_spec(&mut tasks, &clause.spec)?; + } // Versions read BEFORE injection — see the comment on the sibling // planning path in this file for why a post-injection read is unsafe. diff --git a/nodedb/src/control/planner/sql_plan_convert/output_schema/build.rs b/nodedb/src/control/planner/sql_plan_convert/output_schema/build.rs index 80be200fb..56256573c 100644 --- a/nodedb/src/control/planner/sql_plan_convert/output_schema/build.rs +++ b/nodedb/src/control/planner/sql_plan_convert/output_schema/build.rs @@ -10,11 +10,10 @@ use std::collections::HashMap; -use nodedb_physical::physical_plan::ReturningSpec; use nodedb_query::agg_key::canonical_agg_key; use nodedb_sql::catalog::SqlCatalog; use nodedb_sql::types::SqlPlan; -use nodedb_sql::types::query::AggOutputSlot; +use nodedb_sql::types::query::{AggOutputSlot, Projection}; use crate::control::planner::sql_plan_convert::aggregate::agg_expr_to_pair; use crate::control::planner::sql_plan_convert::lateral::collection_name_from_plan; @@ -32,13 +31,14 @@ use super::returning::build_returning_schema; /// A read plan announces the columns its projection names. A write plan /// announces the columns `returning` projects: the clause is stripped from the /// statement text before planning, so the plan itself carries no column list -/// and the caller supplies the parsed spec. `None` means the statement carries -/// no `RETURNING` clause, and a write then announces nothing. +/// and the caller supplies the projection resolved against the target. `None` +/// means the statement carries no `RETURNING` clause, and a write then +/// announces nothing. pub fn build_output_schema( plans: &[SqlPlan], catalog: &C, database_id: nodedb_types::DatabaseId, - returning: Option<&ReturningSpec>, + returning: Option<&[Projection]>, ) -> OutputSchema { let Some(plan) = plans.first() else { return OutputSchema { diff --git a/nodedb/src/control/planner/sql_plan_convert/output_schema/returning.rs b/nodedb/src/control/planner/sql_plan_convert/output_schema/returning.rs index 6f95a8fdb..39acaf1f6 100644 --- a/nodedb/src/control/planner/sql_plan_convert/output_schema/returning.rs +++ b/nodedb/src/control/planner/sql_plan_convert/output_schema/returning.rs @@ -3,22 +3,23 @@ //! Derives the announced [`OutputSchema`] of a DML `RETURNING` clause. //! //! A `RETURNING` clause is a projection over the target collection, so it is -//! typed by the same rule a `SELECT` projection is: the clause's column list is -//! mapped to [`Projection`] entries and handed to +//! typed by the same rule a `SELECT` projection is: the resolved +//! [`Projection`] list is handed to //! [`schema_from_projection`](super::columns::schema_from_projection), against //! the same catalog column types and declared order. There is one derivation, //! so a write and a read of the same column can never announce different types. //! +//! A `CpComputed` entry is announced under its alias and recorded in +//! `cp_computed`, so the response shaper evaluates it per returned row. +//! //! `RETURNING *` sets `is_star`, exactly as `SELECT *` does. The concrete //! column list of a star is only knowable from the returned rows — a //! schemaless row carries fields no catalog column declares — so the shaper //! keeps the row-derived list and renders those cells as text, the same answer //! `SELECT *` gives for the same row. -use nodedb_physical::physical_plan::{ReturningColumns, ReturningSpec}; use nodedb_sql::catalog::SqlCatalog; use nodedb_sql::types::query::Projection; -use nodedb_sql::types_expr::SqlExpr; use crate::control::server::response_shape::schema::OutputSchema; @@ -29,53 +30,24 @@ use super::columns::{column_types_for, ordered_columns_for, schema_from_projecti /// `None` — the statement carries no `RETURNING` clause — announces nothing, /// which is what a write with no result set must say. pub fn build_returning_schema( - returning: Option<&ReturningSpec>, + returning: Option<&[Projection]>, collection: &str, catalog: &C, database_id: nodedb_types::DatabaseId, ) -> OutputSchema { - let Some(spec) = returning else { + let Some(projection) = returning else { return OutputSchema::default(); }; - let projection = returning_projection(spec); let types = column_types_for(catalog, database_id, collection); let ordered_cols = ordered_columns_for(catalog, database_id, collection); - schema_from_projection(&projection, &types, &ordered_cols) -} - -/// Maps a `RETURNING` column list to the projection entries the shared -/// derivation reads. -/// -/// The clause's grammar admits a bare column name and an optional alias, and -/// nothing else: `parse_returning_columns` rejects every expression form with a -/// typed error before a spec exists. So an aliased item maps to a -/// `Projection::Computed` wrapping the column reference — the form that keeps -/// the alias as the display name while the value is still looked up under the -/// source column — and a bare item maps to `Projection::Column`. -fn returning_projection(spec: &ReturningSpec) -> Vec { - match &spec.columns { - ReturningColumns::Star => vec![Projection::Star], - ReturningColumns::Named(items) => items - .iter() - .map(|item| match &item.alias { - Some(alias) => Projection::Computed { - expr: SqlExpr::Column { - table: None, - name: item.name.clone(), - }, - alias: alias.clone(), - }, - None => Projection::Column(item.name.clone()), - }) - .collect(), - } + schema_from_projection(projection, &types, &ordered_cols) } #[cfg(test)] mod tests { use super::*; use crate::control::server::response_shape::types::DdlColType; - use nodedb_physical::physical_plan::ReturningItem; + use nodedb_sql::types_expr::{BinaryOp, SqlExpr, SqlValue}; /// Catalog exposing one `points` collection: a `TIMESTAMP` time key, a /// `TEXT` tag, and a `FLOAT` measurement — the shape a timeseries @@ -128,31 +100,28 @@ mod tests { } } - fn named(items: &[(&str, Option<&str>)]) -> ReturningSpec { - ReturningSpec { - columns: ReturningColumns::Named( - items - .iter() - .map(|(name, alias)| ReturningItem { - name: (*name).to_string(), - alias: alias.map(str::to_string), - }) - .collect(), - ), + fn column(name: &str) -> SqlExpr { + SqlExpr::Column { + table: None, + name: name.to_string(), } } - fn schema(spec: Option<&ReturningSpec>) -> OutputSchema { + fn schema(projection: Option<&[Projection]>) -> OutputSchema { let database_id = nodedb_types::DatabaseId::DEFAULT; - build_returning_schema(spec, "points", &PointsCatalog, database_id) + build_returning_schema(projection, "points", &PointsCatalog, database_id) } /// Named columns carry the declared catalog type, in clause order — the /// same types `SELECT ts, host, v` announces for the same row. #[test] fn named_columns_carry_their_declared_types() { - let spec = named(&[("ts", None), ("host", None), ("v", None)]); - let out = schema(Some(&spec)); + let projection = vec![ + Projection::Column("ts".into()), + Projection::Column("host".into()), + Projection::Column("v".into()), + ]; + let out = schema(Some(&projection)); assert!(!out.is_star); let got: Vec<(&str, DdlColType)> = out .columns @@ -167,26 +136,68 @@ mod tests { ("v", DdlColType::Float8), ] ); + assert!(out.cp_computed.is_empty()); } /// An alias names the output column while the value is still looked up /// under the source column, and it keeps the source column's type. #[test] fn an_alias_renames_the_column_and_keeps_its_type() { - let spec = named(&[("v", Some("reading"))]); - let out = schema(Some(&spec)); + let projection = vec![Projection::Computed { + expr: column("v"), + alias: "reading".into(), + }]; + let out = schema(Some(&projection)); assert_eq!(out.columns.len(), 1); assert_eq!(out.columns[0].display_name, "reading"); assert_eq!(out.columns[0].lookup_key, "v"); assert_eq!(out.columns[0].ty, DdlColType::Float8); } + /// A Control-Plane computed entry is announced under its alias, looked up + /// under that alias, and recorded for the shaper to evaluate. + #[test] + fn a_computed_entry_is_announced_and_recorded() { + let projection = vec![ + Projection::Column("host".into()), + Projection::CpComputed { + expr: SqlExpr::BinaryOp { + left: Box::new(column("v")), + op: BinaryOp::Mul, + right: Box::new(SqlExpr::Literal(SqlValue::Int(2))), + }, + alias: "d".into(), + }, + ]; + let out = schema(Some(&projection)); + assert_eq!(out.columns.len(), 2); + assert_eq!(out.columns[1].display_name, "d"); + assert_eq!(out.columns[1].lookup_key, "d"); + assert_eq!(out.cp_computed.len(), 1); + assert_eq!(out.cp_computed[0].alias, "d"); + } + + /// A bare sequence accessor announces `bigint`. + #[test] + fn a_bare_accessor_is_a_bigint() { + let projection = vec![Projection::CpComputed { + expr: SqlExpr::Function { + name: "nextval".into(), + args: vec![SqlExpr::Literal(SqlValue::String("s".into()))], + distinct: false, + }, + alias: "n".into(), + }]; + let out = schema(Some(&projection)); + assert_eq!(out.columns[0].ty, DdlColType::Int8); + } + /// A column the catalog does not declare falls back to `Text`, the safe /// default for a schemaless field. #[test] fn an_undeclared_column_falls_back_to_text() { - let spec = named(&[("undeclared", None)]); - let out = schema(Some(&spec)); + let projection = vec![Projection::Column("undeclared".into())]; + let out = schema(Some(&projection)); assert_eq!(out.columns[0].ty, DdlColType::Text); } @@ -194,10 +205,8 @@ mod tests { /// column list — the same answer `SELECT *` gives. #[test] fn a_star_is_marked_as_one() { - let spec = ReturningSpec { - columns: ReturningColumns::Star, - }; - let out = schema(Some(&spec)); + let projection = vec![Projection::Star]; + let out = schema(Some(&projection)); assert!(out.is_star); let names: Vec<&str> = out .columns diff --git a/nodedb/src/control/server/pgwire/handler/prepared/parser.rs b/nodedb/src/control/server/pgwire/handler/prepared/parser.rs index 1ae47ff18..11c044fba 100644 --- a/nodedb/src/control/server/pgwire/handler/prepared/parser.rs +++ b/nodedb/src/control/server/pgwire/handler/prepared/parser.rs @@ -18,6 +18,7 @@ use crate::config::auth::AuthMode; use crate::control::security::audit::ArcAuditEmitter; use crate::control::server::response_shape::types::DdlColType; use crate::control::server::shared::authorization::{authorize_database, authorize_task_set}; +use crate::control::server::shared::returning; use crate::control::server::shared::session::{SessionId, SessionStore}; use crate::control::state::SharedState; @@ -193,11 +194,10 @@ impl NodeDbQueryParser { database_id: crate::types::DatabaseId, emitter: &ArcAuditEmitter, ) -> PgWireResult { - let (sql_without_returning, _) = - match crate::control::server::shared::returning::strip_returning(sql) { - Ok(parts) => parts, - Err(_) => return Ok(false), - }; + let (sql_without_returning, _) = match returning::strip_returning(sql) { + Ok(parts) => parts, + Err(_) => return Ok(false), + }; let sql_for_planning = substitute_placeholders_with_null(&sql_without_returning); let query_ctx = crate::control::planner::context::QueryContext::for_state_with_lease(&self.state); @@ -271,6 +271,7 @@ impl NodeDbQueryParser { client_types: &[Option], catalog: &crate::control::planner::catalog_adapter::OriginCatalog, database_id: crate::types::DatabaseId, + tenant_id: crate::types::TenantId, ) -> (Vec>, Vec) { // Placeholder *counting* runs unconditionally so an unplannable SQL // string (e.g. `WHERE id = $1` where the planner needs bound params @@ -280,13 +281,12 @@ impl NodeDbQueryParser { // pass below and survives that pass failing. let param_types = Self::param_types_with_inference(sql, client_types, catalog); - // Strip RETURNING from DML before passing to DataFusion. Retain the - // parsed spec so we can build result fields for Describe. - let (sql_stripped, returning_spec) = - match crate::control::server::shared::returning::strip_returning(sql) { - Ok(pair) => pair, - Err(_) => return (param_types, Vec::new()), - }; + // Strip RETURNING from DML before planning. The item text is kept so + // the result fields for Describe are built from the resolved clause. + let (sql_stripped, returning_items) = match returning::strip_returning(sql) { + Ok(pair) => pair, + Err(_) => return (param_types, Vec::new()), + }; // Parse and plan to get collection info for result schema. // @@ -302,10 +302,24 @@ impl NodeDbQueryParser { Err(_) => return (param_types, Vec::new()), }; + // The RETURNING clause resolves against the planned target exactly as + // it does at execute time, so Describe announces the same columns — + // an expression under its alias, a column under its name. A clause + // that fails to resolve yields no fields; Execute reports the error. + let returning = match returning::resolve_returning_for_plans( + &plans, + returning_items.as_deref(), + catalog, + tenant_id, + ) { + Ok(clause) => clause, + Err(_) => return (param_types, Vec::new()), + }; + // Infer result fields from the planner's authoritative output // schema — the same derivation used to shape response rows, so // Describe's RowDescription always matches what Execute returns. - // A write plan announces what its `RETURNING` spec projects, through + // A write plan announces what its `RETURNING` clause projects, through // that same derivation, so Describe and the simple-query path can // never disagree on a column's type. Empty `plans` (already handled // above) or a plan variant with no resolvable projection yields an @@ -316,7 +330,9 @@ impl NodeDbQueryParser { &plans, catalog, database_id, - returning_spec.as_ref(), + returning + .as_ref() + .map(|clause| clause.projection.as_slice()), ); let result_fields: Vec = output_schema .columns @@ -384,7 +400,7 @@ impl QueryParser for NodeDbQueryParser { // whole could be planned. let catalog = self.build_catalog(identity.tenant_id.as_u64(), database_id); let (param_types, result_fields) = if can_infer_schema { - self.try_infer_types(sql, types, &catalog, database_id) + self.try_infer_types(sql, types, &catalog, database_id, identity.tenant_id) } else { ( Self::param_types_with_inference(sql, types, &catalog), diff --git a/nodedb/src/control/server/pgwire/handler/routing/planning.rs b/nodedb/src/control/server/pgwire/handler/routing/planning.rs index 194f2d032..9e3c16cd4 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/planning.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/planning.rs @@ -62,7 +62,8 @@ impl NodeDbPgHandler { } /// Plan a SQL statement to physical tasks: session auth, RETURNING strip, - /// CHECK constraints, plan cache, RETURNING injection. Returns the task list + /// CHECK constraints, plan cache. The planner resolves and attaches the + /// RETURNING clause, so a cached task set already carries it. Returns the task list /// and descriptor versions; errors stay typed to distinguish a descriptor-drain race from terminal failure. pub(in crate::control::server::pgwire::handler) async fn plan_statement_to_tasks( &self, @@ -159,8 +160,10 @@ impl NodeDbPgHandler { let (statement_sql, scope) = crate::control::server::session_auth::apply_per_query_on_deny(sql, scope); - // Strip RETURNING clause before DataFusion planning. - let (clean_sql, returning_spec) = + // Strip the RETURNING clause before planning. The item text resolves + // inside the planner against the planned target, which announces the + // projection and attaches the Data-Plane spec to every task. + let (clean_sql, returning_items) = returning::strip_returning(&statement_sql).map_err(StatementSetupError::from)?; // Forwards per-session planning GUCs into the shared query context, protocol-neutral @@ -246,7 +249,7 @@ impl NodeDbPgHandler { tenant_id, database_id, &sec, - returning_spec.as_ref(), + returning_items.as_deref(), ) .await .map_err(StatementSetupError::from)?; @@ -279,7 +282,7 @@ impl NodeDbPgHandler { tenant_id, database_id, &sec, - returning_spec.as_ref(), + returning_items.as_deref(), ) .await .map_err(StatementSetupError::from)? @@ -299,21 +302,6 @@ impl NodeDbPgHandler { (planned, output_schema, versions) }; - // Inject RETURNING spec into DML plans. An insert shape with no `returning` - // slot is refused rather than silently dropped. - let tasks = if let Some(ref spec) = returning_spec { - let mut injected = Vec::with_capacity(tasks.len()); - for mut task in tasks { - returning::refuse_unprojectable_insert_returning(&task.plan) - .map_err(StatementSetupError::from)?; - returning::inject_returning_spec(&mut task.plan, spec.clone()); - injected.push(task); - } - injected - } else { - tasks - }; - // Preauthorize before expansion allocates surrogates; descriptor admission // waits for the expanded task set's final authorization in the execute path. let _preauthorized_tasks = self diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs index 194e07970..0e87463ab 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs @@ -6,6 +6,7 @@ use crate::control::planner::context::PlanSecurityContext; use crate::control::security::audit::ArcAuditEmitter; use crate::control::security::identity::{AuthenticatedIdentity, Permission}; use crate::control::security::request_scope::RequestAuthScope; +use crate::control::sequence::SessionSequenceAccess; use crate::control::server::pgwire::types::error_to_sqlstate; use crate::control::server::response_shape::compose::{ShapeOutcome, shape_response_materialized}; use crate::control::server::response_shape::redaction::QueryRedaction; @@ -117,9 +118,10 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a sql: &str, txn_ctx: &DmlTxnCtx<'_>, ) -> Result, DdlError> { - // The clause is stripped from the rebuilt statement before planning and - // re-attached to each plan below — the planner itself does not parse it. - let (sql, returning_spec) = returning::strip_returning(sql).map_err(|error| { + // The clause is stripped from the rebuilt statement before planning. The + // planner resolves the item text against the planned target and attaches + // the Data-Plane spec to every task before they come back here. + let (sql, returning_items) = returning::strip_returning(sql).map_err(|error| { let (_, sqlstate, message) = error_to_sqlstate(&error); ddl_err(sqlstate, message) })?; @@ -131,9 +133,10 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a // it: read filters would not be injected, and the write gates would decide // nothing, on a transport a client can reach directly. // - // Injection happens HERE, before the task set is consumed: implicit-edge - // extraction, authorization, staging, and dispatch all read `tasks` after - // this point, and injecting later would hand them un-injected copies. + // Injection happens inside planning, before the task set is consumed: + // implicit-edge extraction, authorization, staging, and dispatch all read + // `tasks` after this point, and injecting later would hand them + // un-injected copies. let (mut tasks, output_schema, versions) = { let scope = RequestAuthScope::for_database(identity, state.auth_stores(), database_id); let permission_cache = state.permission_cache.read().await; @@ -153,7 +156,7 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a tenant_id, database_id, &sec, - returning_spec.as_ref(), + returning_items.as_deref(), ) .await .map_err(|error| { @@ -162,18 +165,7 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a })?; (tasks, output_schema, versions) }; - - // Attach the projection to every planned write, refusing any insert shape - // that has nowhere to carry it rather than dropping the clause in silence. - if let Some(ref spec) = returning_spec { - for task in &mut tasks { - returning::refuse_unprojectable_insert_returning(&task.plan).map_err(|error| { - let (_, sqlstate, message) = error_to_sqlstate(&error); - ddl_err(sqlstate, message) - })?; - returning::inject_returning_spec(&mut task.plan, spec.clone()); - } - } + let has_returning = returning_items.is_some(); // Extraction marks catalog state and allocates surrogates. Reject an // unauthorized original DML task set before either side effect can occur. @@ -287,7 +279,7 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a // A cross-shard Calvin dispatch returns no per-task payload here, so // there is no stored row to project. Refused rather than answered with // an empty row set, which would read as "the write matched nothing". - if returning_spec.is_some() { + if has_returning { return Err(ddl_err( "0A000", "RETURNING is not supported on a write that spans multiple shards", @@ -358,7 +350,7 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a // here, so the clause cannot be answered on this path. Refused // through the shared rule so this transport's message is the // one the pgwire and native loops give for the same limitation. - if returning_spec.is_some() { + if has_returning { let (_, sqlstate, message) = error_to_sqlstate(&returning::in_transaction_returning_unsupported()); return Err(ddl_err(sqlstate, message)); @@ -426,9 +418,18 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a // Shape the STORED rows the write returned, redacted for the caller — // the same choke point the pgwire dispatch loop uses, so a redaction // policy masks identically on both transports. - if returning_spec.is_some() { + if has_returning { let scope = RequestAuthScope::for_database(identity, state.auth_stores(), database_id); let redaction = QueryRedaction::for_plan(tenant_id, scope.auth(), &task.plan); + // A RETURNING expression is evaluated here, per returned row, and + // a sequence accessor in it resolves against this session's + // `currval` map exactly as the pgwire loop's does. + let sequences = SessionSequenceAccess::for_session( + state, + txn_ctx.sessions.sequence_values(txn_ctx.session_id), + database_id, + tenant_id, + ); let outcome = shape_response_materialized(MaterializedShapeRequest { payload: response.payload.as_bytes(), plan: &task.plan, @@ -440,9 +441,7 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a database_id, tenant_id, redaction: Some(redaction.ctx(&state.redaction)), - // A RETURNING list names stored columns only, never a - // Control-Plane computed column. - sequences: None, + sequences: Some(&sequences), }) .map_err(|error| ddl_err("XX000", error.message().to_string()))?; // Folded rather than pushed: a statement is ONE result set, however diff --git a/nodedb/src/control/server/shared/returning.rs b/nodedb/src/control/server/shared/returning.rs deleted file mode 100644 index a8b2cfd6f..000000000 --- a/nodedb/src/control/server/shared/returning.rs +++ /dev/null @@ -1,735 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! RETURNING clause handling for DML statements: strip it from the text, -//! decide whether the resulting plan can carry it, and attach it. -//! -//! The planner does not parse RETURNING on DML, so the clause is removed from -//! the raw SQL before planning and its projected column list is parsed here. -//! The spec is then injected into the plan variant that will produce the rows. -//! -//! Protocol-neutral: both the pgwire planner and the neutral DDL router's -//! `UPSERT` path go through this module, so a statement's clause is stripped, -//! judged, and attached identically on either transport. - -// Re-export bridge types so callers only import from this module. -pub use nodedb_physical::physical_plan::{ReturningColumns, ReturningItem, ReturningSpec}; - -use crate::Error; -use crate::bridge::envelope::PhysicalPlan; -use nodedb_physical::physical_plan::{ - ColumnarOp, CrdtOp, DocumentOp, KvOp, QueryOp, TimeseriesOp, VectorOp, -}; -use nodedb_sql::parser::preprocess::lex::{find_ascii_keyword, keyword_position_outside_literals}; -use nodedb_types::starts_with_ascii_case_insensitive; - -const RETURNING_KEYWORD: &str = "RETURNING"; - -/// Check if a DML statement contains a RETURNING clause and strip it. -/// -/// Returns `(cleaned_sql, returning_spec)`. The cleaned SQL has the -/// `RETURNING ...` suffix removed so DataFusion can parse it. -/// -/// RETURNING is honored on INSERT, UPSERT, UPDATE, DELETE and MERGE. Whether -/// the resulting plan has a slot to carry the clause is not decidable from the -/// statement text — it depends on the shape the planner produces — so that -/// judgement is made once the plan exists, by -/// [`refuse_unprojectable_insert_returning`]. -/// -/// Arithmetic expressions (e.g. `RETURNING stock * 2`) are rejected with -/// a typed error — only bare column names and `*` are supported. -pub fn strip_returning(sql: &str) -> Result<(String, Option), Error> { - let trimmed = sql.trim_start(); - - // Gated on the DML verbs rather than on "everything that is not a SELECT", - // so an unrelated statement whose text merely contains the word is never - // truncated at it. - if !starts_with_ascii_case_insensitive(trimmed, "INSERT") - && !starts_with_ascii_case_insensitive(trimmed, "UPSERT") - && !starts_with_ascii_case_insensitive(trimmed, "UPDATE") - && !starts_with_ascii_case_insensitive(trimmed, "DELETE") - && !starts_with_ascii_case_insensitive(trimmed, "MERGE") - { - return Ok((sql.to_string(), None)); - } - - if let Some(pos) = keyword_position_outside_literals(sql, RETURNING_KEYWORD) { - let cleaned = sql[..pos].trim_end().to_string(); - let columns_str = sql[pos + RETURNING_KEYWORD.len()..].trim(); - let spec = parse_returning_columns(columns_str)?; - Ok((cleaned, Some(spec))) - } else { - Ok((sql.to_string(), None)) - } -} - -/// Refuse an `INSERT ... RETURNING` whose plan SHAPE has nowhere to carry the -/// clause. -/// -/// Every engine now carries it on its insert op — document (schemaless and -/// strict), key-value, columnar, spatial, timeseries, and vector-primary each -/// own a `returning` slot paired with an `rls_filters` read gate, so the -/// statement returns the STORED post-image bounded by the read policy. What -/// remains here is not an engine gap but a plan-shape one: `INSERT ... SELECT` -/// never reaches the Data Plane as a single insert op, so there is no slot on -/// it for the clause to ride in, whatever engine it targets. -/// -/// Refusing is the honest answer. Silently dropping the clause answered a -/// statement that asked for rows with a bare command tag, and nothing anywhere -/// said the request had been discarded. -/// -/// Still runs against the plan rather than the statement text: the expansion -/// that removes the slot is a planning decision, not a syntactic one. -pub fn refuse_unprojectable_insert_returning(plan: &PhysicalPlan) -> Result<(), Error> { - let unsupported = match plan { - // `INSERT ... SELECT` never reaches the Data Plane as this op: it is - // expanded on the Control Plane into fresh-surrogate insert tasks whose - // rows the expander, not the plan, decides — so there is no slot on - // this plan for the clause to ride in. - PhysicalPlan::Document(DocumentOp::InsertSelect { .. }) => "INSERT ... SELECT", - // Exchange wraps an unresolved child; judge the child. - PhysicalPlan::Query(QueryOp::Exchange(op)) => { - return refuse_unprojectable_insert_returning(&op.child); - } - // Everything else either carries the clause already or is not an - // insert. Enumerated per engine rather than via a catch-all so a new - // `PhysicalPlan` variant forces a decision instead of silently - // inheriting "supported" and dropping the clause. - PhysicalPlan::Document(_) - | PhysicalPlan::Kv(_) - | PhysicalPlan::Vector(_) - | PhysicalPlan::Graph(_) - | PhysicalPlan::Text(_) - | PhysicalPlan::Columnar(_) - | PhysicalPlan::Timeseries(_) - | PhysicalPlan::Spatial(_) - | PhysicalPlan::Crdt(_) - | PhysicalPlan::Query(_) - | PhysicalPlan::Meta(_) - | PhysicalPlan::Array(_) - | PhysicalPlan::ClusterArray(_) - | PhysicalPlan::ClusterEvent(_) => return Ok(()), - }; - Err(Error::BadRequest { - detail: format!( - "RETURNING is not supported on {unsupported}; it is supported on every engine's \ - direct INSERT — document collections (schemaless and strict), key-value, \ - columnar, spatial, timeseries, and vector-primary collections — and on UPDATE, \ - DELETE, and MERGE. Follow the insert with a SELECT on the inserted key to read \ - the stored rows." - ), - }) -} - -/// The error a row-returning write must fail with when an open transaction -/// buffers or stages it instead of executing it. -/// -/// Both in-transaction routes are structurally unable to answer the clause, and -/// for different reasons — which is why this refuses rather than returning an -/// empty row set: -/// -/// - A **buffered** write performs no engine work at all until COMMIT, so at -/// statement time there is no stored row to project. Nothing could be -/// returned however the response were shaped. -/// - A **staged** write does touch the transaction overlay, but every staging -/// handler answers with an affected-count payload; the one payload-bearing -/// staged outcome is reserved for the atomic key-value ops that compute a -/// value. No staged write carries a row image back. -/// -/// COMMIT then answers with a single tag for the whole transaction, so the rows -/// cannot be surfaced later either. Reporting success with no rows is the exact -/// silence this clause exists to remove, so the statement is refused and says -/// which limitation it hit. Verb-agnostic on purpose: it fires for any plan the -/// shaper classifies as row-returning, so INSERT, UPSERT, UPDATE, DELETE and -/// MERGE all behave identically inside a transaction. -pub fn in_transaction_returning_unsupported() -> Error { - Error::BadRequest { - detail: "RETURNING is not supported inside an explicit transaction: the write is staged \ - or buffered until COMMIT, so it has no stored row to project at this point. Run \ - the statement in autocommit, or follow the write with a SELECT after COMMIT." - .to_string(), - } -} - -/// Inject a RETURNING spec into a DML physical plan variant. -/// -/// Only `PointInsert`, `PointPut`, `BatchInsert`, `Upsert`, `PointUpdate`, -/// `BulkUpdate`, `PointDelete`, `BulkDelete`, `UpdateFromJoin`, `Merge`, the KV -/// `Insert` / `InsertIfAbsent` / `InsertOnConflictUpdate` / `Put` / `BatchPut` -/// ops, the columnar `Insert`, the timeseries `Ingest`, the vector -/// `DirectUpsert`, and the CRDT `DocUpsert` / `DocDelete` ops are affected. -/// Every other variant is left unchanged — an insert shape among them has -/// already been refused by [`refuse_unprojectable_insert_returning`], which -/// runs first. -pub fn inject_returning_spec(plan: &mut PhysicalPlan, spec: ReturningSpec) { - match plan { - PhysicalPlan::Document(DocumentOp::PointInsert { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Kv(KvOp::Insert { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Kv(KvOp::InsertIfAbsent { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Kv(KvOp::InsertOnConflictUpdate { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Kv(KvOp::Put { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Kv(KvOp::BatchPut { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Columnar(ColumnarOp::Insert { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Timeseries(TimeseriesOp::Ingest { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Vector(VectorOp::DirectUpsert { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Document(DocumentOp::PointPut { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Document(DocumentOp::BatchInsert { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Document(DocumentOp::Upsert { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Document(DocumentOp::PointUpdate { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Document(DocumentOp::BulkUpdate { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Document(DocumentOp::PointDelete { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Document(DocumentOp::BulkDelete { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Document(DocumentOp::UpdateFromJoin { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Document(DocumentOp::Merge { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Crdt(CrdtOp::DocUpsert { returning, .. }) => { - *returning = Some(spec); - } - PhysicalPlan::Crdt(CrdtOp::DocDelete { returning, .. }) => { - *returning = Some(spec); - } - _ => {} - } -} - -/// Parse the column list that appears after the RETURNING keyword. -/// -/// Supports: -/// - `*` -/// - `col1, col2` -/// - `col1 AS alias1, col2` -/// -/// Rejects arithmetic expressions (e.g. `stock * 2`) with a typed error. -fn parse_returning_columns(columns_str: &str) -> Result { - let columns_str = columns_str.trim(); - if columns_str == "*" { - return Ok(ReturningSpec { - columns: ReturningColumns::Star, - }); - } - - let mut items = Vec::new(); - for raw_item in columns_str.split(',') { - let item = raw_item.trim(); - if item.is_empty() { - continue; - } - - // Reject arithmetic: contains operators that are not part of a name. - if contains_arithmetic(item) { - return Err(Error::BadRequest { - detail: format!( - "RETURNING expression '{item}' is not supported; \ - only bare column names and RETURNING * are allowed" - ), - }); - } - - // Parse `name [AS alias]` — case-insensitive AS. - if let Some(as_pos) = find_ascii_keyword(item, "AS") { - let name = item[..as_pos].trim().to_string(); - let alias = item[as_pos + 2..].trim().to_string(); - if name.is_empty() || alias.is_empty() { - return Err(Error::BadRequest { - detail: format!("invalid RETURNING column expression: '{item}'"), - }); - } - items.push(ReturningItem { - name, - alias: Some(alias), - }); - } else { - let name = item.to_string(); - if !is_valid_column_name(&name) { - return Err(Error::BadRequest { - detail: format!( - "RETURNING expression '{name}' is not supported; \ - only bare column names and RETURNING * are allowed" - ), - }); - } - items.push(ReturningItem { name, alias: None }); - } - } - - if items.is_empty() { - return Err(Error::BadRequest { - detail: "empty RETURNING column list".into(), - }); - } - - Ok(ReturningSpec { - columns: ReturningColumns::Named(items), - }) -} - -/// Return true if the expression token contains arithmetic operators -/// (*, /, +, -) outside of quoted identifiers. -fn contains_arithmetic(expr: &str) -> bool { - let mut in_quote = false; - let mut prev = '\0'; - for ch in expr.chars() { - if ch == '"' { - in_quote = !in_quote; - prev = ch; - continue; - } - if in_quote { - prev = ch; - continue; - } - if matches!(ch, '+' | '/' | '%') { - return true; - } - // `-` is arithmetic only when not a leading sign or part of an identifier. - if ch == '-' && (prev.is_ascii_alphanumeric() || prev == '_') { - return true; - } - // `*` is arithmetic when preceded by an identifier character. - if ch == '*' && (prev.is_ascii_alphanumeric() || prev == '_') { - return true; - } - prev = ch; - } - false -} - -/// Return true if the given name is a valid bare identifier (letters, digits, underscores). -fn is_valid_column_name(name: &str) -> bool { - if name.is_empty() { - return false; - } - name.chars() - .all(|c| c.is_ascii_alphanumeric() || c == '_' || c == '.') -} - -#[cfg(test)] -mod tests { - use super::*; - use nodedb_types::{DatabaseId, QualifiedCollection}; - - /// The document engines carry the clause, so it is stripped and parsed like - /// any other verb's — the statement text alone cannot decide the engine, so - /// nothing is refused here. - #[test] - fn insert_returning_is_stripped_and_parsed() { - let (sql, spec) = - strip_returning("INSERT INTO items (id, name) VALUES ('a', 'alpha') RETURNING *") - .expect("INSERT RETURNING must plan"); - assert_eq!(sql, "INSERT INTO items (id, name) VALUES ('a', 'alpha')"); - assert_eq!(spec.expect("spec").columns, ReturningColumns::Star); - - let (sql, spec) = strip_returning("insert into items (id) values ('a') returning id AS k") - .expect("INSERT RETURNING must plan"); - assert_eq!(sql, "insert into items (id) values ('a')"); - assert_eq!( - spec.expect("spec").columns, - ReturningColumns::Named(vec![ReturningItem { - name: "id".into(), - alias: Some("k".into()), - }]) - ); - } - - /// An insert shape with no `returning` slot is refused at the plan, naming - /// the shape and where the clause IS honored. Silently dropping it left - /// the caller with a command tag for a statement that asked for rows. - #[test] - fn an_insert_plan_with_no_returning_slot_is_refused() { - let plan = PhysicalPlan::Document(DocumentOp::InsertSelect { - target_collection: QualifiedCollection::new(DatabaseId::DEFAULT, "dst"), - source_collection: QualifiedCollection::new(DatabaseId::DEFAULT, "src"), - source_filters: Vec::new(), - source_limit: 0, - column_map: Vec::new(), - }); - let detail = refuse_unprojectable_insert_returning(&plan) - .expect_err("an INSERT ... SELECT cannot carry the clause") - .to_string(); - assert!( - detail.contains("INSERT ... SELECT") && detail.contains("document"), - "the refusal must name the plan shape and where it IS supported; got {detail}" - ); - } - - /// A vector-primary upsert now carries the clause, so the same gate admits - /// it. Pinned beside the refusal above for the same reason the columnar and - /// timeseries cases are: an engine dropped from the refusal without gaining - /// the slot silently drops the clause, and only asserting both halves - /// catches that. - #[test] - fn a_vector_primary_upsert_plan_is_admitted() { - let plan = PhysicalPlan::Vector(VectorOp::DirectUpsert { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "vectors"), - field: "emb".into(), - surrogate: nodedb_types::Surrogate::ZERO, - vector: Vec::new(), - payload: Vec::new(), - quantization: nodedb_types::VectorQuantization::None, - storage_dtype: nodedb_types::VectorStorageDtype::F32, - payload_indexes: Vec::new(), - returning: None, - rls_filters: Vec::new(), - }); - assert!(refuse_unprojectable_insert_returning(&plan).is_ok()); - } - - /// A timeseries ingest now carries the clause, so the same gate admits it. - /// Pinned beside the refusal above for the same reason the columnar case is: - /// an engine dropped from the refusal without gaining the slot silently - /// drops the clause, and only asserting both halves catches that. - #[test] - fn a_timeseries_ingest_plan_is_admitted() { - let plan = PhysicalPlan::Timeseries(TimeseriesOp::Ingest { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "metrics"), - payload: Vec::new(), - format: "ilp".into(), - wal_lsn: None, - surrogates: Vec::new(), - provenance: None, - rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), - returning: None, - rls_filters: Vec::new(), - }); - assert!(refuse_unprojectable_insert_returning(&plan).is_ok()); - } - - /// A columnar insert now carries the clause, so the same gate admits it. - /// This is the assertion that would fail if the columnar arm were ever - /// restored to the refusal while the op kept its `returning` slot — the - /// combination that silently drops the clause. - #[test] - fn a_columnar_insert_plan_is_admitted() { - let plan = PhysicalPlan::Columnar(ColumnarOp::Insert { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "metrics"), - payload: Vec::new(), - format: "msgpack".into(), - intent: nodedb_physical::physical_plan::ColumnarInsertIntent::Insert, - on_conflict_updates: Vec::new(), - surrogates: Vec::new(), - schema_bytes: Vec::new(), - provenance: None, - wal_lsn: None, - rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), - returning: None, - rls_filters: Vec::new(), - }); - assert!(refuse_unprojectable_insert_returning(&plan).is_ok()); - } - - /// A document insert carries the clause, so the same gate admits it. - #[test] - fn a_document_insert_plan_is_admitted() { - let plan = PhysicalPlan::Document(DocumentOp::PointInsert { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "items"), - document_id: "a".into(), - value: Vec::new(), - if_absent: false, - surrogate: nodedb_types::Surrogate::ZERO, - returning: None, - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - deferred_sum_targets: Vec::new(), - }); - assert!(refuse_unprojectable_insert_returning(&plan).is_ok()); - } - - /// An INSERT with no such clause is untouched — planning must not turn - /// ordinary inserts into errors. - #[test] - fn a_plain_insert_is_untouched() { - let sql = "INSERT INTO items (id, name) VALUES ('a', 'alpha')"; - let (out, spec) = strip_returning(sql).expect("a plain insert must plan"); - assert_eq!(out, sql); - assert!(spec.is_none()); - } - - /// The word inside a string literal is data, not a clause. - #[test] - fn returning_inside_a_string_literal_is_not_a_clause() { - let sql = "INSERT INTO items (id, note) VALUES ('a', 'RETURNING soon')"; - let (out, spec) = strip_returning(sql).expect("a quoted keyword is not a clause"); - assert_eq!(out, sql); - assert!(spec.is_none()); - } - - /// Only DML verbs are scanned for the clause: a SELECT whose column name - /// merely embeds the word is left alone. - #[test] - fn a_non_dml_statement_is_not_scanned_for_the_clause() { - let sql = "SELECT returning_count FROM items"; - let (out, spec) = strip_returning(sql).expect("a select must pass through"); - assert_eq!(out, sql); - assert!(spec.is_none()); - } - - #[test] - fn strips_star_returning_from_update() { - let (sql, spec) = - strip_returning("UPDATE products SET stock = 1 WHERE id = 'p1' RETURNING *").unwrap(); - assert_eq!(sql, "UPDATE products SET stock = 1 WHERE id = 'p1'"); - let spec = spec.unwrap(); - assert_eq!(spec.columns, ReturningColumns::Star); - } - - #[test] - fn strips_named_columns_returning_from_update() { - let (sql, spec) = strip_returning( - "UPDATE products SET stock = stock - 1 WHERE id = 'p1' RETURNING id, stock", - ) - .unwrap(); - assert_eq!(sql, "UPDATE products SET stock = stock - 1 WHERE id = 'p1'"); - let spec = spec.unwrap(); - assert_eq!( - spec.columns, - ReturningColumns::Named(vec![ - ReturningItem { - name: "id".into(), - alias: None - }, - ReturningItem { - name: "stock".into(), - alias: None - }, - ]) - ); - } - - #[test] - fn strips_star_returning_from_delete() { - let (sql, spec) = - strip_returning("DELETE FROM products WHERE id = 'p1' RETURNING *").unwrap(); - assert_eq!(sql, "DELETE FROM products WHERE id = 'p1'"); - let spec = spec.unwrap(); - assert_eq!(spec.columns, ReturningColumns::Star); - } - - #[test] - fn strips_named_returning_from_delete() { - let (sql, spec) = - strip_returning("DELETE FROM products WHERE id = 'p1' RETURNING id").unwrap(); - assert_eq!(sql, "DELETE FROM products WHERE id = 'p1'"); - let spec = spec.unwrap(); - assert_eq!( - spec.columns, - ReturningColumns::Named(vec![ReturningItem { - name: "id".into(), - alias: None - }]) - ); - } - - #[test] - fn strips_star_returning_from_merge() { - let (sql, spec) = strip_returning( - "MERGE INTO products t USING staging s ON t.id = s.id \ - WHEN MATCHED THEN UPDATE SET stock = s.stock RETURNING *", - ) - .unwrap(); - assert_eq!( - sql, - "MERGE INTO products t USING staging s ON t.id = s.id \ - WHEN MATCHED THEN UPDATE SET stock = s.stock" - ); - assert_eq!(spec.unwrap().columns, ReturningColumns::Star); - } - - #[test] - fn strips_named_returning_from_merge() { - let (sql, spec) = strip_returning( - "MERGE INTO products t USING staging s ON t.id = s.id \ - WHEN NOT MATCHED THEN INSERT (id) VALUES (s.id) RETURNING id, stock", - ) - .unwrap(); - assert_eq!( - sql, - "MERGE INTO products t USING staging s ON t.id = s.id \ - WHEN NOT MATCHED THEN INSERT (id) VALUES (s.id)" - ); - assert_eq!( - spec.unwrap().columns, - ReturningColumns::Named(vec![ - ReturningItem { - name: "id".into(), - alias: None - }, - ReturningItem { - name: "stock".into(), - alias: None - }, - ]) - ); - } - - #[test] - fn merge_without_returning_is_unchanged() { - let original = "MERGE INTO products t USING staging s ON t.id = s.id \ - WHEN MATCHED THEN DELETE"; - let (sql, spec) = strip_returning(original).unwrap(); - assert!(spec.is_none()); - assert_eq!(sql, original); - } - - #[test] - fn no_returning() { - let (sql, spec) = strip_returning("UPDATE products SET stock = 0 WHERE id = 'p1'").unwrap(); - assert!(spec.is_none()); - assert_eq!(sql, "UPDATE products SET stock = 0 WHERE id = 'p1'"); - } - - #[test] - fn returning_inside_identifier_not_treated_as_keyword() { - // A collection/table whose name embeds "returning" (with `_` as an - // identifier boundary) must NOT match the RETURNING keyword inside the - // name — the real keyword is the trailing one after WHERE. - let (sql, spec) = - strip_returning("DELETE FROM orders_returning WHERE id = 'p1' RETURNING *").unwrap(); - assert_eq!(sql, "DELETE FROM orders_returning WHERE id = 'p1'"); - assert_eq!(spec.unwrap().columns, ReturningColumns::Star); - - // Same identifier with no trailing RETURNING clause → no spec, unchanged. - let (sql, spec) = strip_returning("DELETE FROM orders_returning WHERE id = 'p1'").unwrap(); - assert!(spec.is_none()); - assert_eq!(sql, "DELETE FROM orders_returning WHERE id = 'p1'"); - } - - #[test] - fn returning_in_string_literal_ignored() { - let (sql, spec) = - strip_returning("UPDATE products SET note = 'RETURNING soon' WHERE id = 'p1'").unwrap(); - assert!(spec.is_none()); - assert_eq!( - sql, - "UPDATE products SET note = 'RETURNING soon' WHERE id = 'p1'" - ); - } - - #[test] - fn select_not_affected() { - let (sql, spec) = strip_returning("SELECT * FROM products").unwrap(); - assert!(spec.is_none()); - assert_eq!(sql, "SELECT * FROM products"); - } - - #[test] - fn case_insensitive() { - let (sql, spec) = - strip_returning("update products set stock = 0 where id = 'p1' returning id").unwrap(); - let spec = spec.unwrap(); - assert_eq!(sql, "update products set stock = 0 where id = 'p1'"); - assert_eq!( - spec.columns, - ReturningColumns::Named(vec![ReturningItem { - name: "id".into(), - alias: None - }]) - ); - } - - #[test] - fn unicode_identifier_before_returning_preserves_original_offsets() { - let (sql, spec) = strip_returning("DELETE FROM tffff RETURNING *").unwrap(); - assert_eq!(sql, "DELETE FROM tffff"); - assert_eq!(spec.unwrap().columns, ReturningColumns::Star); - } - - #[test] - fn unicode_returning_column_before_alias_preserves_original_offsets() { - let (_, spec) = strip_returning("UPDATE t SET x = 1 RETURNING ffff AS alias").unwrap(); - assert_eq!( - spec.unwrap().columns, - ReturningColumns::Named(vec![ReturningItem { - name: "ffff".into(), - alias: Some("alias".into()), - }]) - ); - } - - #[test] - fn arithmetic_in_returning_is_error() { - let result = strip_returning("UPDATE t SET x=1 RETURNING x*2"); - assert!(result.is_err()); - let e = result.unwrap_err().to_string(); - assert!( - e.contains("not supported") || e.contains("expression"), - "unexpected error: {e}" - ); - } - - #[test] - fn returning_with_alias() { - let (sql, spec) = - strip_returning("UPDATE t SET x=2 WHERE id='a' RETURNING x AS new_x").unwrap(); - assert_eq!(sql, "UPDATE t SET x=2 WHERE id='a'"); - let spec = spec.unwrap(); - assert_eq!( - spec.columns, - ReturningColumns::Named(vec![ReturningItem { - name: "x".into(), - alias: Some("new_x".into()), - }]) - ); - } - - #[test] - fn output_names_star_returns_none() { - let spec = ReturningSpec { - columns: ReturningColumns::Star, - }; - assert!(spec.output_names().is_none()); - } - - #[test] - fn output_names_named_uses_aliases() { - let spec = ReturningSpec { - columns: ReturningColumns::Named(vec![ - ReturningItem { - name: "id".into(), - alias: None, - }, - ReturningItem { - name: "x".into(), - alias: Some("val".into()), - }, - ]), - }; - assert_eq!( - spec.output_names(), - Some(vec!["id".to_string(), "val".to_string()]) - ); - } -} diff --git a/nodedb/src/control/server/shared/returning/clause.rs b/nodedb/src/control/server/shared/returning/clause.rs new file mode 100644 index 000000000..dcbaecd46 --- /dev/null +++ b/nodedb/src/control/server/shared/returning/clause.rs @@ -0,0 +1,298 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Resolving a stripped `RETURNING` item list against the planned DML target. +//! +//! Once the plans exist, the item list resolves against the target +//! collection through `nodedb_sql`: a bare column and a star ride to the Data +//! Plane as a name projection, and every other item is a Control-Plane +//! computed column the response stage evaluates per returned row. The +//! Data-Plane spec derived from that projection is what the inject step +//! attaches to the plan. + +use nodedb_physical::physical_plan::{ReturningColumns, ReturningItem, ReturningSpec}; + +use crate::Error; +use crate::control::planner::plan_error_map::map_plan_error; +use nodedb_sql::catalog::SqlCatalog; +use nodedb_sql::types::SqlPlan; +use nodedb_sql::types::plan::referenced_columns; +use nodedb_sql::types::query::Projection; + +/// A resolved RETURNING clause: the Control-Plane projection and the +/// Data-Plane spec derived from it. +#[derive(Debug, Clone)] +pub struct ReturningClause { + /// The announced output columns, in clause order. A `CpComputed` entry is + /// evaluated by the response shaper per returned row. + pub projection: Vec, + /// The base columns the Data Plane returns by name. Display names and + /// drops are the shaper's job. + pub spec: ReturningSpec, +} + +/// Resolve the item text after `RETURNING` against `target`. +/// +/// `tenant_id` scopes the planner error mapping so an unknown target names +/// the tenant it was looked up under. +pub fn resolve_returning_clause( + items_sql: &str, + target: &str, + catalog: &dyn SqlCatalog, + tenant_id: crate::types::TenantId, +) -> crate::Result { + let projection = nodedb_sql::resolve_returning_items(items_sql, target, catalog) + .map_err(|error| map_plan_error(error, tenant_id))?; + let spec = spec_from_projection(&projection); + Ok(ReturningClause { projection, spec }) +} + +/// The Data-Plane spec a resolved projection needs: `Star` for a lone star, +/// otherwise every named column plus every base column a computed entry +/// reads, deduplicated in first-seen order. +/// +/// Names are bare: a stored row keys its fields by column name, and the +/// Control-Plane evaluator reads a column by its bare name too, so a +/// `target.col` qualifier is dropped here. +fn spec_from_projection(projection: &[Projection]) -> ReturningSpec { + if matches!(projection, [Projection::Star]) { + return ReturningSpec { + columns: ReturningColumns::Star, + }; + } + let mut names: Vec = Vec::with_capacity(projection.len()); + let mut push = |qualified: &str| { + let bare = qualified.rsplit('.').next().unwrap_or(qualified); + if !names.iter().any(|n| n == bare) { + names.push(bare.to_string()); + } + }; + for entry in projection { + match entry { + Projection::Column(name) => push(name), + Projection::Computed { expr, .. } | Projection::CpComputed { expr, .. } => { + for name in referenced_columns(expr) { + push(&name); + } + } + // A star among named items still needs every stored column; the + // shaper projects the announced list onto whatever came back. + Projection::Star | Projection::QualifiedStar(_) => { + return ReturningSpec { + columns: ReturningColumns::Star, + }; + } + } + } + ReturningSpec { + columns: ReturningColumns::Named( + names + .into_iter() + .map(|name| ReturningItem { name, alias: None }) + .collect(), + ), + } +} + +/// The collection a DML plan's RETURNING clause projects, `None` for a plan +/// that has no such clause. Exhaustive so a new plan variant decides here. +pub fn returning_target_collection(plans: &[SqlPlan]) -> Option { + let plan = plans.first()?; + match plan { + SqlPlan::Insert { collection, .. } + | SqlPlan::KvInsert { collection, .. } + | SqlPlan::Upsert { collection, .. } + | SqlPlan::Update { collection, .. } + | SqlPlan::UpdateFrom { collection, .. } + | SqlPlan::Delete { collection, .. } + | SqlPlan::TimeseriesIngest { collection, .. } + | SqlPlan::VectorPrimaryInsert { collection, .. } => Some(collection.clone()), + SqlPlan::Merge { target, .. } | SqlPlan::InsertSelect { target, .. } => { + Some(target.clone()) + } + SqlPlan::ConstantResult { .. } + | SqlPlan::Scan { .. } + | SqlPlan::PointGet { .. } + | SqlPlan::DocumentIndexLookup { .. } + | SqlPlan::RangeScan { .. } + | SqlPlan::Truncate { .. } + | SqlPlan::Join { .. } + | SqlPlan::Aggregate { .. } + | SqlPlan::TimeseriesScan { .. } + | SqlPlan::VectorSearch { .. } + | SqlPlan::MultiVectorSearch { .. } + | SqlPlan::SparseSearch { .. } + | SqlPlan::TextSearch { .. } + | SqlPlan::HybridSearch { .. } + | SqlPlan::HybridSearchTriple { .. } + | SqlPlan::SpatialScan { .. } + | SqlPlan::Union { .. } + | SqlPlan::Intersect { .. } + | SqlPlan::Except { .. } + | SqlPlan::RecursiveScan { .. } + | SqlPlan::RecursiveValue { .. } + | SqlPlan::Cte { .. } + | SqlPlan::Subquery { .. } + | SqlPlan::CreateArray { .. } + | SqlPlan::DropArray { .. } + | SqlPlan::AlterArray { .. } + | SqlPlan::InsertArray { .. } + | SqlPlan::DeleteArray { .. } + | SqlPlan::ArraySlice { .. } + | SqlPlan::ArrayProject { .. } + | SqlPlan::ArrayAgg { .. } + | SqlPlan::ArrayElementwise { .. } + | SqlPlan::ArrayFlush { .. } + | SqlPlan::ArrayCompact { .. } + | SqlPlan::LateralTopK { .. } + | SqlPlan::LateralLoop { .. } + | SqlPlan::CreateIndex { .. } + | SqlPlan::DropIndex { .. } => None, + } +} + +/// Resolve `returning_items` against the target the planned statement names. +/// +/// `None` when the statement carries no clause. A clause on a plan with no +/// RETURNING target is refused by name rather than dropped. +pub fn resolve_returning_for_plans( + plans: &[SqlPlan], + returning_items: Option<&str>, + catalog: &dyn SqlCatalog, + tenant_id: crate::types::TenantId, +) -> crate::Result> { + let Some(items) = returning_items else { + return Ok(None); + }; + let Some(target) = returning_target_collection(plans) else { + let shape = plans + .first() + .map_or("an empty statement", SqlPlan::variant_name); + return Err(Error::BadRequest { + detail: format!( + "RETURNING is not supported on {shape}; it is supported on INSERT, UPSERT, \ + UPDATE, DELETE, and MERGE against a collection" + ), + }); + }; + resolve_returning_clause(items, &target, catalog, tenant_id).map(Some) +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_sql::types_expr::{BinaryOp, SqlExpr, SqlValue}; + + fn column(name: &str) -> SqlExpr { + SqlExpr::Column { + table: None, + name: name.into(), + } + } + + /// A lone star asks the Data Plane for every stored column. + #[test] + fn spec_for_a_star_is_star() { + let spec = spec_from_projection(&[Projection::Star]); + assert_eq!(spec.columns, ReturningColumns::Star); + } + + /// Named columns and the base columns computed entries read reach the + /// Data Plane as bare names, deduplicated in first-seen order, with no + /// alias: display names are the shaper's job. + #[test] + fn spec_lists_named_and_referenced_base_columns_once() { + let projection = vec![ + Projection::Column("id".into()), + Projection::CpComputed { + expr: SqlExpr::BinaryOp { + left: Box::new(column("score")), + op: BinaryOp::Add, + right: Box::new(column("id")), + }, + alias: "d".into(), + }, + Projection::Computed { + expr: column("score"), + alias: "s".into(), + }, + ]; + let spec = spec_from_projection(&projection); + assert_eq!( + spec.columns, + ReturningColumns::Named(vec![ + ReturningItem { + name: "id".into(), + alias: None, + }, + ReturningItem { + name: "score".into(), + alias: None, + }, + ]) + ); + } + + /// A computed entry that reads no column (a bare accessor) adds nothing + /// to the Data Plane list. + #[test] + fn spec_for_a_bare_accessor_reads_only_the_named_columns() { + let projection = vec![ + Projection::Column("id".into()), + Projection::CpComputed { + expr: SqlExpr::Function { + name: "nextval".into(), + args: vec![SqlExpr::Literal(SqlValue::String("s".into()))], + distinct: false, + }, + alias: "n".into(), + }, + ]; + let spec = spec_from_projection(&projection); + assert_eq!( + spec.columns, + ReturningColumns::Named(vec![ReturningItem { + name: "id".into(), + alias: None, + }]) + ); + } + + #[test] + fn a_read_plan_has_no_returning_target() { + let plan = SqlPlan::ConstantResult { + columns: Vec::new(), + values: Vec::new(), + volatile: false, + }; + assert!(returning_target_collection(&[plan]).is_none()); + assert!(returning_target_collection(&[]).is_none()); + } + + #[test] + fn output_names_star_returns_none() { + let spec = ReturningSpec { + columns: ReturningColumns::Star, + }; + assert!(spec.output_names().is_none()); + } + + #[test] + fn output_names_named_uses_aliases() { + let spec = ReturningSpec { + columns: ReturningColumns::Named(vec![ + ReturningItem { + name: "id".into(), + alias: None, + }, + ReturningItem { + name: "x".into(), + alias: Some("val".into()), + }, + ]), + }; + assert_eq!( + spec.output_names(), + Some(vec!["id".to_string(), "val".to_string()]) + ); + } +} diff --git a/nodedb/src/control/server/shared/returning/inject.rs b/nodedb/src/control/server/shared/returning/inject.rs new file mode 100644 index 000000000..3420c639d --- /dev/null +++ b/nodedb/src/control/server/shared/returning/inject.rs @@ -0,0 +1,302 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Deciding whether a planned DML task can carry a `RETURNING` spec, and +//! attaching it to the plan variants that can. +//! +//! Every engine carries the clause on its insert op — document (schemaless +//! and strict), key-value, columnar, spatial, timeseries, and vector-primary +//! each own a `returning` slot paired with an `rls_filters` read gate, so the +//! statement returns the stored post-image bounded by the read policy. The +//! one plan shape with nowhere to carry it is `INSERT ... SELECT`, refused +//! rather than silently dropped. + +use nodedb_physical::physical_plan::{ + ColumnarOp, CrdtOp, DocumentOp, KvOp, QueryOp, ReturningSpec, TimeseriesOp, VectorOp, +}; +use nodedb_physical::physical_task::PhysicalTask; + +use crate::Error; +use crate::bridge::envelope::PhysicalPlan; + +/// Attach `spec` to every planned task, refusing any insert shape that has +/// nowhere to carry it rather than dropping the clause in silence. +pub fn attach_returning_spec( + tasks: &mut [PhysicalTask], + spec: &ReturningSpec, +) -> crate::Result<()> { + for task in tasks.iter_mut() { + refuse_unprojectable_insert_returning(&task.plan)?; + inject_returning_spec(&mut task.plan, spec.clone()); + } + Ok(()) +} + +/// Refuse an `INSERT ... RETURNING` whose plan SHAPE has nowhere to carry the +/// clause. +/// +/// Every engine carries it on its insert op — document (schemaless and +/// strict), key-value, columnar, spatial, timeseries, and vector-primary each +/// own a `returning` slot paired with an `rls_filters` read gate, so the +/// statement returns the STORED post-image bounded by the read policy. What +/// remains here is not an engine gap but a plan-shape one: `INSERT ... SELECT` +/// never reaches the Data Plane as a single insert op, so there is no slot on +/// it for the clause to ride in, whatever engine it targets. +/// +/// Refusing is the honest answer. Silently dropping the clause answers a +/// statement that asked for rows with a bare command tag, and nothing anywhere +/// says the request was discarded. +/// +/// Runs against the plan rather than the statement text: the expansion that +/// removes the slot is a planning decision, not a syntactic one. +pub fn refuse_unprojectable_insert_returning(plan: &PhysicalPlan) -> Result<(), Error> { + let unsupported = match plan { + // `INSERT ... SELECT` never reaches the Data Plane as this op: it is + // expanded on the Control Plane into fresh-surrogate insert tasks whose + // rows the expander, not the plan, decides — so there is no slot on + // this plan for the clause to ride in. + PhysicalPlan::Document(DocumentOp::InsertSelect { .. }) => "INSERT ... SELECT", + // Exchange wraps an unresolved child; judge the child. + PhysicalPlan::Query(QueryOp::Exchange(op)) => { + return refuse_unprojectable_insert_returning(&op.child); + } + // Everything else either carries the clause already or is not an + // insert. Enumerated per engine rather than via a catch-all so a new + // `PhysicalPlan` variant forces a decision instead of silently + // inheriting "supported" and dropping the clause. + PhysicalPlan::Document(_) + | PhysicalPlan::Kv(_) + | PhysicalPlan::Vector(_) + | PhysicalPlan::Graph(_) + | PhysicalPlan::Text(_) + | PhysicalPlan::Columnar(_) + | PhysicalPlan::Timeseries(_) + | PhysicalPlan::Spatial(_) + | PhysicalPlan::Crdt(_) + | PhysicalPlan::Query(_) + | PhysicalPlan::Meta(_) + | PhysicalPlan::Array(_) + | PhysicalPlan::ClusterArray(_) + | PhysicalPlan::ClusterEvent(_) => return Ok(()), + }; + Err(Error::BadRequest { + detail: format!( + "RETURNING is not supported on {unsupported}; it is supported on every engine's \ + direct INSERT — document collections (schemaless and strict), key-value, \ + columnar, spatial, timeseries, and vector-primary collections — and on UPDATE, \ + DELETE, and MERGE. Follow the insert with a SELECT on the inserted key to read \ + the stored rows." + ), + }) +} + +/// The error a row-returning write must fail with when an open transaction +/// buffers or stages it instead of executing it. +/// +/// Both in-transaction routes are structurally unable to answer the clause, and +/// for different reasons — which is why this refuses rather than returning an +/// empty row set: +/// +/// - A **buffered** write performs no engine work at all until COMMIT, so at +/// statement time there is no stored row to project. Nothing could be +/// returned however the response were shaped. +/// - A **staged** write does touch the transaction overlay, but every staging +/// handler answers with an affected-count payload; the one payload-bearing +/// staged outcome is reserved for the atomic key-value ops that compute a +/// value. No staged write carries a row image back. +/// +/// COMMIT then answers with a single tag for the whole transaction, so the rows +/// cannot be surfaced later either. Reporting success with no rows is the exact +/// silence this clause exists to remove, so the statement is refused and says +/// which limitation it hit. Verb-agnostic on purpose: it fires for any plan the +/// shaper classifies as row-returning, so INSERT, UPSERT, UPDATE, DELETE and +/// MERGE all behave identically inside a transaction. +pub fn in_transaction_returning_unsupported() -> Error { + Error::BadRequest { + detail: "RETURNING is not supported inside an explicit transaction: the write is staged \ + or buffered until COMMIT, so it has no stored row to project at this point. Run \ + the statement in autocommit, or follow the write with a SELECT after COMMIT." + .to_string(), + } +} + +/// Inject a RETURNING spec into a DML physical plan variant. +/// +/// Only `PointInsert`, `PointPut`, `BatchInsert`, `Upsert`, `PointUpdate`, +/// `BulkUpdate`, `PointDelete`, `BulkDelete`, `UpdateFromJoin`, `Merge`, the KV +/// `Insert` / `InsertIfAbsent` / `InsertOnConflictUpdate` / `Put` / `BatchPut` +/// ops, the columnar `Insert`, the timeseries `Ingest`, the vector +/// `DirectUpsert`, and the CRDT `DocUpsert` / `DocDelete` ops are affected. +/// Every other variant is left unchanged — an insert shape among them has +/// already been refused by [`refuse_unprojectable_insert_returning`], which +/// runs first. +pub fn inject_returning_spec(plan: &mut PhysicalPlan, spec: ReturningSpec) { + match plan { + PhysicalPlan::Document(DocumentOp::PointInsert { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Kv(KvOp::Insert { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Kv(KvOp::InsertIfAbsent { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Kv(KvOp::InsertOnConflictUpdate { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Kv(KvOp::Put { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Kv(KvOp::BatchPut { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Columnar(ColumnarOp::Insert { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Timeseries(TimeseriesOp::Ingest { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Vector(VectorOp::DirectUpsert { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Document(DocumentOp::PointPut { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Document(DocumentOp::BatchInsert { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Document(DocumentOp::Upsert { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Document(DocumentOp::PointUpdate { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Document(DocumentOp::BulkUpdate { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Document(DocumentOp::PointDelete { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Document(DocumentOp::BulkDelete { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Document(DocumentOp::UpdateFromJoin { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Document(DocumentOp::Merge { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Crdt(CrdtOp::DocUpsert { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Crdt(CrdtOp::DocDelete { returning, .. }) => { + *returning = Some(spec); + } + _ => {} + } +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_types::{DatabaseId, QualifiedCollection}; + + /// An insert shape with no `returning` slot is refused at the plan, naming + /// the shape and where the clause IS honored. Silently dropping it left + /// the caller with a command tag for a statement that asked for rows. + #[test] + fn an_insert_plan_with_no_returning_slot_is_refused() { + let plan = PhysicalPlan::Document(DocumentOp::InsertSelect { + target_collection: QualifiedCollection::new(DatabaseId::DEFAULT, "dst"), + source_collection: QualifiedCollection::new(DatabaseId::DEFAULT, "src"), + source_filters: Vec::new(), + source_limit: 0, + column_map: Vec::new(), + }); + let detail = refuse_unprojectable_insert_returning(&plan) + .expect_err("an INSERT ... SELECT cannot carry the clause") + .to_string(); + assert!( + detail.contains("INSERT ... SELECT") && detail.contains("document"), + "the refusal must name the plan shape and where it IS supported; got {detail}" + ); + } + + /// A vector-primary upsert carries the clause, so the same gate admits + /// it. Pinned beside the refusal above for the same reason the columnar and + /// timeseries cases are: an engine dropped from the refusal without gaining + /// the slot silently drops the clause, and only asserting both halves + /// catches that. + #[test] + fn a_vector_primary_upsert_plan_is_admitted() { + let plan = PhysicalPlan::Vector(VectorOp::DirectUpsert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "vectors"), + field: "emb".into(), + surrogate: nodedb_types::Surrogate::ZERO, + vector: Vec::new(), + payload: Vec::new(), + quantization: nodedb_types::VectorQuantization::None, + storage_dtype: nodedb_types::VectorStorageDtype::F32, + payload_indexes: Vec::new(), + returning: None, + rls_filters: Vec::new(), + }); + assert!(refuse_unprojectable_insert_returning(&plan).is_ok()); + } + + /// A timeseries ingest carries the clause, so the same gate admits it. + #[test] + fn a_timeseries_ingest_plan_is_admitted() { + let plan = PhysicalPlan::Timeseries(TimeseriesOp::Ingest { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "metrics"), + payload: Vec::new(), + format: "ilp".into(), + wal_lsn: None, + surrogates: Vec::new(), + provenance: None, + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: None, + rls_filters: Vec::new(), + }); + assert!(refuse_unprojectable_insert_returning(&plan).is_ok()); + } + + /// A columnar insert carries the clause, so the same gate admits it. + /// This is the assertion that fails if the columnar arm is ever restored + /// to the refusal while the op keeps its `returning` slot — the + /// combination that silently drops the clause. + #[test] + fn a_columnar_insert_plan_is_admitted() { + let plan = PhysicalPlan::Columnar(ColumnarOp::Insert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "metrics"), + payload: Vec::new(), + format: "msgpack".into(), + intent: nodedb_physical::physical_plan::ColumnarInsertIntent::Insert, + on_conflict_updates: Vec::new(), + surrogates: Vec::new(), + schema_bytes: Vec::new(), + provenance: None, + wal_lsn: None, + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: None, + rls_filters: Vec::new(), + }); + assert!(refuse_unprojectable_insert_returning(&plan).is_ok()); + } + + /// A document insert carries the clause, so the same gate admits it. + #[test] + fn a_document_insert_plan_is_admitted() { + let plan = PhysicalPlan::Document(DocumentOp::PointInsert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "items"), + document_id: "a".into(), + value: Vec::new(), + if_absent: false, + surrogate: nodedb_types::Surrogate::ZERO, + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + deferred_sum_targets: Vec::new(), + }); + assert!(refuse_unprojectable_insert_returning(&plan).is_ok()); + } +} diff --git a/nodedb/src/control/server/shared/returning/mod.rs b/nodedb/src/control/server/shared/returning/mod.rs new file mode 100644 index 000000000..4b5be68dd --- /dev/null +++ b/nodedb/src/control/server/shared/returning/mod.rs @@ -0,0 +1,24 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! RETURNING clause handling for DML statements: strip it from the text, +//! resolve its item list against the planned target, decide whether the +//! resulting plan can carry it, and attach it. +//! +//! Protocol-neutral: the pgwire planner, the neutral DDL router's `UPSERT` +//! path, and the prepared-statement Describe path all go through this +//! module, so a statement's clause is stripped, resolved, judged, and +//! attached identically on every transport. + +mod clause; +mod inject; +mod strip; + +// Re-export bridge types so callers only import from this module. +pub use nodedb_physical::physical_plan::{ReturningColumns, ReturningItem, ReturningSpec}; + +pub use clause::{ReturningClause, resolve_returning_clause, resolve_returning_for_plans}; +pub use inject::{ + attach_returning_spec, in_transaction_returning_unsupported, inject_returning_spec, + refuse_unprojectable_insert_returning, +}; +pub use strip::strip_returning; diff --git a/nodedb/src/control/server/shared/returning/strip.rs b/nodedb/src/control/server/shared/returning/strip.rs new file mode 100644 index 000000000..cb0b618e5 --- /dev/null +++ b/nodedb/src/control/server/shared/returning/strip.rs @@ -0,0 +1,199 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Stripping a `RETURNING` clause from raw DML SQL text before planning. +//! +//! The planner does not parse RETURNING on DML, so the clause is removed from +//! the raw SQL before planning; the item text is resolved later, once the +//! target collection is known. + +use crate::Error; +use nodedb_sql::parser::preprocess::lex::keyword_position_outside_literals; +use nodedb_types::starts_with_ascii_case_insensitive; + +const RETURNING_KEYWORD: &str = "RETURNING"; + +/// Check if a DML statement contains a RETURNING clause and strip it. +/// +/// Returns `(cleaned_sql, returning_items)`. The cleaned SQL has the +/// `RETURNING ...` suffix removed so the planner can parse it, and +/// `returning_items` is the raw item text after the keyword, resolved later by +/// [`super::resolve_returning_clause`] once the target collection is known. +/// +/// RETURNING is honored on INSERT, UPSERT, UPDATE, DELETE and MERGE. Whether +/// the resulting plan has a slot to carry the clause is not decidable from the +/// statement text — it depends on the shape the planner produces — so that +/// judgement is made once the plan exists, by +/// [`super::refuse_unprojectable_insert_returning`]. +pub fn strip_returning(sql: &str) -> Result<(String, Option), Error> { + let trimmed = sql.trim_start(); + + // Gated on the DML verbs rather than on "everything that is not a SELECT", + // so an unrelated statement whose text merely contains the word is never + // truncated at it. + if !starts_with_ascii_case_insensitive(trimmed, "INSERT") + && !starts_with_ascii_case_insensitive(trimmed, "UPSERT") + && !starts_with_ascii_case_insensitive(trimmed, "UPDATE") + && !starts_with_ascii_case_insensitive(trimmed, "DELETE") + && !starts_with_ascii_case_insensitive(trimmed, "MERGE") + { + return Ok((sql.to_string(), None)); + } + + if let Some(pos) = keyword_position_outside_literals(sql, RETURNING_KEYWORD) { + let cleaned = sql[..pos].trim_end().to_string(); + let items = sql[pos + RETURNING_KEYWORD.len()..].trim(); + if items.is_empty() { + return Err(Error::BadRequest { + detail: "empty RETURNING column list".into(), + }); + } + Ok((cleaned, Some(items.to_string()))) + } else { + Ok((sql.to_string(), None)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// The document engines carry the clause, so it is stripped like any + /// other verb's — the statement text alone cannot decide the engine, so + /// nothing is refused here. + #[test] + fn insert_returning_is_stripped() { + let (sql, items) = + strip_returning("INSERT INTO items (id, name) VALUES ('a', 'alpha') RETURNING *") + .expect("INSERT RETURNING must plan"); + assert_eq!(sql, "INSERT INTO items (id, name) VALUES ('a', 'alpha')"); + assert_eq!(items.as_deref(), Some("*")); + + let (sql, items) = strip_returning("insert into items (id) values ('a') returning id AS k") + .expect("INSERT RETURNING must plan"); + assert_eq!(sql, "insert into items (id) values ('a')"); + assert_eq!(items.as_deref(), Some("id AS k")); + } + + /// An INSERT with no such clause is untouched — planning must not turn + /// ordinary inserts into errors. + #[test] + fn a_plain_insert_is_untouched() { + let sql = "INSERT INTO items (id, name) VALUES ('a', 'alpha')"; + let (out, items) = strip_returning(sql).expect("a plain insert must plan"); + assert_eq!(out, sql); + assert!(items.is_none()); + } + + /// The word inside a string literal is data, not a clause. + #[test] + fn returning_inside_a_string_literal_is_not_a_clause() { + let sql = "INSERT INTO items (id, note) VALUES ('a', 'RETURNING soon')"; + let (out, items) = strip_returning(sql).expect("a quoted keyword is not a clause"); + assert_eq!(out, sql); + assert!(items.is_none()); + } + + /// Only DML verbs are scanned for the clause: a SELECT whose column name + /// merely embeds the word is left alone. + #[test] + fn a_non_dml_statement_is_not_scanned_for_the_clause() { + let sql = "SELECT returning_count FROM items"; + let (out, items) = strip_returning(sql).expect("a select must pass through"); + assert_eq!(out, sql); + assert!(items.is_none()); + } + + #[test] + fn strips_star_returning_from_update() { + let (sql, items) = + strip_returning("UPDATE products SET stock = 1 WHERE id = 'p1' RETURNING *").unwrap(); + assert_eq!(sql, "UPDATE products SET stock = 1 WHERE id = 'p1'"); + assert_eq!(items.as_deref(), Some("*")); + } + + #[test] + fn strips_named_columns_returning_from_update() { + let (sql, items) = strip_returning( + "UPDATE products SET stock = stock - 1 WHERE id = 'p1' RETURNING id, stock", + ) + .unwrap(); + assert_eq!(sql, "UPDATE products SET stock = stock - 1 WHERE id = 'p1'"); + assert_eq!(items.as_deref(), Some("id, stock")); + } + + #[test] + fn strips_returning_from_delete() { + let (sql, items) = + strip_returning("DELETE FROM products WHERE id = 'p1' RETURNING id").unwrap(); + assert_eq!(sql, "DELETE FROM products WHERE id = 'p1'"); + assert_eq!(items.as_deref(), Some("id")); + } + + #[test] + fn strips_returning_from_merge() { + let (sql, items) = strip_returning( + "MERGE INTO products t USING staging s ON t.id = s.id \ + WHEN NOT MATCHED THEN INSERT (id) VALUES (s.id) RETURNING id, stock", + ) + .unwrap(); + assert_eq!( + sql, + "MERGE INTO products t USING staging s ON t.id = s.id \ + WHEN NOT MATCHED THEN INSERT (id) VALUES (s.id)" + ); + assert_eq!(items.as_deref(), Some("id, stock")); + } + + #[test] + fn merge_without_returning_is_unchanged() { + let original = "MERGE INTO products t USING staging s ON t.id = s.id \ + WHEN MATCHED THEN DELETE"; + let (sql, items) = strip_returning(original).unwrap(); + assert!(items.is_none()); + assert_eq!(sql, original); + } + + #[test] + fn returning_inside_identifier_not_treated_as_keyword() { + // A collection whose name embeds "returning" (with `_` as an + // identifier boundary) must NOT match the RETURNING keyword inside the + // name — the real keyword is the trailing one after WHERE. + let (sql, items) = + strip_returning("DELETE FROM orders_returning WHERE id = 'p1' RETURNING *").unwrap(); + assert_eq!(sql, "DELETE FROM orders_returning WHERE id = 'p1'"); + assert_eq!(items.as_deref(), Some("*")); + + let (sql, items) = strip_returning("DELETE FROM orders_returning WHERE id = 'p1'").unwrap(); + assert!(items.is_none()); + assert_eq!(sql, "DELETE FROM orders_returning WHERE id = 'p1'"); + } + + #[test] + fn case_insensitive() { + let (sql, items) = + strip_returning("update products set stock = 0 where id = 'p1' returning id").unwrap(); + assert_eq!(sql, "update products set stock = 0 where id = 'p1'"); + assert_eq!(items.as_deref(), Some("id")); + } + + #[test] + fn unicode_identifier_before_returning_preserves_original_offsets() { + let (sql, items) = strip_returning("DELETE FROM tffff RETURNING *").unwrap(); + assert_eq!(sql, "DELETE FROM tffff"); + assert_eq!(items.as_deref(), Some("*")); + } + + /// The raw item text is kept verbatim: an expression is resolved later + /// against the target, never rejected at the strip. + #[test] + fn an_expression_survives_the_strip() { + let (sql, items) = strip_returning("UPDATE t SET x=1 RETURNING x*2 AS d").unwrap(); + assert_eq!(sql, "UPDATE t SET x=1"); + assert_eq!(items.as_deref(), Some("x*2 AS d")); + } + + #[test] + fn an_empty_item_list_is_refused() { + assert!(strip_returning("UPDATE t SET x=1 RETURNING ").is_err()); + } +} diff --git a/nodedb/tests/wire/cases/pgwire_returning_dml.rs b/nodedb/tests/wire/cases/pgwire_returning_dml.rs index a8ba13163..6f66b6cfe 100644 --- a/nodedb/tests/wire/cases/pgwire_returning_dml.rs +++ b/nodedb/tests/wire/cases/pgwire_returning_dml.rs @@ -537,20 +537,55 @@ async fn unicode_identifier_before_returning_preserves_connection() { // Arithmetic expression in RETURNING — error path // --------------------------------------------------------------------------- -/// A RETURNING clause containing an arithmetic expression must be rejected -/// with a typed error. NodeDB only supports column references and aliases -/// in RETURNING, not computed expressions. +/// An arithmetic `RETURNING` expression evaluates on the Control Plane +/// against the stored post-image, once per returned row. #[tokio::test] -async fn returning_arithmetic_expression_rejected() { +async fn returning_arithmetic_expression_evaluates() { let server = TestServer::start().await; seed_docs(&server).await; - server - .expect_error( - "UPDATE items SET score = 1 WHERE id = 'a' RETURNING score + 1", - "not supported", - ) - .await; + let rows = server + .query_rows("UPDATE items SET score = 1 WHERE id = 'a' RETURNING score + 1 AS next_score") + .await + .expect("an arithmetic RETURNING expression must evaluate"); + assert_eq!(rows, vec![vec!["2".to_string()]]); +} + +/// A function call in `RETURNING` evaluates against the returned row. +#[tokio::test] +async fn returning_function_expression_evaluates() { + let server = TestServer::start().await; + seed_docs(&server).await; + + let rows = server + .query_rows("UPDATE items SET score = 5 WHERE id = 'b' RETURNING id, upper(name) AS u") + .await + .expect("a function RETURNING expression must evaluate"); + assert_eq!(rows, vec![vec!["b".to_string(), "BETA".to_string()]]); +} + +/// The base columns an expression reads never leak into the result: the +/// client sees exactly the announced list. +#[tokio::test] +async fn returning_expression_keeps_only_requested_columns() { + let server = TestServer::start().await; + seed_docs(&server).await; + + let msgs = server + .client + .simple_query("UPDATE items SET score = 7 WHERE id = 'c' RETURNING score * 2 AS d") + .await + .expect("an expression-only RETURNING must succeed"); + let row = msgs + .iter() + .find_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(row) => Some(row), + _ => None, + }) + .expect("one returned row"); + assert_eq!(row.len(), 1, "only the requested column is shipped"); + assert_eq!(row.columns()[0].name(), "d"); + assert_eq!(row.get(0), Some("14")); } // --------------------------------------------------------------------------- From 2a755940b01ae322c28b9586290fdb3f156a7bf6 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 18 Sep 2026 12:04:24 +0800 Subject: [PATCH 12/15] feat(query): support RETURNING on KV UPDATE and DELETE Extend keyed and predicate-filtered KV delete/update ops with a returning spec and RLS filter bytes, so the executor can read the stored pre-image (delete) or post-image (update) of every affected row and project it per spec instead of a bare row count. Thread the new fields through the physical plan, planner RLS injection, WAL replication encode/decode, response shaping, and transaction staging. Split the KV scan and transfer dispatch arms out of dispatch.rs into their own files to keep it within size limits alongside the new params. --- nodedb-physical/src/physical_plan/kv/op.rs | 28 ++ nodedb/src/bridge/admission_chokepoint.rs | 2 + .../src/control/planner/calvin/write_class.rs | 2 + .../src/control/planner/rls_injection/kv.rs | 91 ++++++- .../dml/update_delete/delete.rs | 6 + .../dml/update_delete/update.rs | 6 + .../native/dispatch/plan_builder/document.rs | 3 + .../server/native/dispatch/plan_builder/kv.rs | 3 + .../src/control/server/pgwire/handler/plan.rs | 2 + .../src/control/server/resp/handler_hash.rs | 3 + .../control/server/resp/handler_kv/strings.rs | 3 + .../src/control/server/resp/handler_sorted.rs | 3 + .../server/response_shape/types/plan_kind.rs | 21 +- .../control/server/shared/clone_write/kv.rs | 17 ++ .../server/shared/ddl/neutral/rate_gate.rs | 2 + .../control/server/shared/returning/inject.rs | 248 +++++++++++++++++- .../shared/write_admission/lock_keys.rs | 4 + .../predicate/txn_buffering/classify.rs | 8 + .../control/server/wal_dispatch_kv/append.rs | 5 + .../wal_replication/decode/entry_kv.rs | 51 +++- .../src/control/wal_replication/decode/kv.rs | 22 +- .../wal_replication/encode/entry_kv.rs | 47 +++- .../src/control/wal_replication/encode/kv.rs | 22 +- .../wal_replication/types/replicated_write.rs | 24 ++ .../data/executor/handlers/kv/crud/delete.rs | 57 +++- .../src/data/executor/handlers/kv/crud/mod.rs | 2 +- .../data/executor/handlers/kv/crud/types.rs | 16 ++ .../src/data/executor/handlers/kv/dispatch.rs | 94 ++----- .../executor/handlers/kv/dispatch_scan.rs | 57 ++++ .../executor/handlers/kv/dispatch_transfer.rs | 96 +++++++ nodedb/src/data/executor/handlers/kv/field.rs | 34 ++- nodedb/src/data/executor/handlers/kv/mod.rs | 4 +- .../executor/handlers/kv/predicate/apply.rs | 37 ++- .../executor/handlers/kv/resolve/dispatch.rs | 50 +++- .../handlers/kv/resolve/predicate_ops.rs | 106 +++++--- .../executor/handlers/kv/resolve/write_ops.rs | 60 +++-- .../handlers/transaction/resolve/entry.rs | 2 + .../transaction/stage_write/stage_kv.rs | 4 + .../stage_write/stage_kv_transfer.rs | 5 + .../transaction/sub_plan_kv_writes.rs | 24 +- .../src/data/executor/wal_replay_kv_field.rs | 8 + .../inproc/cases/executor_tests/test_kv.rs | 2 + .../cases/executor_tests/test_kv_advanced.rs | 6 + .../test_tenant_isolation_kv_negative.rs | 2 + .../test_transaction_matrix_kv.rs | 2 + nodedb/tests/wire/cases/mod.rs | 1 + .../wire/cases/pgwire_returning_dml_kv.rs | 210 +++++++++++++++ 47 files changed, 1319 insertions(+), 183 deletions(-) create mode 100644 nodedb/src/data/executor/handlers/kv/dispatch_scan.rs create mode 100644 nodedb/src/data/executor/handlers/kv/dispatch_transfer.rs create mode 100644 nodedb/tests/wire/cases/pgwire_returning_dml_kv.rs diff --git a/nodedb-physical/src/physical_plan/kv/op.rs b/nodedb-physical/src/physical_plan/kv/op.rs index 317761917..56d03a231 100644 --- a/nodedb-physical/src/physical_plan/kv/op.rs +++ b/nodedb-physical/src/physical_plan/kv/op.rs @@ -132,6 +132,13 @@ pub enum KvOp { /// predicate is attached. Only `RlsWriteCheck::Predicate` makes the /// handler read the pre-image at all. rls_write_check: RlsWriteCheck, + /// When `Some`, return the STORED pre-image of every removed row + /// (row as `SELECT` showed it, `key` included). + #[serde(default)] + returning: Option, + /// See `Put::rls_filters`. + #[serde(default)] + rls_filters: Vec, }, /// Cursor-based scan with optional filter predicate. @@ -275,6 +282,13 @@ pub enum KvOp { /// Write policy evaluated against the merged body, which exists only /// after the stored row is read and updates applied. rls_write_check: RlsWriteCheck, + /// When `Some`, return the STORED post-image (merged row as `SELECT` + /// shows it, `key` included). Never the caller's submitted updates. + #[serde(default)] + returning: Option, + /// See `Put::rls_filters`. + #[serde(default)] + rls_filters: Vec, }, /// Truncate: delete ALL entries in a KV collection. @@ -497,6 +511,13 @@ pub enum KvOp { /// matched row's post-image once the assignments have been applied, or /// the reason no predicate is attached. rls_write_check: RlsWriteCheck, + /// When `Some`, return one row per matched key — the STORED + /// post-image of each, in scan order — projected per spec. + #[serde(default)] + returning: Option, + /// See `Put::rls_filters`. + #[serde(default)] + rls_filters: Vec, }, /// Delete every row matching `filters`. @@ -510,5 +531,12 @@ pub enum KvOp { /// Compiled row-level-security WRITE predicate, evaluated against the /// pre-image of every row this removes. rls_write_check: RlsWriteCheck, + /// When `Some`, return one row per removed key — the STORED + /// pre-image of each, in scan order — projected per spec. + #[serde(default)] + returning: Option, + /// See `Put::rls_filters`. + #[serde(default)] + rls_filters: Vec, }, } diff --git a/nodedb/src/bridge/admission_chokepoint.rs b/nodedb/src/bridge/admission_chokepoint.rs index 7189c954e..ab130da81 100644 --- a/nodedb/src/bridge/admission_chokepoint.rs +++ b/nodedb/src/bridge/admission_chokepoint.rs @@ -214,6 +214,8 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), keys: vec![b"k".to_vec()], rls_write_check: check, + returning: None, + rls_filters: Vec::new(), }) } diff --git a/nodedb/src/control/planner/calvin/write_class.rs b/nodedb/src/control/planner/calvin/write_class.rs index b26888129..4dec7a9da 100644 --- a/nodedb/src/control/planner/calvin/write_class.rs +++ b/nodedb/src/control/planner/calvin/write_class.rs @@ -434,6 +434,8 @@ mod tests { updates: vec![("field".to_owned(), b"v".to_vec())], surrogate: Surrogate::new(1), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: None, + rls_filters: Vec::new(), }); assert!(is_write_plan(&plan), "KvOp::FieldSet must be a write"); } diff --git a/nodedb/src/control/planner/rls_injection/kv.rs b/nodedb/src/control/planner/rls_injection/kv.rs index 69a7a10f4..5113a64e7 100644 --- a/nodedb/src/control/planner/rls_injection/kv.rs +++ b/nodedb/src/control/planner/rls_injection/kv.rs @@ -114,34 +114,47 @@ pub(super) fn inject_kv(ctx: &RlsCtx<'_>, op: &mut KvOp) -> crate::Result<()> { ctx.set_post_filters(collection, rls_filters) } + // Ship the predicate and gate `RETURNING` as a read: the removed row + // or field merge exists only where it's persisted, and the rows the + // clause hands back are bounded by the same read policy a `SELECT` + // by this principal is. Row set of the predicate forms is resolved by + // the Data Plane scan. Mirrors `ColumnarOp::{Update, Delete}`. KvOp::Delete { collection, rls_write_check, + rls_filters, .. } - | KvOp::Expire { + | KvOp::FieldSet { collection, rls_write_check, + rls_filters, .. } - | KvOp::Persist { + | KvOp::PredicateUpdate { collection, rls_write_check, + rls_filters, .. } - | KvOp::FieldSet { + | KvOp::PredicateDelete { collection, rls_write_check, + rls_filters, .. + } => { + ctx.set_write_check(collection, rls_write_check)?; + ctx.set_post_filters(collection, rls_filters) } - // Row set resolved by the Data Plane scan; images exist only where - // persisted. Mirrors `ColumnarOp::{Update, Delete}`. - | KvOp::PredicateUpdate { + + // Ship the predicate: the TTL body is the stored row, which exists + // only where it's persisted. + KvOp::Expire { collection, rls_write_check, .. } - | KvOp::PredicateDelete { + | KvOp::Persist { collection, rls_write_check, .. @@ -271,6 +284,8 @@ mod tests { ), keys: vec![b"k1".to_vec()], rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: None, + rls_filters: Vec::new(), }) } @@ -419,11 +434,73 @@ mod tests { updates: Vec::new(), surrogate: nodedb_types::Surrogate::ZERO, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: None, + rls_filters: Vec::new(), }); assert!(inject(&mut plan, &store).is_ok()); assert!(write_check(&plan).has_predicate()); } + /// `RETURNING` on a keyed or predicate UPDATE / DELETE ships rows back, + /// so a read policy lands in each op's post-filter slot the same way it + /// does for the KV insert ops. + #[test] + fn kv_update_and_delete_receive_the_read_policy_filter() { + let store = store_with_read_policy("sessions"); + let collection = || { + nodedb_types::QualifiedCollection::new(nodedb_types::DatabaseId::DEFAULT, "sessions") + }; + let ops = [ + KvOp::Delete { + collection: collection(), + keys: vec![b"k1".to_vec()], + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: None, + rls_filters: Vec::new(), + }, + KvOp::FieldSet { + collection: collection(), + key: b"k1".to_vec(), + updates: Vec::new(), + surrogate: nodedb_types::Surrogate::ZERO, + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: None, + rls_filters: Vec::new(), + }, + KvOp::PredicateUpdate { + collection: collection(), + filters: Vec::new(), + updates: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: None, + rls_filters: Vec::new(), + }, + KvOp::PredicateDelete { + collection: collection(), + filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: None, + rls_filters: Vec::new(), + }, + ]; + for op in ops { + let mut plan = PhysicalPlan::Kv(op); + assert!(inject(&mut plan, &store).is_ok()); + match &plan { + PhysicalPlan::Kv( + KvOp::Delete { rls_filters, .. } + | KvOp::FieldSet { rls_filters, .. } + | KvOp::PredicateUpdate { rls_filters, .. } + | KvOp::PredicateDelete { rls_filters, .. }, + ) => assert!( + !rls_filters.is_empty(), + "the read policy must gate RETURNING output for {plan:?}" + ), + other => panic!("plan shape changed: {other:?}"), + } + } + } + /// The incremented value is computed inside the engine, so the predicate /// rides along for the engine to decide the computed image against. #[test] diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs index 2161fb4d2..26bd958f3 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs @@ -46,6 +46,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_delete( collection: qualified_collection.clone(), filters: filter_bytes, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // Attached by `inject_returning_spec` after plan conversion. + returning: None, + rls_filters: Vec::new(), }), post_set_op: PostSetOp::None, txn_id: None, @@ -62,6 +65,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_delete( // Filled by the RLS injection pass, which runs after plan // conversion. rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // Attached by `inject_returning_spec` after plan conversion. + returning: None, + rls_filters: Vec::new(), }), post_set_op: PostSetOp::None, txn_id: None, diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs index f9c75d93f..aaf4b4f06 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs @@ -99,6 +99,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update( filters: filter_bytes, updates: literal_updates, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // Attached by `inject_returning_spec` after plan conversion. + returning: None, + rls_filters: Vec::new(), }), post_set_op: PostSetOp::None, txn_id: None, @@ -131,6 +134,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update( surrogate, // Filled by the RLS injection pass, after plan conversion. rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // Attached by `inject_returning_spec` after plan conversion. + returning: None, + rls_filters: Vec::new(), }), post_set_op: PostSetOp::None, txn_id: None, diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/document.rs b/nodedb/src/control/server/native/dispatch/plan_builder/document.rs index 1a6760a78..0c2b6b3a3 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/document.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/document.rs @@ -141,6 +141,9 @@ pub(crate) fn build_point_delete( keys: vec![doc_id.into_bytes()], // Filled by the RLS injection pass this dispatch path runs. rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // The native point-delete carries no RETURNING clause. + returning: None, + rls_filters: Vec::new(), })), Some(CollectionType::Columnar(ColumnarProfile::Timeseries { .. })) => { Err(crate::Error::BadRequest { diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs b/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs index 36c161d03..9a6d47c33 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs @@ -186,6 +186,9 @@ pub(crate) fn build_field_set( updates, surrogate, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // The native field-set carries no RETURNING clause. + returning: None, + rls_filters: Vec::new(), })) } diff --git a/nodedb/src/control/server/pgwire/handler/plan.rs b/nodedb/src/control/server/pgwire/handler/plan.rs index e0c87288f..84a8f34f4 100644 --- a/nodedb/src/control/server/pgwire/handler/plan.rs +++ b/nodedb/src/control/server/pgwire/handler/plan.rs @@ -276,6 +276,8 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "items"), keys: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: None, + rls_filters: Vec::new(), }); assert!(!is_calvin_foldable(&delete)); assert!(calvin_tag_for_plan(&delete).is_err()); diff --git a/nodedb/src/control/server/resp/handler_hash.rs b/nodedb/src/control/server/resp/handler_hash.rs index b95851434..8a478da41 100644 --- a/nodedb/src/control/server/resp/handler_hash.rs +++ b/nodedb/src/control/server/resp/handler_hash.rs @@ -149,6 +149,9 @@ pub(super) async fn handle_hset( surrogate, // Filled by the RLS injection pass `dispatch_kv_write` runs. rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // RESP has no RETURNING clause. + returning: None, + rls_filters: Vec::new(), }); match dispatch_kv_write(state, session, plan).await { diff --git a/nodedb/src/control/server/resp/handler_kv/strings.rs b/nodedb/src/control/server/resp/handler_kv/strings.rs index 586d6110e..e5d978117 100644 --- a/nodedb/src/control/server/resp/handler_kv/strings.rs +++ b/nodedb/src/control/server/resp/handler_kv/strings.rs @@ -162,6 +162,9 @@ pub(in crate::control::server::resp) async fn handle_del( keys, // Filled by the RLS injection pass `dispatch_kv_write` runs. rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // RESP has no RETURNING clause. + returning: None, + rls_filters: Vec::new(), }); match dispatch_kv_write(state, session, plan).await { diff --git a/nodedb/src/control/server/resp/handler_sorted.rs b/nodedb/src/control/server/resp/handler_sorted.rs index f2f9a520e..db4f1d748 100644 --- a/nodedb/src/control/server/resp/handler_sorted.rs +++ b/nodedb/src/control/server/resp/handler_sorted.rs @@ -113,6 +113,9 @@ pub(super) async fn handle_zrem( keys, // Filled by the RLS injection pass `dispatch_kv_write` runs. rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // RESP has no RETURNING clause. + returning: None, + rls_filters: Vec::new(), }); match dispatch_kv_write(state, session, plan).await { diff --git a/nodedb/src/control/server/response_shape/types/plan_kind.rs b/nodedb/src/control/server/response_shape/types/plan_kind.rs index 1eb811cfb..93476319a 100644 --- a/nodedb/src/control/server/response_shape/types/plan_kind.rs +++ b/nodedb/src/control/server/response_shape/types/plan_kind.rs @@ -187,12 +187,31 @@ pub fn describe_plan(plan: &PhysicalPlan) -> PlanKind { PhysicalPlan::Document(DocumentOp::Truncate { .. }) => DmlResult("TRUNCATE"), + // A KV update/delete with a projection returns real stored rows and + // must be decoded and redacted, exactly like the KV insert ops above. + PhysicalPlan::Kv( + KvOp::FieldSet { + returning: Some(_), .. + } + | KvOp::PredicateUpdate { + returning: Some(_), .. + } + | KvOp::Delete { + returning: Some(_), .. + } + | KvOp::PredicateDelete { + returning: Some(_), .. + }, + ) => PlanKind::ReturningRows, // KV delete/truncate count the keys removed — `Execution` would discard that. PhysicalPlan::Kv(KvOp::Delete { .. }) | PhysicalPlan::Kv(KvOp::PredicateDelete { .. }) => { DmlResult("DELETE") } // Reports `{"affected": n}` — `Execution` would discard that count. - PhysicalPlan::Kv(KvOp::PredicateUpdate { .. }) => DmlResult("UPDATE"), + // `FieldSet` is the keyed UPDATE, so it tags the same way. + PhysicalPlan::Kv(KvOp::FieldSet { .. }) | PhysicalPlan::Kv(KvOp::PredicateUpdate { .. }) => { + DmlResult("UPDATE") + } PhysicalPlan::Kv(KvOp::Truncate { .. }) => DmlResult("TRUNCATE"), PhysicalPlan::Document(DocumentOp::InsertSelect { .. }) => DmlResult("INSERT"), diff --git a/nodedb/src/control/server/shared/clone_write/kv.rs b/nodedb/src/control/server/shared/clone_write/kv.rs index 3b7b0dadc..d84093828 100644 --- a/nodedb/src/control/server/shared/clone_write/kv.rs +++ b/nodedb/src/control/server/shared/clone_write/kv.rs @@ -36,6 +36,8 @@ pub(super) async fn intercept_kv_clone_write( collection, keys, rls_write_check, + returning, + rls_filters, }) => { // Delete may have multiple keys; handle each. We serialize here // (one tombstone per key) and return Handled with synthetic OK. @@ -58,6 +60,18 @@ pub(super) async fn intercept_kv_clone_write( CloneStatus::Materialized => return Ok(CloneWriteOutcome::Passthrough), CloneStatus::Shadowed | CloneStatus::Materializing { .. } => {} } + // A row this delete hides only by tombstone lives in the source, + // so the clone has no stored pre-image to project for it. The + // reply below is a synthesized count, which the RETURNING renderer + // cannot decode as rows; refusing beats answering the wrong shape. + if returning.is_some() { + return Err(crate::Error::BadRequest { + detail: "RETURNING is not supported on a DELETE against a shadowed clone: \ + rows hidden by tombstone have no stored pre-image to project. \ + Materialize the clone first, or SELECT the rows before deleting." + .to_string(), + }); + } let emitter = ArcAuditEmitter(Arc::clone(&state.audit)); authorize_collection( @@ -150,6 +164,9 @@ pub(super) async fn intercept_kv_clone_write( // ungoverned one for exactly the keys that resolve to // real target rows. rls_write_check: rls_write_check.clone(), + // Same statement, same projection and read gate. + returning: returning.clone(), + rls_filters: rls_filters.clone(), }); let vshard_id = VShardId::from_collection_in_database(db_id, collection_qualified); let resp = dispatch_data_plane_raw(state, tenant_id, vshard_id, db_id, delete_plan) diff --git a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs index b904ceea9..21d5ed7c4 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs @@ -249,6 +249,8 @@ pub async fn rate_reset( keys: vec![rate_key.as_bytes().to_vec()], // Internal bookkeeping collection — see `rate_gate`'s INCR above. rls_write_check: nodedb_types::RlsWriteCheck::system_internal_collection(), + returning: None, + rls_filters: Vec::new(), }); match crate::control::server::dispatch_utils::dispatch_to_data_plane( diff --git a/nodedb/src/control/server/shared/returning/inject.rs b/nodedb/src/control/server/shared/returning/inject.rs index 3420c639d..f7ab70de3 100644 --- a/nodedb/src/control/server/shared/returning/inject.rs +++ b/nodedb/src/control/server/shared/returning/inject.rs @@ -11,7 +11,8 @@ //! rather than silently dropped. use nodedb_physical::physical_plan::{ - ColumnarOp, CrdtOp, DocumentOp, KvOp, QueryOp, ReturningSpec, TimeseriesOp, VectorOp, + ArrayOp, ClusterArrayOp, ClusterEventOp, ColumnarOp, CrdtOp, DocumentOp, GraphOp, KvOp, MetaOp, + QueryOp, ReturningSpec, SpatialOp, TextOp, TimeseriesOp, VectorOp, }; use nodedb_physical::physical_task::PhysicalTask; @@ -124,11 +125,11 @@ pub fn in_transaction_returning_unsupported() -> Error { /// Only `PointInsert`, `PointPut`, `BatchInsert`, `Upsert`, `PointUpdate`, /// `BulkUpdate`, `PointDelete`, `BulkDelete`, `UpdateFromJoin`, `Merge`, the KV /// `Insert` / `InsertIfAbsent` / `InsertOnConflictUpdate` / `Put` / `BatchPut` -/// ops, the columnar `Insert`, the timeseries `Ingest`, the vector -/// `DirectUpsert`, and the CRDT `DocUpsert` / `DocDelete` ops are affected. -/// Every other variant is left unchanged — an insert shape among them has -/// already been refused by [`refuse_unprojectable_insert_returning`], which -/// runs first. +/// / `FieldSet` / `PredicateUpdate` / `Delete` / `PredicateDelete` ops, the +/// columnar `Insert`, the timeseries `Ingest`, the vector `DirectUpsert`, and +/// the CRDT `DocUpsert` / `DocDelete` ops are affected. Every other variant is +/// left unchanged — an insert shape among them has already been refused by +/// [`refuse_unprojectable_insert_returning`], which runs first. pub fn inject_returning_spec(plan: &mut PhysicalPlan, spec: ReturningSpec) { match plan { PhysicalPlan::Document(DocumentOp::PointInsert { returning, .. }) => { @@ -149,6 +150,18 @@ pub fn inject_returning_spec(plan: &mut PhysicalPlan, spec: ReturningSpec) { PhysicalPlan::Kv(KvOp::BatchPut { returning, .. }) => { *returning = Some(spec); } + PhysicalPlan::Kv(KvOp::FieldSet { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Kv(KvOp::PredicateUpdate { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Kv(KvOp::Delete { returning, .. }) => { + *returning = Some(spec); + } + PhysicalPlan::Kv(KvOp::PredicateDelete { returning, .. }) => { + *returning = Some(spec); + } PhysicalPlan::Columnar(ColumnarOp::Insert { returning, .. }) => { *returning = Some(spec); } @@ -191,7 +204,228 @@ pub fn inject_returning_spec(plan: &mut PhysicalPlan, spec: ReturningSpec) { PhysicalPlan::Crdt(CrdtOp::DocDelete { returning, .. }) => { *returning = Some(spec); } - _ => {} + // Every variant with no `returning` slot, listed so a new variant + // forces a decision here instead of inheriting silence. + PhysicalPlan::Document( + DocumentOp::PointGet { .. } + | DocumentOp::Scan { .. } + | DocumentOp::RangeScan { .. } + | DocumentOp::Register { .. } + | DocumentOp::IndexLookup { .. } + | DocumentOp::IndexedFetch { .. } + | DocumentOp::DropIndex { .. } + | DocumentOp::BackfillIndex { .. } + | DocumentOp::Truncate { .. } + | DocumentOp::EstimateCount { .. } + | DocumentOp::InsertSelect { .. } + | DocumentOp::MaterializeScan { .. } + | DocumentOp::ApplyBalanceDelta { .. } + | DocumentOp::ResolveWrite(_) + | DocumentOp::ResolvedWrite { .. }, + ) + | PhysicalPlan::Kv( + KvOp::Get { .. } + | KvOp::Scan { .. } + | KvOp::Expire { .. } + | KvOp::Persist { .. } + | KvOp::GetTtl { .. } + | KvOp::BatchGet { .. } + | KvOp::RegisterIndex { .. } + | KvOp::DropIndex { .. } + | KvOp::FieldGet { .. } + | KvOp::Truncate { .. } + | KvOp::Incr { .. } + | KvOp::IncrFloat { .. } + | KvOp::Cas { .. } + | KvOp::GetSet { .. } + | KvOp::Transfer { .. } + | KvOp::TransferItem { .. } + | KvOp::RegisterSortedIndex { .. } + | KvOp::DropSortedIndex { .. } + | KvOp::SortedIndexRank { .. } + | KvOp::SortedIndexTopK { .. } + | KvOp::SortedIndexRange { .. } + | KvOp::SortedIndexCount { .. } + | KvOp::SortedIndexScore { .. } + | KvOp::MaterializeScan { .. } + | KvOp::ResolveWrite(_) + | KvOp::ResolvedWrite { .. }, + ) + | PhysicalPlan::Vector( + VectorOp::Search { .. } + | VectorOp::Insert { .. } + | VectorOp::BatchInsert { .. } + | VectorOp::MultiSearch { .. } + | VectorOp::Delete { .. } + | VectorOp::DeleteBySurrogate { .. } + | VectorOp::SetParams { .. } + | VectorOp::DropIndex { .. } + | VectorOp::QueryStats { .. } + | VectorOp::Seal { .. } + | VectorOp::CompactIndex { .. } + | VectorOp::Rebuild { .. } + | VectorOp::SparseInsert { .. } + | VectorOp::SparseSearch { .. } + | VectorOp::SparseDelete { .. } + | VectorOp::MultiVectorInsert { .. } + | VectorOp::MultiVectorDelete { .. } + | VectorOp::MultiVectorScoreSearch { .. }, + ) + | PhysicalPlan::Graph( + GraphOp::EdgePut { .. } + | GraphOp::EdgePutBatch { .. } + | GraphOp::EdgeDelete { .. } + | GraphOp::EdgeDeleteBatch { .. } + | GraphOp::Hop { .. } + | GraphOp::Neighbors { .. } + | GraphOp::NeighborsMulti { .. } + | GraphOp::Path { .. } + | GraphOp::Subgraph { .. } + | GraphOp::RagFusion { .. } + | GraphOp::Algo { .. } + | GraphOp::Match { .. } + | GraphOp::MatchContinuation { .. } + | GraphOp::MatchVarLenResume { .. } + | GraphOp::SetNodeLabels { .. } + | GraphOp::RemoveNodeLabels { .. } + | GraphOp::TemporalNeighbors { .. } + | GraphOp::TemporalAlgorithm { .. } + | GraphOp::Stats { .. } + | GraphOp::ResolveEdgeDelete(_) + | GraphOp::BspSuperstep(_) + | GraphOp::WccSuperstep(_), + ) + | PhysicalPlan::Text( + TextOp::Search { .. } + | TextOp::BM25ScoreScan { .. } + | TextOp::PhraseSearch { .. } + | TextOp::HybridSearch { .. } + | TextOp::FtsIndexDoc { .. } + | TextOp::FtsDeleteDoc { .. } + | TextOp::HybridSearchTriple { .. } + | TextOp::SetTextConfig { .. }, + ) + | PhysicalPlan::Columnar( + ColumnarOp::Scan { .. } + | ColumnarOp::Update { .. } + | ColumnarOp::Delete { .. } + | ColumnarOp::ResolvedUpdate { .. } + | ColumnarOp::ResolvedDelete { .. } + | ColumnarOp::ResolveDml { .. } + | ColumnarOp::MaterializeScan { .. }, + ) + | PhysicalPlan::Timeseries(TimeseriesOp::Scan { .. } | TimeseriesOp::ResolveIngest(_)) + | PhysicalPlan::Spatial( + SpatialOp::Insert { .. } | SpatialOp::Delete { .. } | SpatialOp::Scan { .. }, + ) + | PhysicalPlan::Crdt( + CrdtOp::Read { .. } + | CrdtOp::Apply { .. } + | CrdtOp::ApplyAuthenticated { .. } + | CrdtOp::ImportSnapshot { .. } + | CrdtOp::SetConstraints { .. } + | CrdtOp::DropConstraints { .. } + | CrdtOp::ReadConstraints { .. } + | CrdtOp::SetPolicy { .. } + | CrdtOp::GetPolicy { .. } + | CrdtOp::ReadAtVersion { .. } + | CrdtOp::GetVersionVector { .. } + | CrdtOp::ExportDelta { .. } + | CrdtOp::RestoreToVersion { .. } + | CrdtOp::CompactAtVersion { .. } + | CrdtOp::ListInsert { .. } + | CrdtOp::ListDelete { .. } + | CrdtOp::ListMove { .. } + | CrdtOp::PreviewApply { .. }, + ) + | PhysicalPlan::Query( + QueryOp::ProviderScan { .. } + | QueryOp::PostProcess { .. } + | QueryOp::SetOp { .. } + | QueryOp::Aggregate { .. } + | QueryOp::PartialAggregate { .. } + | QueryOp::PartialAggregateState { .. } + | QueryOp::HashJoin { .. } + | QueryOp::ShuffleJoinConsume { .. } + | QueryOp::ShuffleAggregateConsume { .. } + | QueryOp::NestedLoopJoin { .. } + | QueryOp::SortMergeJoin { .. } + | QueryOp::FacetCounts { .. } + | QueryOp::RecursiveScan { .. } + | QueryOp::RecursiveValue { .. } + | QueryOp::LateralTopK { .. } + | QueryOp::LateralLoop { .. } + | QueryOp::Exchange(_), + ) + | PhysicalPlan::Meta( + MetaOp::WalAppend { .. } + | MetaOp::Cancel { .. } + | MetaOp::TransactionBatch { .. } + | MetaOp::CreateSnapshot + | MetaOp::Compact + | MetaOp::Checkpoint + | MetaOp::RegisterContinuousAggregate { .. } + | MetaOp::UnregisterContinuousAggregate { .. } + | MetaOp::ListContinuousAggregates + | MetaOp::ConvertCollection { .. } + | MetaOp::CreateTenantSnapshot { .. } + | MetaOp::RestoreTenantSnapshot { .. } + | MetaOp::PurgeTenant { .. } + | MetaOp::UnregisterCollection { .. } + | MetaOp::UnregisterMaterializedView { .. } + | MetaOp::QueryCollectionSize { .. } + | MetaOp::EnforceTimeseriesRetention { .. } + | MetaOp::TemporalPurgeEdgeStore { .. } + | MetaOp::TemporalPurgeDocumentStrict { .. } + | MetaOp::TemporalPurgeColumnar { .. } + | MetaOp::TemporalPurgeCrdt { .. } + | MetaOp::TemporalPurgeArray { .. } + | MetaOp::AlterArray { .. } + | MetaOp::ApplyContinuousAggRetention + | MetaOp::QueryAggregateWatermark { .. } + | MetaOp::QueryLastValues { .. } + | MetaOp::QueryLastValue { .. } + | MetaOp::CalvinExecuteStatic { .. } + | MetaOp::CalvinExecutePassive { .. } + | MetaOp::CalvinExecuteActive { .. } + | MetaOp::RebuildIndex { .. } + | MetaOp::PutSynonymGroup { .. } + | MetaOp::DeleteSynonymGroup { .. } + | MetaOp::RenameCollection { .. } + | MetaOp::StageWrite { .. } + | MetaOp::DropTxnOverlay { .. } + | MetaOp::MarkSavepoint { .. } + | MetaOp::RollbackToSavepoint { .. } + | MetaOp::RecordCalvinWriteVersions { .. } + | MetaOp::CalvinFlush { .. } + | MetaOp::CalvinDrop { .. } + | MetaOp::ResolveTxn { .. } + | MetaOp::CalvinResolve { .. }, + ) + | PhysicalPlan::Array( + ArrayOp::OpenArray { .. } + | ArrayOp::Put { .. } + | ArrayOp::Delete { .. } + | ArrayOp::Slice { .. } + | ArrayOp::Project { .. } + | ArrayOp::Aggregate { .. } + | ArrayOp::Elementwise { .. } + | ArrayOp::Flush { .. } + | ArrayOp::Compact { .. } + | ArrayOp::SurrogateBitmapScan { .. } + | ArrayOp::DropArray { .. } + | ArrayOp::RestoreArrayDrop { .. } + | ArrayOp::PurgeArrayDrop { .. }, + ) + | PhysicalPlan::ClusterArray( + ClusterArrayOp::Slice { .. } + | ClusterArrayOp::Agg { .. } + | ClusterArrayOp::Put { .. } + | ClusterArrayOp::Delete { .. }, + ) + | PhysicalPlan::ClusterEvent( + ClusterEventOp::ConsumeStream { .. } | ClusterEventOp::PublishTopic { .. }, + ) => {} } } diff --git a/nodedb/src/control/server/shared/write_admission/lock_keys.rs b/nodedb/src/control/server/shared/write_admission/lock_keys.rs index 829b146ca..615f58023 100644 --- a/nodedb/src/control/server/shared/write_admission/lock_keys.rs +++ b/nodedb/src/control/server/shared/write_admission/lock_keys.rs @@ -309,6 +309,8 @@ mod tests { updates: vec![], surrogate: Surrogate::new(1), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: None, + rls_filters: Vec::new(), }), LockKey::Kv { collection: Arc::from("counters"), @@ -339,6 +341,8 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "counters"), keys: vec![b"k1".to_vec(), b"k2".to_vec()], rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: None, + rls_filters: Vec::new(), }), None ); diff --git a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs index 071834fe6..61de2e372 100644 --- a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs +++ b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs @@ -1167,17 +1167,23 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), keys: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), }), PhysicalPlan::Kv(KvOp::PredicateUpdate { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), filters: Vec::new(), updates: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), }), PhysicalPlan::Kv(KvOp::PredicateDelete { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), }), PhysicalPlan::Kv(KvOp::Scan { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), @@ -1230,6 +1236,8 @@ mod tests { updates: Vec::new(), surrogate: Surrogate::ZERO, rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), }), PhysicalPlan::Kv(KvOp::Incr { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), diff --git a/nodedb/src/control/server/wal_dispatch_kv/append.rs b/nodedb/src/control/server/wal_dispatch_kv/append.rs index 04b06e4bd..fd9c7b402 100644 --- a/nodedb/src/control/server/wal_dispatch_kv/append.rs +++ b/nodedb/src/control/server/wal_dispatch_kv/append.rs @@ -273,6 +273,9 @@ pub fn wal_append_kv_op( updates, // Per-request authorization input, not part of the durable image. rls_write_check: _, + // Per-request reply projection, not part of the durable image. + returning: _, + rls_filters: _, } => { let entry = encode_kv_predicate_update(collection.as_str(), filters, updates)?; Some(wal.append_put(tenant_id, vshard_id, database_id, &entry)?) @@ -281,6 +284,8 @@ pub fn wal_append_kv_op( collection, filters, rls_write_check: _, + returning: _, + rls_filters: _, } => { let entry = encode_kv_predicate_delete(collection.as_str(), filters)?; Some(wal.append_delete(tenant_id, vshard_id, database_id, &entry)?) diff --git a/nodedb/src/control/wal_replication/decode/entry_kv.rs b/nodedb/src/control/wal_replication/decode/entry_kv.rs index 05c62d877..0ed5d56c8 100644 --- a/nodedb/src/control/wal_replication/decode/entry_kv.rs +++ b/nodedb/src/control/wal_replication/decode/entry_kv.rs @@ -43,7 +43,19 @@ pub(super) fn decode_arm( }, )? } - ReplicatedWrite::KvDelete { collection, keys } => kv::delete(collection, keys), + ReplicatedWrite::KvDelete { + collection, + keys, + returning, + rls_filters, + } => kv::delete( + collection, + keys, + ReturningFields { + returning: decode_returning(returning)?, + rls_filters, + }, + ), ReplicatedWrite::KvInsert { collection, key, @@ -215,7 +227,19 @@ pub(super) fn decode_arm( key, updates, surrogate, - } => kv::field_set(ctx, collection, key, updates, *surrogate)?, + returning, + rls_filters, + } => kv::field_set( + ctx, + collection, + key, + updates, + *surrogate, + ReturningFields { + returning: decode_returning(returning)?, + rls_filters, + }, + )?, ReplicatedWrite::KvTransfer { collection, source_key, @@ -244,11 +268,30 @@ pub(super) fn decode_arm( collection, filters, updates, - } => kv::predicate_update(collection, filters, updates), + returning, + rls_filters, + } => kv::predicate_update( + collection, + filters, + updates, + ReturningFields { + returning: decode_returning(returning)?, + rls_filters, + }, + ), ReplicatedWrite::KvPredicateDelete { collection, filters, - } => kv::predicate_delete(collection, filters), + returning, + rls_filters, + } => kv::predicate_delete( + collection, + filters, + ReturningFields { + returning: decode_returning(returning)?, + rls_filters, + }, + ), ReplicatedWrite::KvTransferItem { source_collection, dest_collection, diff --git a/nodedb/src/control/wal_replication/decode/kv.rs b/nodedb/src/control/wal_replication/decode/kv.rs index ce60c464b..5e50e9541 100644 --- a/nodedb/src/control/wal_replication/decode/kv.rs +++ b/nodedb/src/control/wal_replication/decode/kv.rs @@ -44,11 +44,17 @@ pub(super) fn put( /// Every plan reconstructed here carries `RlsWriteCheck::already_decided_elsewhere()` /// — the writing identity isn't available on this node, so re-deciding at /// recovery time would make it non-deterministic. -pub(super) fn delete(collection: &str, keys: &[Vec]) -> PhysicalPlan { +pub(super) fn delete( + collection: &str, + keys: &[Vec], + returning: ReturningFields<'_>, +) -> PhysicalPlan { PhysicalPlan::Kv(KvOp::Delete { collection: nodedb_types::QualifiedCollection::from_stored(collection.to_owned()), keys: keys.to_vec(), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + returning: returning.returning, + rls_filters: returning.rls_filters.to_vec(), }) } @@ -59,21 +65,30 @@ pub(super) fn predicate_update( collection: &str, filters: &[u8], updates: &[(String, Vec)], + returning: ReturningFields<'_>, ) -> PhysicalPlan { PhysicalPlan::Kv(KvOp::PredicateUpdate { collection: nodedb_types::QualifiedCollection::from_stored(collection.to_owned()), filters: filters.to_vec(), updates: updates.to_vec(), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + returning: returning.returning, + rls_filters: returning.rls_filters.to_vec(), }) } /// Reconstruct a KV predicate `DELETE` plan — see [`predicate_update`]. -pub(super) fn predicate_delete(collection: &str, filters: &[u8]) -> PhysicalPlan { +pub(super) fn predicate_delete( + collection: &str, + filters: &[u8], + returning: ReturningFields<'_>, +) -> PhysicalPlan { PhysicalPlan::Kv(KvOp::PredicateDelete { collection: nodedb_types::QualifiedCollection::from_stored(collection.to_owned()), filters: filters.to_vec(), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + returning: returning.returning, + rls_filters: returning.rls_filters.to_vec(), }) } @@ -362,6 +377,7 @@ pub(super) fn field_set( key: &[u8], updates: &[(String, Vec)], surrogate: u32, + returning: ReturningFields<'_>, ) -> crate::Result { let carried = nodedb_types::Surrogate::new(surrogate); let surrogate = bind_or_lookup(ctx, collection, key, carried)?; @@ -371,6 +387,8 @@ pub(super) fn field_set( updates: updates.to_vec(), surrogate, rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + returning: returning.returning, + rls_filters: returning.rls_filters.to_vec(), })) } diff --git a/nodedb/src/control/wal_replication/encode/entry_kv.rs b/nodedb/src/control/wal_replication/encode/entry_kv.rs index 83e30d7ae..e43a2a5fd 100644 --- a/nodedb/src/control/wal_replication/encode/entry_kv.rs +++ b/nodedb/src/control/wal_replication/encode/entry_kv.rs @@ -43,7 +43,16 @@ pub(super) fn kv_write(op: &KvOp) -> crate::Result> { keys, // A follower has no writing identity; decode stamps `already_decided_elsewhere()`. rls_write_check: _, - } => kv::delete(collection.as_str(), keys), + returning, + rls_filters, + } => kv::delete( + collection.as_str(), + keys, + WireReturning { + returning, + rls_filters, + }, + ), KvOp::Insert { collection, key, @@ -222,7 +231,18 @@ pub(super) fn kv_write(op: &KvOp) -> crate::Result> { updates, surrogate, rls_write_check: _, - } => kv::field_set(collection.as_str(), key, updates, surrogate.as_u32()), + returning, + rls_filters, + } => kv::field_set( + collection.as_str(), + key, + updates, + surrogate.as_u32(), + WireReturning { + returning, + rls_filters, + }, + ), // A follower has no writing identity; decode stamps `already_decided_elsewhere()`. KvOp::Transfer { collection, @@ -277,17 +297,36 @@ pub(super) fn kv_write(op: &KvOp) -> crate::Result> { filters, updates, rls_write_check, + returning, + rls_filters, } => { refuse_governed_predicate_dml(collection.as_str(), rls_write_check)?; - kv::predicate_update(collection.as_str(), filters, updates) + kv::predicate_update( + collection.as_str(), + filters, + updates, + WireReturning { + returning, + rls_filters, + }, + ) } KvOp::PredicateDelete { collection, filters, rls_write_check, + returning, + rls_filters, } => { refuse_governed_predicate_dml(collection.as_str(), rls_write_check)?; - kv::predicate_delete(collection.as_str(), filters) + kv::predicate_delete( + collection.as_str(), + filters, + WireReturning { + returning, + rls_filters, + }, + ) } // Not a write — reads/scans/sorted-index queries. `ResolveWrite` mutates diff --git a/nodedb/src/control/wal_replication/encode/kv.rs b/nodedb/src/control/wal_replication/encode/kv.rs index 706520ebb..fe935db27 100644 --- a/nodedb/src/control/wal_replication/encode/kv.rs +++ b/nodedb/src/control/wal_replication/encode/kv.rs @@ -52,26 +52,41 @@ pub(super) fn predicate_update( collection: &str, filters: &[u8], updates: &[(String, Vec)], + returning: WireReturning<'_>, ) -> ReplicatedWrite { ReplicatedWrite::KvPredicateUpdate { collection: collection.to_owned(), filters: filters.to_vec(), updates: updates.to_vec(), + returning: encode_returning(returning.returning), + rls_filters: returning.rls_filters.to_vec(), } } /// Encode a KV predicate `DELETE` — see [`predicate_update`]. -pub(super) fn predicate_delete(collection: &str, filters: &[u8]) -> ReplicatedWrite { +pub(super) fn predicate_delete( + collection: &str, + filters: &[u8], + returning: WireReturning<'_>, +) -> ReplicatedWrite { ReplicatedWrite::KvPredicateDelete { collection: collection.to_owned(), filters: filters.to_vec(), + returning: encode_returning(returning.returning), + rls_filters: returning.rls_filters.to_vec(), } } -pub(super) fn delete(collection: &str, keys: &[Vec]) -> ReplicatedWrite { +pub(super) fn delete( + collection: &str, + keys: &[Vec], + returning: WireReturning<'_>, +) -> ReplicatedWrite { ReplicatedWrite::KvDelete { collection: collection.to_owned(), keys: keys.to_vec(), + returning: encode_returning(returning.returning), + rls_filters: returning.rls_filters.to_vec(), } } @@ -294,12 +309,15 @@ pub(super) fn field_set( key: &[u8], updates: &[(String, Vec)], surrogate: u32, + returning: WireReturning<'_>, ) -> ReplicatedWrite { ReplicatedWrite::KvFieldSet { collection: collection.to_owned(), key: key.to_vec(), updates: updates.to_vec(), surrogate, + returning: encode_returning(returning.returning), + rls_filters: returning.rls_filters.to_vec(), } } diff --git a/nodedb/src/control/wal_replication/types/replicated_write.rs b/nodedb/src/control/wal_replication/types/replicated_write.rs index 11d74c603..f3ef22700 100644 --- a/nodedb/src/control/wal_replication/types/replicated_write.rs +++ b/nodedb/src/control/wal_replication/types/replicated_write.rs @@ -388,6 +388,12 @@ pub enum ReplicatedWrite { KvDelete { collection: String, keys: Vec>, + /// See `ReplicatedWrite::PointPut::returning`. + #[serde(default)] + returning: Option>, + /// See `ReplicatedWrite::PointPut::rls_filters`. + #[serde(default)] + rls_filters: Vec, }, KvInsert { collection: String, @@ -522,6 +528,12 @@ pub enum ReplicatedWrite { key: Vec, updates: Vec<(String, Vec)>, surrogate: u32, + /// See `ReplicatedWrite::PointPut::returning`. + #[serde(default)] + returning: Option>, + /// See `ReplicatedWrite::PointPut::rls_filters`. + #[serde(default)] + rls_filters: Vec, }, KvTransfer { collection: String, @@ -788,6 +800,12 @@ pub enum ReplicatedWrite { /// Serialized `Vec`. Empty matches every row. filters: Vec, updates: Vec<(String, Vec)>, + /// See `ReplicatedWrite::PointPut::returning`. + #[serde(default)] + returning: Option>, + /// See `ReplicatedWrite::PointPut::rls_filters`. + #[serde(default)] + rls_filters: Vec, }, /// KV predicate `DELETE` on a collection with NO write policy — see @@ -796,6 +814,12 @@ pub enum ReplicatedWrite { collection: String, /// Serialized `Vec`. Empty matches every row. filters: Vec, + /// See `ReplicatedWrite::PointPut::returning`. + #[serde(default)] + returning: Option>, + /// See `ReplicatedWrite::PointPut::rls_filters`. + #[serde(default)] + rls_filters: Vec, }, /// Resolved form of a deferred document write (`PointUpdate`, diff --git a/nodedb/src/data/executor/handlers/kv/crud/delete.rs b/nodedb/src/data/executor/handlers/kv/crud/delete.rs index 6afd2cd08..81f29f645 100644 --- a/nodedb/src/data/executor/handlers/kv/crud/delete.rs +++ b/nodedb/src/data/executor/handlers/kv/crud/delete.rs @@ -4,8 +4,10 @@ use tracing::debug; +use super::types::KvDeleteParams; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::returning_rows::KvStoredRow; use crate::data::executor::response_codec; use crate::data::executor::task::ExecutionTask; use crate::engine::kv::current_ms; @@ -13,36 +15,51 @@ use crate::engine::kv::current_ms; impl CoreLoop { /// `rls_write_check` is the compiled RLS write policy. The row a delete /// removes is the image the policy decides, and a delete otherwise reads - /// nothing at all — so a non-empty check is what makes the pre-image be - /// read in the first place, and one rejected key fails the whole statement - /// before any key is removed. + /// nothing at all — so a non-empty check, or a `RETURNING` clause, is what + /// makes the pre-image be read in the first place, and one rejected key + /// fails the whole statement before any key is removed. pub(in crate::data::executor) fn execute_kv_delete( &mut self, task: &ExecutionTask, - did: u64, - tid: u64, - collection: &str, - keys: &[Vec], - rls_write_check: &nodedb_types::RlsWriteCheck, + params: KvDeleteParams<'_>, ) -> Response { + let KvDeleteParams { + did, + tid, + collection, + keys, + rls_write_check, + returning, + rls_filters, + } = params; debug!(core = self.core_id, %collection, count = keys.len(), "kv delete"); let now_ms = current_ms(); - if !matches!( + let gated = !matches!( rls_write_check.decision(), nodedb_types::WriteGateDecision::AdmitAll - ) { + ); + // Pre-images of the keys that exist, in `keys` order: the rows the + // write gate decides and the rows `RETURNING` projects. An absent key + // removes no row, so it has no image and counts as not-deleted. + let mut pre_images: Vec<(&[u8], Vec)> = Vec::with_capacity(keys.len()); + if gated || returning.is_some() { for key in keys { - // An absent key removes no row, so there is no image to decide; - // the delete simply counts it as not-deleted. let Some(body) = self.kv_engine.get(did, tid, collection, key, now_ms) else { continue; }; - if let Err(e) = - super::super::rls::admit_kv_row(rls_write_check, &body, key, tid, collection) + if gated + && let Err(e) = super::super::rls::admit_kv_row( + rls_write_check, + &body, + key, + tid, + collection, + ) { return self.response_error(task, e); } + pre_images.push((key.as_slice(), body)); } } @@ -66,6 +83,18 @@ impl CoreLoop { } } + if let Some(spec) = returning { + // The pre-images ARE the removed rows: the core loop runs ops + // serially, so nothing slips between the read and the delete. A + // delete that matched no key ships an EMPTY row set, never a + // count, so the RETURNING renderer decodes the shape it asked for. + let rows: Vec> = pre_images + .iter() + .map(|(key, body)| (*key, body.as_slice())) + .collect(); + return self.kv_stored_returning_response(task, spec, rls_filters, &rows); + } + match response_codec::encode_count("deleted", count) { Ok(payload) => self.response_with_payload(task, payload), Err(e) => self.response_error( diff --git a/nodedb/src/data/executor/handlers/kv/crud/mod.rs b/nodedb/src/data/executor/handlers/kv/crud/mod.rs index 1b5a7b817..0d9799356 100644 --- a/nodedb/src/data/executor/handlers/kv/crud/mod.rs +++ b/nodedb/src/data/executor/handlers/kv/crud/mod.rs @@ -9,5 +9,5 @@ mod write_basic; mod write_upsert; pub(in crate::data::executor) use types::{ - KvGetParams, KvInsertOnConflictUpdateParams, KvWriteParams, + KvDeleteParams, KvGetParams, KvInsertOnConflictUpdateParams, KvWriteParams, }; diff --git a/nodedb/src/data/executor/handlers/kv/crud/types.rs b/nodedb/src/data/executor/handlers/kv/crud/types.rs index c47e503f3..c5654f4ef 100644 --- a/nodedb/src/data/executor/handlers/kv/crud/types.rs +++ b/nodedb/src/data/executor/handlers/kv/crud/types.rs @@ -25,6 +25,22 @@ pub(in crate::data::executor) struct KvInsertOnConflictUpdateParams<'a> { pub rls_filters: &'a [u8], } +/// Parameters for a KV `DELETE` by primary key(s). +pub(in crate::data::executor) struct KvDeleteParams<'a> { + pub did: u64, + pub tid: u64, + pub collection: &'a str, + pub keys: &'a [Vec], + /// Compiled row-level-security WRITE predicate. The row a delete removes + /// is the image the policy decides. + pub rls_write_check: &'a nodedb_types::RlsWriteCheck, + /// When `Some`, project the STORED pre-image of every removed row per + /// spec instead of reporting a bare count. + pub returning: Option<&'a nodedb_physical::physical_plan::ReturningSpec>, + /// Compiled read policy bounding which of those rows may be shown back. + pub rls_filters: &'a [u8], +} + /// Parameters for a KV point `GET`. pub(in crate::data::executor) struct KvGetParams<'a> { pub did: u64, diff --git a/nodedb/src/data/executor/handlers/kv/dispatch.rs b/nodedb/src/data/executor/handlers/kv/dispatch.rs index 259d1d422..b32d4cc79 100644 --- a/nodedb/src/data/executor/handlers/kv/dispatch.rs +++ b/nodedb/src/data/executor/handlers/kv/dispatch.rs @@ -129,33 +129,21 @@ impl CoreLoop { collection, keys, rls_write_check, - } => self.execute_kv_delete(task, did, tid, collection.as_str(), keys, rls_write_check), - KvOp::Scan { - collection, - cursor, - count, - filters, - projection, - computed_columns, - match_pattern, - sort_keys, - surrogate_ceiling, - } => self.execute_kv_scan( + returning, + rls_filters, + } => self.execute_kv_delete( task, - super::scan::KvScanHandlerParams { + super::crud::KvDeleteParams { did, tid, collection: collection.as_str(), - cursor, - count: *count, - match_pattern: match_pattern.as_deref(), - filters, - projection, - computed_columns_bytes: computed_columns, - sort_keys, - surrogate_ceiling: *surrogate_ceiling, + keys, + rls_write_check, + returning: returning.as_ref(), + rls_filters, }, ), + KvOp::Scan { .. } => self.dispatch_kv_scan(task, did, tid, op), KvOp::Expire { collection, key, @@ -252,6 +240,8 @@ impl CoreLoop { updates, surrogate, rls_write_check, + returning, + rls_filters, } => self.execute_kv_field_set( super::atomic::KvAtomicCtx { task, @@ -262,7 +252,11 @@ impl CoreLoop { surrogate: *surrogate, rls_write_check, }, - updates, + super::field::KvFieldSetArgs { + updates, + returning: returning.as_ref(), + rls_filters, + }, ), KvOp::GetTtl { collection, key } => { self.execute_kv_get_ttl(task, did, tid, collection.as_str(), key) @@ -403,52 +397,8 @@ impl CoreLoop { index_name, primary_key, } => self.execute_kv_sorted_index_score(task, did, tid, index_name, primary_key), - KvOp::Transfer { - collection, - source_key, - dest_key, - field, - amount, - debit_surrogate, - credit_surrogate, - rls_write_check, - } => self.execute_kv_transfer( - task, - super::transfer::TransferParams { - did, - tid, - collection: collection.as_str(), - source_key, - dest_key, - field, - amount: *amount, - debit_surrogate: *debit_surrogate, - credit_surrogate: *credit_surrogate, - rls_write_check, - }, - ), - KvOp::TransferItem { - source_collection, - dest_collection, - item_key, - dest_key, - surrogate, - source_rls_write_check, - dest_rls_write_check, - } => self.execute_kv_transfer_item( - task, - super::transfer::TransferItemParams { - did, - tid, - source_collection: source_collection.as_str(), - dest_collection: dest_collection.as_str(), - item_key, - dest_key, - surrogate: *surrogate, - source_rls_write_check, - dest_rls_write_check, - }, - ), + KvOp::Transfer { .. } => self.dispatch_kv_transfer(task, did, tid, op), + KvOp::TransferItem { .. } => self.dispatch_kv_transfer_item(task, did, tid, op), KvOp::MaterializeScan { collection, cursor, @@ -466,6 +416,8 @@ impl CoreLoop { filters, updates, rls_write_check, + returning, + rls_filters, } => self.execute_kv_predicate_update( task, super::predicate::KvPredicateCtx { @@ -474,6 +426,8 @@ impl CoreLoop { collection: collection.as_str(), filters, rls_write_check, + returning: returning.as_ref(), + rls_filters, }, updates, ), @@ -481,6 +435,8 @@ impl CoreLoop { collection, filters, rls_write_check, + returning, + rls_filters, } => self.execute_kv_predicate_delete( task, super::predicate::KvPredicateCtx { @@ -489,6 +445,8 @@ impl CoreLoop { collection: collection.as_str(), filters, rls_write_check, + returning: returning.as_ref(), + rls_filters, }, ), KvOp::ResolveWrite(inner) => self.execute_kv_resolve_write(task, did, tid, inner), diff --git a/nodedb/src/data/executor/handlers/kv/dispatch_scan.rs b/nodedb/src/data/executor/handlers/kv/dispatch_scan.rs new file mode 100644 index 000000000..008ac3dff --- /dev/null +++ b/nodedb/src/data/executor/handlers/kv/dispatch_scan.rs @@ -0,0 +1,57 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! KV dispatch for the cursor `Scan` op. Split from `dispatch.rs` by op +//! family; fills the handler's params from the op's fields exactly as the +//! single match arm did. + +use crate::bridge::envelope::{ErrorCode, Response}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::task::ExecutionTask; +use nodedb_physical::physical_plan::KvOp; + +impl CoreLoop { + /// Run a `Scan`: cursor-paginated read with optional filters and sort. + pub(super) fn dispatch_kv_scan( + &mut self, + task: &ExecutionTask, + did: u64, + tid: u64, + op: &KvOp, + ) -> Response { + let KvOp::Scan { + collection, + cursor, + count, + filters, + projection, + computed_columns, + match_pattern, + sort_keys, + surrogate_ceiling, + } = op + else { + return self.response_error( + task, + ErrorCode::Internal { + detail: "dispatch_kv_scan: plan is not Scan".into(), + }, + ); + }; + self.execute_kv_scan( + task, + super::scan::KvScanHandlerParams { + did, + tid, + collection: collection.as_str(), + cursor, + count: *count, + match_pattern: match_pattern.as_deref(), + filters, + projection, + computed_columns_bytes: computed_columns, + sort_keys, + surrogate_ceiling: *surrogate_ceiling, + }, + ) + } +} diff --git a/nodedb/src/data/executor/handlers/kv/dispatch_transfer.rs b/nodedb/src/data/executor/handlers/kv/dispatch_transfer.rs new file mode 100644 index 000000000..e9b2451fc --- /dev/null +++ b/nodedb/src/data/executor/handlers/kv/dispatch_transfer.rs @@ -0,0 +1,96 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! KV dispatch for the two-key transfer ops, `Transfer` and `TransferItem`. +//! Split from `dispatch.rs` by op family; each fills the handler's params +//! from the op's fields exactly as the single match arm did. + +use crate::bridge::envelope::{ErrorCode, Response}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::task::ExecutionTask; +use nodedb_physical::physical_plan::KvOp; + +impl CoreLoop { + /// Run a `Transfer`: atomic fungible move of `field` between two keys. + pub(super) fn dispatch_kv_transfer( + &mut self, + task: &ExecutionTask, + did: u64, + tid: u64, + op: &KvOp, + ) -> Response { + let KvOp::Transfer { + collection, + source_key, + dest_key, + field, + amount, + debit_surrogate, + credit_surrogate, + rls_write_check, + } = op + else { + return self.response_error( + task, + ErrorCode::Internal { + detail: "dispatch_kv_transfer: plan is not Transfer".into(), + }, + ); + }; + self.execute_kv_transfer( + task, + super::transfer::TransferParams { + did, + tid, + collection: collection.as_str(), + source_key, + dest_key, + field, + amount: *amount, + debit_surrogate: *debit_surrogate, + credit_surrogate: *credit_surrogate, + rls_write_check, + }, + ) + } + + /// Run a `TransferItem`: atomic non-fungible move between collections. + pub(super) fn dispatch_kv_transfer_item( + &mut self, + task: &ExecutionTask, + did: u64, + tid: u64, + op: &KvOp, + ) -> Response { + let KvOp::TransferItem { + source_collection, + dest_collection, + item_key, + dest_key, + surrogate, + source_rls_write_check, + dest_rls_write_check, + } = op + else { + return self.response_error( + task, + ErrorCode::Internal { + detail: "dispatch_kv_transfer_item: plan is not TransferItem".into(), + }, + ); + }; + self.execute_kv_transfer_item( + task, + super::transfer::TransferItemParams { + did, + tid, + source_collection: source_collection.as_str(), + dest_collection: dest_collection.as_str(), + item_key, + dest_key, + surrogate: *surrogate, + source_rls_write_check, + dest_rls_write_check, + }, + ) + } +} diff --git a/nodedb/src/data/executor/handlers/kv/field.rs b/nodedb/src/data/executor/handlers/kv/field.rs index 197113caa..b87999074 100644 --- a/nodedb/src/data/executor/handlers/kv/field.rs +++ b/nodedb/src/data/executor/handlers/kv/field.rs @@ -23,6 +23,18 @@ pub(in crate::data::executor) struct KvFieldGetArgs<'a> { pub rls_filters: &'a [u8], } +/// Arguments for [`CoreLoop::execute_kv_field_set`] beyond the shared +/// single-key identity context. +pub(in crate::data::executor) struct KvFieldSetArgs<'a> { + /// Field name → new value (msgpack-encoded bytes). + pub updates: &'a [(String, Vec)], + /// When `Some`, project the STORED post-image (the merged row) per spec + /// instead of reporting the field-count payload. + pub returning: Option<&'a nodedb_physical::physical_plan::ReturningSpec>, + /// Compiled read policy bounding which of those rows may be shown back. + pub rls_filters: &'a [u8], +} + impl CoreLoop { pub(in crate::data::executor) fn execute_kv_field_get( &self, @@ -100,7 +112,7 @@ impl CoreLoop { pub(in crate::data::executor) fn execute_kv_field_set( &mut self, ctx: super::atomic::KvAtomicCtx<'_>, - updates: &[(String, Vec)], + args: KvFieldSetArgs<'_>, ) -> Response { let super::atomic::KvAtomicCtx { task, @@ -111,6 +123,11 @@ impl CoreLoop { surrogate, rls_write_check, } = ctx; + let KvFieldSetArgs { + updates, + returning, + rls_filters, + } = args; debug!(core = self.core_id, %collection, field_count = updates.len(), "kv field set"); let now_ms = current_ms(); @@ -146,8 +163,21 @@ impl CoreLoop { surrogate, }); self.note_kv_write_lsn(task, did, tid, collection, key); + if let Some(spec) = returning { + // `computed.new_value` IS the stored body: the merge is persisted + // verbatim, so projecting it is projecting the post-image. + return self.kv_stored_returning_response( + task, + spec, + rls_filters, + &[(key, computed.new_value.as_slice())], + ); + } + // `affected` is the row count the SQL `UPDATE` tag reads; + // `fields_added` is what RESP `HSET` reports. The merge always + // persists exactly one row. match response_codec::encode_json_as_msgpack( - &serde_json::json!({ "fields_added": computed.fields_added }), + &serde_json::json!({ "affected": 1, "fields_added": computed.fields_added }), ) { Ok(payload) => self.response_with_payload(task, payload), Err(e) => self.response_error( diff --git a/nodedb/src/data/executor/handlers/kv/mod.rs b/nodedb/src/data/executor/handlers/kv/mod.rs index ce6d7a2d2..c28ce718e 100644 --- a/nodedb/src/data/executor/handlers/kv/mod.rs +++ b/nodedb/src/data/executor/handlers/kv/mod.rs @@ -6,7 +6,9 @@ pub(in crate::data::executor) mod atomic; pub(in crate::data::executor) mod batch; pub(in crate::data::executor) mod crud; mod dispatch; -mod field; +mod dispatch_scan; +mod dispatch_transfer; +pub(in crate::data::executor) mod field; mod index; mod materialize_scan; pub(in crate::data::executor) mod predicate; diff --git a/nodedb/src/data/executor/handlers/kv/predicate/apply.rs b/nodedb/src/data/executor/handlers/kv/predicate/apply.rs index 47113afdf..2226e32b5 100644 --- a/nodedb/src/data/executor/handlers/kv/predicate/apply.rs +++ b/nodedb/src/data/executor/handlers/kv/predicate/apply.rs @@ -5,13 +5,16 @@ //! keyed path verbatim — re-deriving either here is how a predicate write //! would drift from the keyed one. +use nodedb_physical::physical_plan::ReturningSpec; use nodedb_types::{RlsWriteCheck, Surrogate}; use tracing::debug; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::kv::crud::KvDeleteParams; use crate::data::executor::handlers::kv::field_compute::merge_field_updates; use crate::data::executor::handlers::kv::rls::admit_kv_row; +use crate::data::executor::handlers::returning_rows::KvStoredRow; use crate::data::executor::response_codec; use crate::data::executor::task::ExecutionTask; use crate::engine::kv::{KvPutParams, current_ms}; @@ -23,6 +26,12 @@ pub(in crate::data::executor) struct KvPredicateCtx<'a> { pub collection: &'a str, pub filters: &'a [u8], pub rls_write_check: &'a RlsWriteCheck, + /// When `Some`, project one STORED row per matched key per spec instead + /// of reporting a bare count: the post-image for an update, the + /// pre-image for a delete. + pub returning: Option<&'a ReturningSpec>, + /// Compiled read policy bounding which of those rows may be shown back. + pub rls_filters: &'a [u8], } impl CoreLoop { @@ -41,6 +50,8 @@ impl CoreLoop { collection, filters, rls_write_check, + returning, + rls_filters, } = ctx; debug!(core = self.core_id, %collection, "kv predicate update"); let now_ms = current_ms(); @@ -93,6 +104,17 @@ impl CoreLoop { self.note_kv_write_lsn(task, did, tid, collection, key); } + if let Some(spec) = returning { + // Each `new_value` IS the stored body: the merge is persisted + // verbatim, so projecting it is projecting the post-image. Zero + // matches ships an EMPTY row set, never a count. + let rows: Vec> = writes + .iter() + .map(|(key, _old_body, new_value)| (key.as_slice(), new_value.as_slice())) + .collect(); + return self.kv_stored_returning_response(task, spec, rls_filters, &rows); + } + match response_codec::encode_count("affected", writes.len()) { Ok(payload) => self.response_with_payload(task, payload), Err(e) => self.response_error( @@ -118,6 +140,8 @@ impl CoreLoop { collection, filters, rls_write_check, + returning, + rls_filters, } = ctx; debug!(core = self.core_id, %collection, "kv predicate delete"); let now_ms = current_ms(); @@ -127,6 +151,17 @@ impl CoreLoop { Err(e) => return self.response_error(task, e), }; let keys: Vec> = matched.into_iter().map(|(key, _body)| key).collect(); - self.execute_kv_delete(task, did, tid, collection, &keys, rls_write_check) + self.execute_kv_delete( + task, + KvDeleteParams { + did, + tid, + collection, + keys: &keys, + rls_write_check, + returning, + rls_filters, + }, + ) } } diff --git a/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs b/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs index ba9e202bd..32b38744b 100644 --- a/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs +++ b/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs @@ -11,7 +11,9 @@ use tracing::debug; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::kv::atomic::KvAtomicCtx; -use crate::data::executor::handlers::kv::crud::KvInsertOnConflictUpdateParams; +use crate::data::executor::handlers::kv::crud::{KvDeleteParams, KvInsertOnConflictUpdateParams}; +use crate::data::executor::handlers::kv::field::KvFieldSetArgs; +use crate::data::executor::handlers::kv::predicate::KvPredicateCtx; use crate::data::executor::handlers::kv::transfer::{TransferItemParams, TransferParams}; use crate::data::executor::handlers::kv::ttl::KvTtlTarget; use crate::data::executor::response_codec; @@ -62,7 +64,17 @@ impl CoreLoop { collection, keys, rls_write_check, - } => self.resolve_kv_delete(did, tid, collection.as_str(), keys, rls_write_check), + returning, + rls_filters, + } => self.resolve_kv_delete(KvDeleteParams { + did, + tid, + collection: collection.as_str(), + keys, + rls_write_check, + returning: returning.as_ref(), + rls_filters, + }), KvOp::Expire { collection, key, @@ -96,6 +108,8 @@ impl CoreLoop { updates, surrogate, rls_write_check, + returning, + rls_filters, } => self.resolve_kv_field_set( KvAtomicCtx { task, @@ -106,7 +120,11 @@ impl CoreLoop { surrogate: *surrogate, rls_write_check, }, - updates, + KvFieldSetArgs { + updates, + returning: returning.as_ref(), + rls_filters, + }, ), KvOp::Incr { collection, @@ -231,25 +249,35 @@ impl CoreLoop { filters, updates, rls_write_check, + returning, + rls_filters, } => self.resolve_kv_predicate_update( - did, - tid, - collection.as_str(), - filters, + KvPredicateCtx { + did, + tid, + collection: collection.as_str(), + filters, + rls_write_check, + returning: returning.as_ref(), + rls_filters, + }, updates, - rls_write_check, ), KvOp::PredicateDelete { collection, filters, rls_write_check, - } => self.resolve_kv_predicate_delete( + returning, + rls_filters, + } => self.resolve_kv_predicate_delete(KvPredicateCtx { did, tid, - collection.as_str(), + collection: collection.as_str(), filters, rls_write_check, - ), + returning: returning.as_ref(), + rls_filters, + }), // Every other op is unwrapped by `resolver_for_plan` already. other => Err(ErrorCode::Internal { detail: format!( diff --git a/nodedb/src/data/executor/handlers/kv/resolve/predicate_ops.rs b/nodedb/src/data/executor/handlers/kv/resolve/predicate_ops.rs index 17300b1e3..74375f4d4 100644 --- a/nodedb/src/data/executor/handlers/kv/resolve/predicate_ops.rs +++ b/nodedb/src/data/executor/handlers/kv/resolve/predicate_ops.rs @@ -5,12 +5,14 @@ //! scan the live handler uses, computes each post-image with the same merge, //! and reports the mutations instead of applying them. -use nodedb_types::{RlsWriteCheck, Surrogate}; +use nodedb_types::Surrogate; use super::context::{ResolveResult, ResolvedPut, delete_mutation, put_mutation}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::kv::field_compute::merge_field_updates; +use crate::data::executor::handlers::kv::predicate::KvPredicateCtx; use crate::data::executor::handlers::kv::rls::admit_kv_row; +use crate::data::executor::handlers::returning_rows::kv_stored_rows_payload; use crate::data::executor::response_codec; use crate::engine::kv::current_ms; use nodedb_physical::physical_plan::KvResolveOutcome; @@ -18,38 +20,59 @@ use nodedb_physical::physical_plan::KvResolveOutcome; impl CoreLoop { /// Resolve a predicate `UPDATE`. Each matched row's stored body becomes /// its mutation's `precondition`, so a moved-past resolution applies nothing. + /// A `RETURNING` projects the post-images the mutations carry. pub(super) fn resolve_kv_predicate_update( &self, - did: u64, - tid: u64, - collection: &str, - filters: &[u8], + ctx: KvPredicateCtx<'_>, updates: &[(String, Vec)], - rls_write_check: &RlsWriteCheck, ) -> ResolveResult { + let KvPredicateCtx { + did, + tid, + collection, + filters, + rls_write_check, + returning, + rls_filters, + } = ctx; let now_ms = current_ms(); let matched = self.kv_predicate_matches(did, tid, collection, filters, now_ms)?; - let mut mutations = Vec::with_capacity(matched.len()); + let mut writes: Vec<(Vec, Vec, Vec)> = Vec::with_capacity(matched.len()); for (key, body) in matched { let computed = merge_field_updates(Some(body.as_slice()), updates)?; admit_kv_row(rls_write_check, &computed.new_value, &key, tid, collection)?; - mutations.push(put_mutation(ResolvedPut { - collection, - key: &key, - value: computed.new_value, - // `execute_kv_predicate_update` writes with `ttl_ms: 0`, the - // keyed field merge's behaviour. Preserved verbatim. - ttl_ms: 0, - expire_at_ms: 0, - // The row exists, so its bound surrogate must survive the - // merge — `ZERO` leaves it alone. - surrogate: Surrogate::ZERO, - precondition: Some(body), - })); + writes.push((key, body, computed.new_value)); } - let response_payload = response_codec::encode_count("affected", mutations.len())?; + let response_payload = match returning { + Some(spec) => { + let rows: Vec<(&[u8], &[u8])> = writes + .iter() + .map(|(key, _old_body, new_value)| (key.as_slice(), new_value.as_slice())) + .collect(); + kv_stored_rows_payload(spec, rls_filters, &rows)? + } + None => response_codec::encode_count("affected", writes.len())?, + }; + let mutations = writes + .into_iter() + .map(|(key, body, new_value)| { + put_mutation(ResolvedPut { + collection, + key: &key, + value: new_value, + // `execute_kv_predicate_update` writes with `ttl_ms: 0`, the + // keyed field merge's behaviour. Preserved verbatim. + ttl_ms: 0, + expire_at_ms: 0, + // The row exists, so its bound surrogate must survive the + // merge — `ZERO` leaves it alone. + surrogate: Surrogate::ZERO, + precondition: Some(body), + }) + }) + .collect(); Ok(KvResolveOutcome { mutations, response_payload, @@ -57,25 +80,38 @@ impl CoreLoop { } /// Resolve a predicate `DELETE`. Counts and replies exactly as - /// `resolve_kv_delete` does for a keyed one. - pub(super) fn resolve_kv_predicate_delete( - &self, - did: u64, - tid: u64, - collection: &str, - filters: &[u8], - rls_write_check: &RlsWriteCheck, - ) -> ResolveResult { + /// `resolve_kv_delete` does for a keyed one, pre-image `RETURNING` included. + pub(super) fn resolve_kv_predicate_delete(&self, ctx: KvPredicateCtx<'_>) -> ResolveResult { + let KvPredicateCtx { + did, + tid, + collection, + filters, + rls_write_check, + returning, + rls_filters, + } = ctx; let now_ms = current_ms(); let matched = self.kv_predicate_matches(did, tid, collection, filters, now_ms)?; - let mut mutations = Vec::with_capacity(matched.len()); - for (key, body) in matched { - admit_kv_row(rls_write_check, &body, &key, tid, collection)?; - mutations.push(delete_mutation(collection, &key, Some(body))); + for (key, body) in &matched { + admit_kv_row(rls_write_check, body, key, tid, collection)?; } - let response_payload = response_codec::encode_count("deleted", mutations.len())?; + let response_payload = match returning { + Some(spec) => { + let rows: Vec<(&[u8], &[u8])> = matched + .iter() + .map(|(key, body)| (key.as_slice(), body.as_slice())) + .collect(); + kv_stored_rows_payload(spec, rls_filters, &rows)? + } + None => response_codec::encode_count("deleted", matched.len())?, + }; + let mutations = matched + .into_iter() + .map(|(key, body)| delete_mutation(collection, &key, Some(body))) + .collect(); Ok(KvResolveOutcome { mutations, response_payload, diff --git a/nodedb/src/data/executor/handlers/kv/resolve/write_ops.rs b/nodedb/src/data/executor/handlers/kv/resolve/write_ops.rs index 724f29231..29167e3f7 100644 --- a/nodedb/src/data/executor/handlers/kv/resolve/write_ops.rs +++ b/nodedb/src/data/executor/handlers/kv/resolve/write_ops.rs @@ -14,7 +14,8 @@ use super::context::{ use crate::bridge::envelope::ErrorCode; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::kv::atomic::KvAtomicCtx; -use crate::data::executor::handlers::kv::crud::KvInsertOnConflictUpdateParams; +use crate::data::executor::handlers::kv::crud::{KvDeleteParams, KvInsertOnConflictUpdateParams}; +use crate::data::executor::handlers::kv::field::KvFieldSetArgs; use crate::data::executor::handlers::kv::rls::admit_kv_row; use crate::data::executor::handlers::kv::ttl::KvTtlTarget; use crate::data::executor::handlers::returning_rows::kv_stored_rows_payload; @@ -103,26 +104,42 @@ impl CoreLoop { } /// Resolve a KV `DELETE`. An absent key contributes no mutation and is - /// counted as not-deleted, same as `execute_kv_delete`. - pub(super) fn resolve_kv_delete( - &self, - did: u64, - tid: u64, - collection: &str, - keys: &[Vec], - rls_write_check: &nodedb_types::RlsWriteCheck, - ) -> ResolveResult { + /// counted as not-deleted, same as `execute_kv_delete`. A `RETURNING` + /// projects the pre-images the mutations carry, same as the live handler. + pub(super) fn resolve_kv_delete(&self, params: KvDeleteParams<'_>) -> ResolveResult { + let KvDeleteParams { + did, + tid, + collection, + keys, + rls_write_check, + returning, + rls_filters, + } = params; let now_ms = current_ms(); - let mut mutations = Vec::new(); + let mut pre_images: Vec<(&[u8], Vec)> = Vec::with_capacity(keys.len()); for key in keys { let Some(body) = self.kv_resolve_read(did, tid, collection, key, now_ms) else { continue; }; admit_kv_row(rls_write_check, &body, key, tid, collection)?; - mutations.push(delete_mutation(collection, key, Some(body))); + pre_images.push((key.as_slice(), body)); } - let response_payload = response_codec::encode_count("deleted", mutations.len())?; + let response_payload = match returning { + Some(spec) => { + let rows: Vec<(&[u8], &[u8])> = pre_images + .iter() + .map(|(key, body)| (*key, body.as_slice())) + .collect(); + kv_stored_rows_payload(spec, rls_filters, &rows)? + } + None => response_codec::encode_count("deleted", pre_images.len())?, + }; + let mutations = pre_images + .into_iter() + .map(|(key, body)| delete_mutation(collection, key, Some(body))) + .collect(); Ok(KvResolveOutcome { mutations, response_payload, @@ -197,7 +214,7 @@ impl CoreLoop { pub(super) fn resolve_kv_field_set( &self, ctx: KvAtomicCtx<'_>, - updates: &[(String, Vec)], + args: KvFieldSetArgs<'_>, ) -> ResolveResult { let KvAtomicCtx { did, @@ -208,6 +225,11 @@ impl CoreLoop { rls_write_check, .. } = ctx; + let KvFieldSetArgs { + updates, + returning, + rls_filters, + } = args; let now_ms = current_ms(); let current = self.kv_resolve_read(did, tid, collection, key, now_ms); let computed = crate::data::executor::handlers::kv::field_compute::merge_field_updates( @@ -216,9 +238,13 @@ impl CoreLoop { )?; admit_kv_row(rls_write_check, &computed.new_value, key, tid, collection)?; - let response_payload = response_codec::encode_json_as_msgpack( - &serde_json::json!({ "fields_added": computed.fields_added }), - )?; + let response_payload = match returning { + Some(spec) => kv_stored_rows_payload(spec, rls_filters, &[(key, &computed.new_value)])?, + // Same shape `execute_kv_field_set` reports. + None => response_codec::encode_json_as_msgpack( + &serde_json::json!({ "affected": 1, "fields_added": computed.fields_added }), + )?, + }; Ok(one( put_mutation(ResolvedPut { collection, diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs b/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs index f231c16f6..6d5780716 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs @@ -633,6 +633,8 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "kvc"), keys: vec![b"gone".to_vec()], rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), })], ); let redo = decode_redo(&resp); diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs index 74cf20569..709ac56e4 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs @@ -104,6 +104,10 @@ impl CoreLoop { collection, keys, rls_write_check, + // The Control Plane refuses `RETURNING` inside a transaction + // before the write is staged, so no row image is projected here. + returning: _, + rls_filters: _, } => self.stage_kv_delete(task, tid, txn_id, collection.as_str(), keys, rls_write_check), KvOp::BatchPut { .. } | KvOp::Incr { .. } diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_transfer.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_transfer.rs index e99c379ce..6c3e4ad0a 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_transfer.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_transfer.rs @@ -89,6 +89,10 @@ impl CoreLoop { // keys its own slots (see module doc) and ignores it. surrogate: _, rls_write_check, + // The Control Plane refuses `RETURNING` inside a transaction + // before the write is staged, so no row image is projected here. + returning: _, + rls_filters: _, } => { let ctx = self.kv_atomic_stage_ctx(task, tid, txn_id, collection.as_str(), key); self.stage_kv_field_set(&ctx, key, updates, rls_write_check) @@ -160,6 +164,7 @@ impl CoreLoop { return self.response_error(ctx.task, e); } match response_codec::encode_json_as_msgpack(&serde_json::json!({ + "affected": 1, "fields_added": computed.fields_added, })) { Ok(payload) => self.response_with_payload(ctx.task, payload), diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_writes.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_writes.rs index b3be21d0f..cd58095ee 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_writes.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_writes.rs @@ -168,6 +168,7 @@ impl CoreLoop { collection, keys, rls_write_check, + .. } => { let now_ms = current_ms(); // Capture prior values for all keys that exist before deleting. @@ -180,13 +181,19 @@ impl CoreLoop { Some((k.clone(), v)) }) .collect(); + // In-transaction writes never carry `RETURNING`: the Control + // Plane refuses the clause before staging (see `BatchPut`). let resp = self.execute_kv_delete( task, - did, - tid, - collection.as_str(), - keys, - rls_write_check, + crate::data::executor::handlers::kv::crud::KvDeleteParams { + did, + tid, + collection: collection.as_str(), + keys, + rls_write_check, + returning: None, + rls_filters: &[], + }, ); if resp.status == Status::Error { return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { @@ -249,6 +256,7 @@ impl CoreLoop { updates, surrogate, rls_write_check, + .. } => { let now_ms = current_ms(); let prior = self @@ -264,7 +272,11 @@ impl CoreLoop { surrogate: *surrogate, rls_write_check, }, - updates, + crate::data::executor::handlers::kv::field::KvFieldSetArgs { + updates, + returning: None, + rls_filters: &[], + }, ); if resp.status == Status::Error { return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { diff --git a/nodedb/src/data/executor/wal_replay_kv_field.rs b/nodedb/src/data/executor/wal_replay_kv_field.rs index 7a38531f9..93f449ec5 100644 --- a/nodedb/src/data/executor/wal_replay_kv_field.rs +++ b/nodedb/src/data/executor/wal_replay_kv_field.rs @@ -188,6 +188,8 @@ mod tests { updates: vec![("mana".to_string(), json_field_bytes(serde_json::json!(5)))], surrogate: Surrogate::new(1), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + returning: None, + rls_filters: Vec::new(), }); let records = append_via_autocommit(&[put_p1, field_set]); @@ -221,6 +223,8 @@ mod tests { updates: vec![("hp".to_string(), json_field_bytes(serde_json::json!(100)))], surrogate: Surrogate::new(3), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + returning: None, + rls_filters: Vec::new(), }); let records = append_via_autocommit(&[field_set]); @@ -262,6 +266,8 @@ mod tests { updates: vec![("hp".to_string(), json_field_bytes(serde_json::json!(1)))], surrogate: Surrogate::new(2), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + returning: None, + rls_filters: Vec::new(), }); let records = append_via_autocommit(&[put_scalar, field_set]); @@ -293,6 +299,8 @@ mod tests { updates: vec![("hp".to_string(), json_field_bytes(serde_json::json!(7)))], surrogate: Surrogate::new(99), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + returning: None, + rls_filters: Vec::new(), }); let records = append_via_autocommit(&[field_set]); diff --git a/nodedb/tests/inproc/cases/executor_tests/test_kv.rs b/nodedb/tests/inproc/cases/executor_tests/test_kv.rs index 6c4474402..0c2c68a7d 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_kv.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_kv.rs @@ -63,6 +63,8 @@ fn kv_put_get_delete() { ), keys: vec![b"key1".to_vec()], rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), }), ); let json = payload_value(&payload); diff --git a/nodedb/tests/inproc/cases/executor_tests/test_kv_advanced.rs b/nodedb/tests/inproc/cases/executor_tests/test_kv_advanced.rs index 5f0bb210e..9994cf03d 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_kv_advanced.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_kv_advanced.rs @@ -117,6 +117,8 @@ fn kv_protocol_command_sequence() { ), keys: vec![b"key1".to_vec()], rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), }), ); let json: serde_json::Value = payload_value(&payload); @@ -193,6 +195,8 @@ fn kv_protocol_command_sequence() { ), keys: vec![b"a".to_vec(), b"b".to_vec(), b"c".to_vec()], rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), }), ); let json: serde_json::Value = payload_value(&payload); @@ -458,6 +462,8 @@ fn kv_field_get_and_set() { )], surrogate: nodedb_types::Surrogate::ZERO, rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), }), ); diff --git a/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_kv_negative.rs b/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_kv_negative.rs index 1bdaa52ef..8241d0954 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_kv_negative.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_kv_negative.rs @@ -125,6 +125,8 @@ fn kv_cross_tenant_delete_does_not_affect_owner() { ), keys: vec![b"sess_xyz".to_vec()], rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), }), ); // Either Ok (deleted 0 rows from B's namespace) or NotFound — both correct. diff --git a/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_kv.rs b/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_kv.rs index 26ec6506f..69ae94ff7 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_kv.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_kv.rs @@ -190,6 +190,8 @@ fn rollback_matrix_kv_delete_then_doc_fail() { ), keys: vec![b"del_key".to_vec()], rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), }), doc_insert_conflict("docs"), ], diff --git a/nodedb/tests/wire/cases/mod.rs b/nodedb/tests/wire/cases/mod.rs index 92a757c3d..ca9df93ac 100644 --- a/nodedb/tests/wire/cases/mod.rs +++ b/nodedb/tests/wire/cases/mod.rs @@ -121,6 +121,7 @@ mod pgwire_int_width_range_enforcement; mod pgwire_orm_conformance; mod pgwire_reset_parameters; mod pgwire_returning_dml; +mod pgwire_returning_dml_kv; mod pgwire_returning_dml_strict; mod pgwire_show_dispatch; mod pgwire_tenant_scoping; diff --git a/nodedb/tests/wire/cases/pgwire_returning_dml_kv.rs b/nodedb/tests/wire/cases/pgwire_returning_dml_kv.rs new file mode 100644 index 000000000..510902aec --- /dev/null +++ b/nodedb/tests/wire/cases/pgwire_returning_dml_kv.rs @@ -0,0 +1,210 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `UPDATE ... RETURNING` and `DELETE ... RETURNING` on the key-value engine. +//! +//! An update returns the STORED post-image (the merged row as a `SELECT` +//! shows it), a delete returns the pre-image of every row it removed, and a +//! statement that matched nothing returns zero rows rather than a count. The +//! keyed form (`WHERE = ...`) and the predicate form (`WHERE = +//! ...`) go through different `KvOp`s and are pinned separately. + +use crate::harness::TestServer; + +/// Rows of `sql`, each row's columns joined by `|`. +async fn rows(server: &TestServer, sql: &str) -> Vec { + server + .query_rows(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")) + .into_iter() + .map(|r| r.join("|")) + .collect() +} + +/// A KV collection holding three rows: two owned by `alice`, one by `bob`. +async fn seed(server: &TestServer, collection: &str) { + server + .exec(&format!( + "CREATE COLLECTION {collection} \ + (id TEXT PRIMARY KEY, owner TEXT, n INT) \ + WITH (engine='kv')" + )) + .await + .unwrap_or_else(|e| panic!("create {collection}: {e}")); + for (id, owner, n) in [("r1", "alice", 1), ("r2", "alice", 2), ("r3", "bob", 3)] { + server + .exec(&format!( + "INSERT INTO {collection} (id, owner, n) VALUES ('{id}', '{owner}', {n})" + )) + .await + .unwrap_or_else(|e| panic!("seed {collection}/{id}: {e}")); + } +} + +/// A keyed UPDATE returns the post-image: the assigned value, the untouched +/// columns, and the key, exactly as a `SELECT` of that key reports them. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn kv_keyed_update_returning_ships_the_post_image() { + let server = TestServer::start().await; + seed(&server, "kv_ret_upd_key").await; + + let returned = server + .query_named_rows("UPDATE kv_ret_upd_key SET n = 99 WHERE id = 'r1' RETURNING *") + .await + .expect("KV keyed UPDATE RETURNING * must return the updated row"); + + assert_eq!(returned.len(), 1, "one updated row: {returned:?}"); + assert_eq!(returned[0].get("id").map(String::as_str), Some("r1")); + assert_eq!(returned[0].get("owner").map(String::as_str), Some("alice")); + assert_eq!(returned[0].get("n").map(String::as_str), Some("99")); + assert_eq!( + rows( + &server, + "SELECT id, owner, n FROM kv_ret_upd_key WHERE id = 'r1'" + ) + .await, + vec!["r1|alice|99".to_string()], + "RETURNING must report the row a SELECT sees" + ); +} + +/// A predicate UPDATE returns one post-image per matched row, and only the +/// matched rows. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn kv_predicate_update_returning_ships_every_matched_post_image() { + let server = TestServer::start().await; + seed(&server, "kv_ret_upd_pred").await; + + let mut returned = rows( + &server, + "UPDATE kv_ret_upd_pred SET n = 7 WHERE owner = 'alice' RETURNING id, owner, n", + ) + .await; + returned.sort(); + assert_eq!( + returned, + vec!["r1|alice|7".to_string(), "r2|alice|7".to_string()], + "both of alice's rows come back with the new value" + ); + assert_eq!( + rows(&server, "SELECT id, n FROM kv_ret_upd_pred ORDER BY id").await, + vec!["r1|7".to_string(), "r2|7".to_string(), "r3|3".to_string()], + "bob's row is untouched" + ); +} + +/// A keyed DELETE returns the pre-image of the row it removed, and the row is +/// gone afterwards. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn kv_keyed_delete_returning_ships_the_pre_image() { + let server = TestServer::start().await; + seed(&server, "kv_ret_del_key").await; + + let returned = server + .query_named_rows("DELETE FROM kv_ret_del_key WHERE id = 'r3' RETURNING *") + .await + .expect("KV keyed DELETE RETURNING * must return the removed row"); + + assert_eq!(returned.len(), 1, "one removed row: {returned:?}"); + assert_eq!(returned[0].get("id").map(String::as_str), Some("r3")); + assert_eq!(returned[0].get("owner").map(String::as_str), Some("bob")); + assert_eq!(returned[0].get("n").map(String::as_str), Some("3")); + assert_eq!( + rows(&server, "SELECT id FROM kv_ret_del_key ORDER BY id").await, + vec!["r1".to_string(), "r2".to_string()], + "the returned row must no longer be stored" + ); +} + +/// A predicate DELETE returns the pre-image of every row it removed. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn kv_predicate_delete_returning_ships_every_removed_pre_image() { + let server = TestServer::start().await; + seed(&server, "kv_ret_del_pred").await; + + let mut returned = rows( + &server, + "DELETE FROM kv_ret_del_pred WHERE owner = 'alice' RETURNING id, n", + ) + .await; + returned.sort(); + assert_eq!( + returned, + vec!["r1|1".to_string(), "r2|2".to_string()], + "both of alice's rows come back as they were stored" + ); + assert_eq!( + rows(&server, "SELECT id FROM kv_ret_del_pred").await, + vec!["r3".to_string()], + "only bob's row survives" + ); +} + +/// An UPDATE that matches no row returns zero rows: never a count payload, +/// which the RETURNING renderer cannot decode, and never an error. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn kv_update_returning_with_zero_matches_returns_zero_rows() { + let server = TestServer::start().await; + seed(&server, "kv_ret_upd_none").await; + + let returned = server + .query_rows("UPDATE kv_ret_upd_none SET n = 0 WHERE owner = 'nobody' RETURNING *") + .await + .expect("a zero-match UPDATE RETURNING must succeed"); + assert!(returned.is_empty(), "no row matched: {returned:?}"); + assert_eq!( + rows(&server, "SELECT id, n FROM kv_ret_upd_none ORDER BY id").await, + vec!["r1|1".to_string(), "r2|2".to_string(), "r3|3".to_string()], + "nothing was written" + ); +} + +/// A DELETE that matches no row returns zero rows, keyed and predicate alike. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn kv_delete_returning_with_zero_matches_returns_zero_rows() { + let server = TestServer::start().await; + seed(&server, "kv_ret_del_none").await; + + for sql in [ + "DELETE FROM kv_ret_del_none WHERE id = 'missing' RETURNING *", + "DELETE FROM kv_ret_del_none WHERE owner = 'nobody' RETURNING *", + ] { + let returned = server + .query_rows(sql) + .await + .unwrap_or_else(|e| panic!("a zero-match DELETE RETURNING must succeed: {sql}: {e}")); + assert!( + returned.is_empty(), + "no row matched for {sql}: {returned:?}" + ); + } + assert_eq!( + rows(&server, "SELECT id FROM kv_ret_del_none ORDER BY id").await, + vec!["r1".to_string(), "r2".to_string(), "r3".to_string()], + "nothing was removed" + ); +} + +/// A RETURNING expression evaluates against the removed row's pre-image, and +/// only the requested column is shipped. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn kv_delete_returning_expression_evaluates_on_the_pre_image() { + let server = TestServer::start().await; + seed(&server, "kv_ret_del_expr").await; + + let msgs = server + .client + .simple_query("DELETE FROM kv_ret_del_expr WHERE id = 'r2' RETURNING n * 2 AS d") + .await + .expect("an expression-only RETURNING must succeed"); + let row = msgs + .iter() + .find_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(row) => Some(row), + _ => None, + }) + .expect("one returned row"); + assert_eq!(row.len(), 1, "only the requested column is shipped"); + assert_eq!(row.columns()[0].name(), "d"); + assert_eq!(row.get(0), Some("4")); +} From 7f093f09266d82e27428439016645c6da5a96d02 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 18 Sep 2026 13:23:58 +0800 Subject: [PATCH 13/15] feat(kv): make SQL UPDATE a no-op against an absent key Add if_present to KvOp::FieldSet so SQL UPDATE and the RESP hash-set family diverge on a missing key: UPDATE reports UPDATE 0 (RETURNING yields no rows) instead of materializing a row from nothing, while HSET keeps creating it. Thread the flag through the physical plan, planner conversion, WAL encode/decode, replication encode/decode, the executor's live and transaction-staged field-set paths, and WAL replay, so every read of the flag agrees on the same decision. Switch HSET's per-field encoding from JSON to msgpack so RESP writes and the SQL UPDATE lowering feed the field-set merge the same wire format. --- nodedb-physical/src/physical_plan/kv/op.rs | 7 ++- .../src/control/planner/calvin/write_class.rs | 1 + .../src/control/planner/rls_injection/kv.rs | 2 + .../dml/update_delete/update.rs | 2 + .../server/native/dispatch/plan_builder/kv.rs | 2 + .../src/control/server/resp/handler_hash.rs | 11 ++-- .../shared/write_admission/lock_keys.rs | 1 + .../predicate/txn_buffering/classify.rs | 1 + .../control/server/wal_dispatch_kv/append.rs | 9 ++- .../control/server/wal_dispatch_kv/encode.rs | 27 ++++++--- .../wal_replication/decode/entry_kv.rs | 2 + .../src/control/wal_replication/decode/kv.rs | 2 + .../wal_replication/encode/entry_kv.rs | 2 + .../src/control/wal_replication/encode/kv.rs | 2 + .../wal_replication/types/replicated_write.rs | 3 + .../src/data/executor/handlers/kv/dispatch.rs | 2 + nodedb/src/data/executor/handlers/kv/field.rs | 23 ++++++++ .../executor/handlers/kv/resolve/dispatch.rs | 2 + .../executor/handlers/kv/resolve/write_ops.rs | 17 ++++++ .../stage_write/stage_kv_transfer.rs | 17 +++++- .../transaction/sub_plan_kv_writes.rs | 2 + .../src/data/executor/wal_replay_kv_field.rs | 26 +++++++-- .../cases/executor_tests/test_kv_advanced.rs | 2 + .../kv_field_transfer_surrogate_identity.rs | 19 +++++-- .../inproc/cases/resp_row_level_security.rs | 26 +++++++++ .../wire/cases/pgwire_returning_dml_kv.rs | 57 +++++++++++++++++++ 26 files changed, 240 insertions(+), 27 deletions(-) diff --git a/nodedb-physical/src/physical_plan/kv/op.rs b/nodedb-physical/src/physical_plan/kv/op.rs index 56d03a231..822cf4b32 100644 --- a/nodedb-physical/src/physical_plan/kv/op.rs +++ b/nodedb-physical/src/physical_plan/kv/op.rs @@ -274,11 +274,16 @@ pub enum KvOp { FieldSet { collection: QualifiedCollection, key: Vec, - /// Field name → new value (JSON-encoded bytes). + /// Field name → new value (msgpack-encoded bytes; empty = NULL). updates: Vec<(String, Vec)>, /// Content-addressed identity on `(collection, key)`, threaded to the /// write-back so a field merge keeps the row's original surrogate. surrogate: Surrogate, + /// Update only an existing row. `true` for SQL UPDATE (an absent key + /// is a no-op); `false` for the RESP hash-set family, which creates + /// the row. + #[serde(default)] + if_present: bool, /// Write policy evaluated against the merged body, which exists only /// after the stored row is read and updates applied. rls_write_check: RlsWriteCheck, diff --git a/nodedb/src/control/planner/calvin/write_class.rs b/nodedb/src/control/planner/calvin/write_class.rs index 4dec7a9da..031cd1fd9 100644 --- a/nodedb/src/control/planner/calvin/write_class.rs +++ b/nodedb/src/control/planner/calvin/write_class.rs @@ -433,6 +433,7 @@ mod tests { key: b"k".to_vec(), updates: vec![("field".to_owned(), b"v".to_vec())], surrogate: Surrogate::new(1), + if_present: false, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: None, rls_filters: Vec::new(), diff --git a/nodedb/src/control/planner/rls_injection/kv.rs b/nodedb/src/control/planner/rls_injection/kv.rs index 5113a64e7..1bc53d6d8 100644 --- a/nodedb/src/control/planner/rls_injection/kv.rs +++ b/nodedb/src/control/planner/rls_injection/kv.rs @@ -433,6 +433,7 @@ mod tests { key: b"k1".to_vec(), updates: Vec::new(), surrogate: nodedb_types::Surrogate::ZERO, + if_present: false, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: None, rls_filters: Vec::new(), @@ -463,6 +464,7 @@ mod tests { key: b"k1".to_vec(), updates: Vec::new(), surrogate: nodedb_types::Surrogate::ZERO, + if_present: false, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: None, rls_filters: Vec::new(), diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs index aaf4b4f06..3f31e4027 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs @@ -132,6 +132,8 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update( key: key_bytes, updates: field_updates, surrogate, + // SQL UPDATE: an absent key is `UPDATE 0`, never a create. + if_present: true, // Filled by the RLS injection pass, after plan conversion. rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), // Attached by `inject_returning_spec` after plan conversion. diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs b/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs index 9a6d47c33..34bf24e58 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs @@ -185,6 +185,8 @@ pub(crate) fn build_field_set( key, updates, surrogate, + // Native `field_set` is the RESP HSET family: an absent key is created. + if_present: false, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), // The native field-set carries no RETURNING clause. returning: None, diff --git a/nodedb/src/control/server/resp/handler_hash.rs b/nodedb/src/control/server/resp/handler_hash.rs index 8a478da41..d0bd154fb 100644 --- a/nodedb/src/control/server/resp/handler_hash.rs +++ b/nodedb/src/control/server/resp/handler_hash.rs @@ -2,8 +2,6 @@ //! Hash field RESP command handlers: HGET, HMGET, HSET, FLUSHDB. -use sonic_rs; - use crate::bridge::envelope::{PhysicalPlan, Status}; use crate::control::state::SharedState; use nodedb_physical::physical_plan::KvOp; @@ -117,13 +115,14 @@ pub(super) async fn handle_hset( } let key = cmd.args[0].clone(); + // Each value is msgpack-encoded: the field-set merge decodes every + // update as msgpack, the same encoding the SQL UPDATE lowering emits. let updates: Vec<(String, Vec)> = cmd.args[1..] .chunks(2) .filter_map(|pair| { let field = std::str::from_utf8(&pair[0]).ok()?.to_string(); - let json_value = - serde_json::Value::String(String::from_utf8_lossy(&pair[1]).into_owned()); - Some((field, sonic_rs::to_vec(&json_value).ok()?)) + let value = nodedb_types::Value::String(String::from_utf8_lossy(&pair[1]).into_owned()); + Some((field, nodedb_types::value_to_msgpack(&value).ok()?)) }) .collect(); @@ -147,6 +146,8 @@ pub(super) async fn handle_hset( key, updates, surrogate, + // RESP HSET semantics: an absent key is created. + if_present: false, // Filled by the RLS injection pass `dispatch_kv_write` runs. rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), // RESP has no RETURNING clause. diff --git a/nodedb/src/control/server/shared/write_admission/lock_keys.rs b/nodedb/src/control/server/shared/write_admission/lock_keys.rs index 615f58023..789e6d903 100644 --- a/nodedb/src/control/server/shared/write_admission/lock_keys.rs +++ b/nodedb/src/control/server/shared/write_admission/lock_keys.rs @@ -308,6 +308,7 @@ mod tests { key: b"k1".to_vec(), updates: vec![], surrogate: Surrogate::new(1), + if_present: false, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: None, rls_filters: Vec::new(), diff --git a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs index 61de2e372..6b83f2606 100644 --- a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs +++ b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs @@ -1235,6 +1235,7 @@ mod tests { key: Vec::new(), updates: Vec::new(), surrogate: Surrogate::ZERO, + if_present: false, rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, returning: None, rls_filters: Vec::new(), diff --git a/nodedb/src/control/server/wal_dispatch_kv/append.rs b/nodedb/src/control/server/wal_dispatch_kv/append.rs index fd9c7b402..273f95061 100644 --- a/nodedb/src/control/server/wal_dispatch_kv/append.rs +++ b/nodedb/src/control/server/wal_dispatch_kv/append.rs @@ -174,9 +174,16 @@ pub fn wal_append_kv_op( key, updates, surrogate, + if_present, .. } => { - let entry = encode_kv_field_set(collection.as_str(), key, updates, surrogate.as_u32())?; + let entry = encode_kv_field_set( + collection.as_str(), + key, + updates, + surrogate.as_u32(), + *if_present, + )?; Some(wal.append_put(tenant_id, vshard_id, database_id, &entry)?) } KvOp::Incr { diff --git a/nodedb/src/control/server/wal_dispatch_kv/encode.rs b/nodedb/src/control/server/wal_dispatch_kv/encode.rs index 14be361e7..93d6aa0c5 100644 --- a/nodedb/src/control/server/wal_dispatch_kv/encode.rs +++ b/nodedb/src/control/server/wal_dispatch_kv/encode.rs @@ -161,17 +161,27 @@ pub(crate) fn encode_kv_incr_float( } /// Encode a `kv_field_set` WAL payload: `("kv_field_set", collection, key, -/// updates, surrogate)`. Delta record: `updates` carries field-level inputs, -/// not the post-merge document — replay re-runs `merge_field_updates`. +/// updates, surrogate, if_present)`. Delta record: `updates` carries +/// field-level inputs, not the post-merge document — replay re-runs +/// `merge_field_updates`. `if_present` pins the SQL-UPDATE-vs-HSET no-op +/// rule so replay matches the live decision. pub(crate) fn encode_kv_field_set( collection: &str, key: &[u8], updates: &[(String, Vec)], surrogate: u32, + if_present: bool, ) -> crate::Result> { encode( "field set", - &("kv_field_set", collection, key, updates, surrogate), + &( + "kv_field_set", + collection, + key, + updates, + surrogate, + if_present, + ), ) } @@ -578,16 +588,19 @@ mod tests { ("score".to_string(), b"42".to_vec()), ("name".to_string(), b"alice".to_vec()), ]; - let entry = encode_kv_field_set("players", b"p1", &updates, 11).unwrap(); + let entry = encode_kv_field_set("players", b"p1", &updates, 11, true).unwrap(); - let (disc, collection, key, decoded_updates, surrogate) = - zerompk::from_msgpack::<(&str, String, Vec, Vec<(String, Vec)>, u32)>(&entry) - .unwrap(); + let (disc, collection, key, decoded_updates, surrogate, if_present) = + zerompk::from_msgpack::<(&str, String, Vec, Vec<(String, Vec)>, u32, bool)>( + &entry, + ) + .unwrap(); assert_eq!(disc, "kv_field_set"); assert_eq!(collection, "players"); assert_eq!(key, b"p1"); assert_eq!(decoded_updates, updates); assert_eq!(surrogate, 11); + assert!(if_present); } #[test] diff --git a/nodedb/src/control/wal_replication/decode/entry_kv.rs b/nodedb/src/control/wal_replication/decode/entry_kv.rs index 0ed5d56c8..8ca769fad 100644 --- a/nodedb/src/control/wal_replication/decode/entry_kv.rs +++ b/nodedb/src/control/wal_replication/decode/entry_kv.rs @@ -227,6 +227,7 @@ pub(super) fn decode_arm( key, updates, surrogate, + if_present, returning, rls_filters, } => kv::field_set( @@ -235,6 +236,7 @@ pub(super) fn decode_arm( key, updates, *surrogate, + *if_present, ReturningFields { returning: decode_returning(returning)?, rls_filters, diff --git a/nodedb/src/control/wal_replication/decode/kv.rs b/nodedb/src/control/wal_replication/decode/kv.rs index 5e50e9541..b3e227b5d 100644 --- a/nodedb/src/control/wal_replication/decode/kv.rs +++ b/nodedb/src/control/wal_replication/decode/kv.rs @@ -377,6 +377,7 @@ pub(super) fn field_set( key: &[u8], updates: &[(String, Vec)], surrogate: u32, + if_present: bool, returning: ReturningFields<'_>, ) -> crate::Result { let carried = nodedb_types::Surrogate::new(surrogate); @@ -386,6 +387,7 @@ pub(super) fn field_set( key: key.to_vec(), updates: updates.to_vec(), surrogate, + if_present, rls_write_check: RlsWriteCheck::already_decided_elsewhere(), returning: returning.returning, rls_filters: returning.rls_filters.to_vec(), diff --git a/nodedb/src/control/wal_replication/encode/entry_kv.rs b/nodedb/src/control/wal_replication/encode/entry_kv.rs index e43a2a5fd..2ed550a9e 100644 --- a/nodedb/src/control/wal_replication/encode/entry_kv.rs +++ b/nodedb/src/control/wal_replication/encode/entry_kv.rs @@ -230,6 +230,7 @@ pub(super) fn kv_write(op: &KvOp) -> crate::Result> { key, updates, surrogate, + if_present, rls_write_check: _, returning, rls_filters, @@ -238,6 +239,7 @@ pub(super) fn kv_write(op: &KvOp) -> crate::Result> { key, updates, surrogate.as_u32(), + *if_present, WireReturning { returning, rls_filters, diff --git a/nodedb/src/control/wal_replication/encode/kv.rs b/nodedb/src/control/wal_replication/encode/kv.rs index fe935db27..b5895bcb1 100644 --- a/nodedb/src/control/wal_replication/encode/kv.rs +++ b/nodedb/src/control/wal_replication/encode/kv.rs @@ -309,6 +309,7 @@ pub(super) fn field_set( key: &[u8], updates: &[(String, Vec)], surrogate: u32, + if_present: bool, returning: WireReturning<'_>, ) -> ReplicatedWrite { ReplicatedWrite::KvFieldSet { @@ -316,6 +317,7 @@ pub(super) fn field_set( key: key.to_vec(), updates: updates.to_vec(), surrogate, + if_present, returning: encode_returning(returning.returning), rls_filters: returning.rls_filters.to_vec(), } diff --git a/nodedb/src/control/wal_replication/types/replicated_write.rs b/nodedb/src/control/wal_replication/types/replicated_write.rs index f3ef22700..c12a43dfd 100644 --- a/nodedb/src/control/wal_replication/types/replicated_write.rs +++ b/nodedb/src/control/wal_replication/types/replicated_write.rs @@ -528,6 +528,9 @@ pub enum ReplicatedWrite { key: Vec, updates: Vec<(String, Vec)>, surrogate: u32, + /// See `KvOp::FieldSet::if_present`. + #[serde(default)] + if_present: bool, /// See `ReplicatedWrite::PointPut::returning`. #[serde(default)] returning: Option>, diff --git a/nodedb/src/data/executor/handlers/kv/dispatch.rs b/nodedb/src/data/executor/handlers/kv/dispatch.rs index b32d4cc79..8a772d31a 100644 --- a/nodedb/src/data/executor/handlers/kv/dispatch.rs +++ b/nodedb/src/data/executor/handlers/kv/dispatch.rs @@ -239,6 +239,7 @@ impl CoreLoop { key, updates, surrogate, + if_present, rls_write_check, returning, rls_filters, @@ -254,6 +255,7 @@ impl CoreLoop { }, super::field::KvFieldSetArgs { updates, + if_present: *if_present, returning: returning.as_ref(), rls_filters, }, diff --git a/nodedb/src/data/executor/handlers/kv/field.rs b/nodedb/src/data/executor/handlers/kv/field.rs index b87999074..8c93e8f2c 100644 --- a/nodedb/src/data/executor/handlers/kv/field.rs +++ b/nodedb/src/data/executor/handlers/kv/field.rs @@ -28,6 +28,8 @@ pub(in crate::data::executor) struct KvFieldGetArgs<'a> { pub(in crate::data::executor) struct KvFieldSetArgs<'a> { /// Field name → new value (msgpack-encoded bytes). pub updates: &'a [(String, Vec)], + /// See `KvOp::FieldSet::if_present`. + pub if_present: bool, /// When `Some`, project the STORED post-image (the merged row) per spec /// instead of reporting the field-count payload. pub returning: Option<&'a nodedb_physical::physical_plan::ReturningSpec>, @@ -125,6 +127,7 @@ impl CoreLoop { } = ctx; let KvFieldSetArgs { updates, + if_present, returning, rls_filters, } = args; @@ -134,6 +137,26 @@ impl CoreLoop { // Read current value. let current = self.kv_engine.get(did, tid, collection, key, now_ms); + // SQL UPDATE against an absent key is `UPDATE 0`, not a create: the + // RESP hash-set family (`if_present: false`) is the only caller that + // may create a row from nothing. + if if_present && current.is_none() { + if let Some(spec) = returning { + return self.kv_stored_returning_response(task, spec, rls_filters, &[]); + } + return match response_codec::encode_json_as_msgpack( + &serde_json::json!({ "affected": 0, "fields_added": 0 }), + ) { + Ok(payload) => self.response_with_payload(task, payload), + Err(e) => self.response_error( + task, + ErrorCode::Internal { + detail: e.to_string(), + }, + ), + }; + } + // Merge field updates via the pure computation shared with the // in-transaction staging handler (`stage_kv_transfer.rs`), so a // staged value and its COMMIT-time durable replay never diverge. diff --git a/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs b/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs index 32b38744b..f7e18121d 100644 --- a/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs +++ b/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs @@ -107,6 +107,7 @@ impl CoreLoop { key, updates, surrogate, + if_present, rls_write_check, returning, rls_filters, @@ -122,6 +123,7 @@ impl CoreLoop { }, KvFieldSetArgs { updates, + if_present: *if_present, returning: returning.as_ref(), rls_filters, }, diff --git a/nodedb/src/data/executor/handlers/kv/resolve/write_ops.rs b/nodedb/src/data/executor/handlers/kv/resolve/write_ops.rs index 29167e3f7..39de68e5b 100644 --- a/nodedb/src/data/executor/handlers/kv/resolve/write_ops.rs +++ b/nodedb/src/data/executor/handlers/kv/resolve/write_ops.rs @@ -227,11 +227,28 @@ impl CoreLoop { } = ctx; let KvFieldSetArgs { updates, + if_present, returning, rls_filters, } = args; let now_ms = current_ms(); let current = self.kv_resolve_read(did, tid, collection, key, now_ms); + + // SQL UPDATE against an absent key is `UPDATE 0`, not a create — + // mirrors `execute_kv_field_set`'s decision. + if if_present && current.is_none() { + let response_payload = match returning { + Some(spec) => kv_stored_rows_payload(spec, rls_filters, &[])?, + None => response_codec::encode_json_as_msgpack( + &serde_json::json!({ "affected": 0, "fields_added": 0 }), + )?, + }; + return Ok(KvResolveOutcome { + mutations: Vec::new(), + response_payload, + }); + } + let computed = crate::data::executor::handlers::kv::field_compute::merge_field_updates( current.as_deref(), updates, diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_transfer.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_transfer.rs index 6c3e4ad0a..ce8b6beed 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_transfer.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_transfer.rs @@ -88,6 +88,7 @@ impl CoreLoop { // Durable identity binds at COMMIT-time replay; the overlay // keys its own slots (see module doc) and ignores it. surrogate: _, + if_present, rls_write_check, // The Control Plane refuses `RETURNING` inside a transaction // before the write is staged, so no row image is projected here. @@ -95,7 +96,7 @@ impl CoreLoop { rls_filters: _, } => { let ctx = self.kv_atomic_stage_ctx(task, tid, txn_id, collection.as_str(), key); - self.stage_kv_field_set(&ctx, key, updates, rls_write_check) + self.stage_kv_field_set(&ctx, key, updates, *if_present, rls_write_check) } KvOp::Transfer { collection, @@ -150,9 +151,23 @@ impl CoreLoop { ctx: &StageCtx<'_>, key: &[u8], updates: &[(String, Vec)], + if_present: bool, rls_write_check: &nodedb_types::RlsWriteCheck, ) -> Response { let current = self.resolve_kv_current(ctx, key); + + // SQL UPDATE against an absent key is `UPDATE 0`, not a create — + // mirrors `execute_kv_field_set`'s decision. RETURNING is already + // refused inside a transaction, so no empty-projection branch here. + if if_present && current.is_none() { + return match response_codec::encode_json_as_msgpack(&serde_json::json!({ + "affected": 0, + "fields_added": 0, + })) { + Ok(payload) => self.response_with_payload(ctx.task, payload), + Err(e) => self.response_error(ctx.task, e), + }; + } let computed = match merge_field_updates(current.as_deref(), updates) { Ok(c) => c, Err(e) => return self.response_error(ctx.task, e), diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_writes.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_writes.rs index cd58095ee..f82662eb0 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_writes.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_writes.rs @@ -255,6 +255,7 @@ impl CoreLoop { key, updates, surrogate, + if_present, rls_write_check, .. } => { @@ -274,6 +275,7 @@ impl CoreLoop { }, crate::data::executor::handlers::kv::field::KvFieldSetArgs { updates, + if_present: *if_present, returning: None, rls_filters: &[], }, diff --git a/nodedb/src/data/executor/wal_replay_kv_field.rs b/nodedb/src/data/executor/wal_replay_kv_field.rs index 93f449ec5..02de3a38e 100644 --- a/nodedb/src/data/executor/wal_replay_kv_field.rs +++ b/nodedb/src/data/executor/wal_replay_kv_field.rs @@ -24,9 +24,10 @@ impl CoreLoop { /// /// Returns `None` when `payload` does not match the `kv_field_set` /// discriminator shape (caller tries the next candidate arm), otherwise - /// `Some(puts)` — `1` if the merge and write applied, `0` if tombstoned - /// or the merge failed (a re-encode error against the previously-durable - /// record, logged and skipped rather than fabricating a partial value). + /// `Some(puts)` — `1` if the merge and write applied, `0` if tombstoned, + /// a SQL-UPDATE record replayed against a still-absent key, or the merge + /// failed (a re-encode error against the previously-durable record, + /// logged and skipped rather than fabricating a partial value). pub(super) fn try_replay_kv_field_set( &mut self, payload: &[u8], @@ -36,9 +37,11 @@ impl CoreLoop { record_lsn: u64, tombstones: &nodedb_wal::TombstoneSet, ) -> Option { - let (disc, collection, key, updates, surrogate) = - zerompk::from_msgpack::<(&str, String, Vec, Vec<(String, Vec)>, u32)>(payload) - .ok()?; + let (disc, collection, key, updates, surrogate, if_present) = + zerompk::from_msgpack::<(&str, String, Vec, Vec<(String, Vec)>, u32, bool)>( + payload, + ) + .ok()?; if disc != "kv_field_set" { return None; } @@ -50,6 +53,12 @@ impl CoreLoop { let current = self .kv_engine .get(database_id, tenant_id, &collection, &key, now_ms); + // Mirrors the live decision in `execute_kv_field_set`: a SQL-UPDATE + // record replayed against a key that is still absent is a no-op, not + // a create. + if if_present && current.is_none() { + return Some(0); + } let computed = match merge_field_updates(current.as_deref(), &updates) { Ok(c) => c, Err(e) => { @@ -187,6 +196,7 @@ mod tests { key: b"p1".to_vec(), updates: vec![("mana".to_string(), json_field_bytes(serde_json::json!(5)))], surrogate: Surrogate::new(1), + if_present: false, rls_write_check: RlsWriteCheck::already_decided_elsewhere(), returning: None, rls_filters: Vec::new(), @@ -222,6 +232,8 @@ mod tests { key: b"fresh".to_vec(), updates: vec![("hp".to_string(), json_field_bytes(serde_json::json!(100)))], surrogate: Surrogate::new(3), + // RESP HSET semantics: proves the create-on-absent path this test names. + if_present: false, rls_write_check: RlsWriteCheck::already_decided_elsewhere(), returning: None, rls_filters: Vec::new(), @@ -265,6 +277,7 @@ mod tests { key: b"p2".to_vec(), updates: vec![("hp".to_string(), json_field_bytes(serde_json::json!(1)))], surrogate: Surrogate::new(2), + if_present: false, rls_write_check: RlsWriteCheck::already_decided_elsewhere(), returning: None, rls_filters: Vec::new(), @@ -298,6 +311,7 @@ mod tests { key: b"p3".to_vec(), updates: vec![("hp".to_string(), json_field_bytes(serde_json::json!(7)))], surrogate: Surrogate::new(99), + if_present: false, rls_write_check: RlsWriteCheck::already_decided_elsewhere(), returning: None, rls_filters: Vec::new(), diff --git a/nodedb/tests/inproc/cases/executor_tests/test_kv_advanced.rs b/nodedb/tests/inproc/cases/executor_tests/test_kv_advanced.rs index 9994cf03d..91cb9af5e 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_kv_advanced.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_kv_advanced.rs @@ -461,6 +461,8 @@ fn kv_field_get_and_set() { .unwrap(), )], surrogate: nodedb_types::Surrogate::ZERO, + // HSET semantics: an absent key is created. + if_present: false, rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, returning: None, rls_filters: Vec::new(), diff --git a/nodedb/tests/inproc/cases/kv_field_transfer_surrogate_identity.rs b/nodedb/tests/inproc/cases/kv_field_transfer_surrogate_identity.rs index 3462a4f5e..b95614ded 100644 --- a/nodedb/tests/inproc/cases/kv_field_transfer_surrogate_identity.rs +++ b/nodedb/tests/inproc/cases/kv_field_transfer_surrogate_identity.rs @@ -26,7 +26,7 @@ use nodedb_test_support::pgwire_harness::TestServer; use nodedb_types::{DatabaseId, Surrogate, TenantId}; #[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn kv_field_set_on_fresh_key_persists_a_real_surrogate() { +async fn kv_field_set_on_existing_key_persists_a_real_surrogate() { let server = TestServer::start().await; server @@ -34,14 +34,21 @@ async fn kv_field_set_on_fresh_key_persists_a_real_surrogate() { .await .expect("create kv collection"); + // SQL `UPDATE` only ever touches an existing row (an absent key is + // `UPDATE 0`, never a create), so seed the row first. + server + .exec("INSERT INTO cf (key, n) VALUES ('fresh', 0)") + .await + .expect("seed key"); + // A KV `UPDATE` with a literal RHS on a PK-equality WHERE lowers to a - // `FieldSet` (HSET-style read-modify-write). Run it on a key that was NEVER - // inserted: the field merge materializes the row, so its cross-engine - // identity must be allocated + persisted on this path. + // `FieldSet` (HSET-style read-modify-write) gated to the existing row. + // Its write-back must carry the row's real persisted surrogate, not + // `Surrogate::ZERO`. server .exec("UPDATE cf SET n = 5 WHERE key = 'fresh'") .await - .expect("kv field-set on fresh key"); + .expect("kv field-set on existing key"); let catalog = server.shared.credentials.catalog(); let bindings = catalog @@ -52,7 +59,7 @@ async fn kv_field_set_on_fresh_key_persists_a_real_surrogate() { bindings.len(), 1, "the field-atomic op must persist exactly one PK->surrogate binding for \ - the fresh key (the bug allocated none and stored Surrogate::ZERO), \ + the key (the bug allocated none and stored Surrogate::ZERO), \ got: {bindings:?}" ); assert!( diff --git a/nodedb/tests/inproc/cases/resp_row_level_security.rs b/nodedb/tests/inproc/cases/resp_row_level_security.rs index 5112328e0..fc497fbb7 100644 --- a/nodedb/tests/inproc/cases/resp_row_level_security.rs +++ b/nodedb/tests/inproc/cases/resp_row_level_security.rs @@ -363,6 +363,32 @@ async fn hget_reports_policy_excluded_rows_as_absent() { ); } +/// `HSET` on a key that was never inserted creates the row: the RESP +/// hash-set family is unaffected by SQL `UPDATE`'s absent-key no-op rule. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn hset_on_absent_key_creates_the_row() { + let server = TestServer::start().await; + seed(&server, "resp_hset_create", "resp_hset_user").await; + let addr = start_resp_listener(&server).await; + + let mut client = session(addr, "resp_hset_user", "resp_hset_create").await; + let reply = client.cmd(&["HSET", "fresh", "val", "created"]).await; + assert!( + !reply.is_error(), + "HSET on an absent key must succeed: {reply:?}" + ); + + let rows = server + .query_rows("SELECT val FROM resp_hset_create WHERE id = 'fresh'") + .await + .unwrap_or_else(|e| panic!("read back created row: {e}")); + assert_eq!( + rows, + vec![vec!["created".to_string()]], + "HSET on an absent key must create the row" + ); +} + /// `SELECT` writes client input straight into the session's collection slot. /// The internal catalog collection must be refused where the client names it. #[tokio::test(flavor = "multi_thread", worker_threads = 4)] diff --git a/nodedb/tests/wire/cases/pgwire_returning_dml_kv.rs b/nodedb/tests/wire/cases/pgwire_returning_dml_kv.rs index 510902aec..ee2f0f2a2 100644 --- a/nodedb/tests/wire/cases/pgwire_returning_dml_kv.rs +++ b/nodedb/tests/wire/cases/pgwire_returning_dml_kv.rs @@ -68,6 +68,63 @@ async fn kv_keyed_update_returning_ships_the_post_image() { ); } +/// A keyed UPDATE against a key that was never inserted reports `UPDATE 0` +/// and creates no row — RESP `HSET` semantics (create on absent) do not +/// leak into SQL `UPDATE`. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn keyed_update_on_absent_key_affects_zero_rows() { + let server = TestServer::start().await; + seed(&server, "kv_upd_absent").await; + + let messages = server + .client + .simple_query("UPDATE kv_upd_absent SET n = 5 WHERE id = 'missing'") + .await + .expect("a keyed UPDATE on an absent key must succeed"); + let mut count = None; + for message in messages { + if let tokio_postgres::SimpleQueryMessage::CommandComplete(n) = message { + count = Some(n); + } + } + assert_eq!( + count, + Some(0), + "an absent keyed UPDATE must report UPDATE 0, not create the row" + ); + assert_eq!( + rows(&server, "SELECT id FROM kv_upd_absent WHERE id = 'missing'").await, + Vec::::new(), + "the absent key must not be created" + ); +} + +/// A keyed `UPDATE ... RETURNING` against a key that was never inserted +/// returns zero rows, never a post-image materialized from nothing. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn keyed_update_returning_on_absent_key_returns_no_rows() { + let server = TestServer::start().await; + seed(&server, "kv_upd_absent_ret").await; + + let returned = server + .query_rows("UPDATE kv_upd_absent_ret SET n = 5 WHERE id = 'missing' RETURNING *") + .await + .expect("a keyed UPDATE RETURNING on an absent key must succeed"); + assert!( + returned.is_empty(), + "an absent keyed UPDATE RETURNING must return no rows: {returned:?}" + ); + assert_eq!( + rows( + &server, + "SELECT id FROM kv_upd_absent_ret WHERE id = 'missing'" + ) + .await, + Vec::::new(), + "the absent key must not be created" + ); +} + /// A predicate UPDATE returns one post-image per matched row, and only the /// matched rows. #[tokio::test(flavor = "multi_thread", worker_threads = 4)] From 7193e2274b65a4215b02dd28607d714a002a2a17 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 18 Sep 2026 13:24:33 +0800 Subject: [PATCH 14/15] fix(resp): surface rejected KV writes as errors, not fallback replies HSET, SET, DEL, and GETSET matched on Status::Ok (or on Ok(_) alone) and fell back to a zero-ish reply (0 fields added, OK, 0 deleted, nil) whenever the dispatch payload didn't decode as expected, masking a rejected write behind a misleading success reply. Route every kv write dispatch through payload_or_typed_error so a rejected write returns the RESP error it actually is. --- .../src/control/server/resp/handler_hash.rs | 13 ++++++--- .../control/server/resp/handler_kv/strings.rs | 28 +++++++++++++------ 2 files changed, 29 insertions(+), 12 deletions(-) diff --git a/nodedb/src/control/server/resp/handler_hash.rs b/nodedb/src/control/server/resp/handler_hash.rs index d0bd154fb..828c256e1 100644 --- a/nodedb/src/control/server/resp/handler_hash.rs +++ b/nodedb/src/control/server/resp/handler_hash.rs @@ -3,6 +3,7 @@ //! Hash field RESP command handlers: HGET, HMGET, HSET, FLUSHDB. use crate::bridge::envelope::{PhysicalPlan, Status}; +use crate::control::server::shared::response_payload::payload_or_typed_error; use crate::control::state::SharedState; use nodedb_physical::physical_plan::KvOp; use nodedb_types::QualifiedCollection; @@ -155,12 +156,16 @@ pub(super) async fn handle_hset( rls_filters: Vec::new(), }); - match dispatch_kv_write(state, session, plan).await { - Ok(resp) if resp.status == Status::Ok => { - let added = payload_field_i64(&resp.payload, "fields_added").unwrap_or(0); + // A rejected write surfaces as the error it is, never as `0` fields + // added. + match dispatch_kv_write(state, session, plan) + .await + .and_then(payload_or_typed_error) + { + Ok(payload) => { + let added = payload_field_i64(&payload, "fields_added").unwrap_or(0); RespValue::integer(added) } - Ok(_) => RespValue::integer(0), Err(e) => RespValue::from_error(&e), } } diff --git a/nodedb/src/control/server/resp/handler_kv/strings.rs b/nodedb/src/control/server/resp/handler_kv/strings.rs index e5d978117..bfa7aa09b 100644 --- a/nodedb/src/control/server/resp/handler_kv/strings.rs +++ b/nodedb/src/control/server/resp/handler_kv/strings.rs @@ -3,6 +3,7 @@ //! Single-key string commands: GET, SET, DEL, EXISTS, GETSET. use crate::bridge::envelope::{PhysicalPlan, Status}; +use crate::control::server::shared::response_payload::payload_or_typed_error; use crate::control::state::SharedState; use nodedb_physical::physical_plan::KvOp; use nodedb_types::{DatabaseId, QualifiedCollection}; @@ -141,7 +142,11 @@ pub(in crate::control::server::resp) async fn handle_set( rls_filters: Vec::new(), }); - match dispatch_kv_write(state, session, plan).await { + // A rejected write surfaces as the error it is, never as `OK`. + match dispatch_kv_write(state, session, plan) + .await + .and_then(payload_or_typed_error) + { Ok(_) => RespValue::ok(), Err(e) => RespValue::from_error(&e), } @@ -167,9 +172,13 @@ pub(in crate::control::server::resp) async fn handle_del( rls_filters: Vec::new(), }); - match dispatch_kv_write(state, session, plan).await { - Ok(resp) => { - let count = payload_field_i64(&resp.payload, "deleted").unwrap_or(0); + // A rejected delete surfaces as the error it is, never as `0` deleted. + match dispatch_kv_write(state, session, plan) + .await + .and_then(payload_or_typed_error) + { + Ok(payload) => { + let count = payload_field_i64(&payload, "deleted").unwrap_or(0); RespValue::integer(count) } Err(e) => RespValue::from_error(&e), @@ -238,10 +247,13 @@ pub(in crate::control::server::resp) async fn handle_getset( rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), }); - match dispatch_kv_write(state, session, plan).await { - Ok(resp) => { - if let Some(serde_json::Value::String(b64)) = - payload_json(&resp.payload).get("old_value") + // A rejected write surfaces as the error it is, never as `nil`. + match dispatch_kv_write(state, session, plan) + .await + .and_then(payload_or_typed_error) + { + Ok(payload) => { + if let Some(serde_json::Value::String(b64)) = payload_json(&payload).get("old_value") && let Ok(mut data) = base64::Engine::decode(&base64::engine::general_purpose::STANDARD, b64) { From 5168e18577a84dc8aee8d98d4d2ba3229dcf1b65 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 18 Sep 2026 13:29:10 +0800 Subject: [PATCH 15/15] docs(sql): document sequence DDL, functions, and evaluation rules Cover CREATE/DROP/ALTER SEQUENCE, SHOW SEQUENCES, DESCRIBE SEQUENCE, the nextval/currval/setval functions, their error SQLSTATEs, and where a sequence accessor is allowed to run per query context. --- docs/query-language.md | 54 ++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 54 insertions(+) diff --git a/docs/query-language.md b/docs/query-language.md index 1879df585..e89590b76 100644 --- a/docs/query-language.md +++ b/docs/query-language.md @@ -260,6 +260,60 @@ INSERT INTO events (id, kind) VALUES ('e1', 'click') RETURNING id, nextval('even `RETURNING *` expands to all columns. Every item is a scalar expression over the target collection: a bare column, a column under an alias (`col AS name`), arithmetic, a function call, or a sequence accessor (`nextval`, `currval`, `setval`). The Data Plane returns the base columns an expression reads, and the Control Plane evaluates the expression once per returned row. A sequence accessor advances once per row, in row order. The clause works in both the simple-query and extended-query (prepared statement) protocols, and `Describe` announces an expression under its alias. +### Sequences + +A sequence is a named `bigint` counter, independent of any collection. + +```sql +CREATE SEQUENCE event_seq START WITH 1 INCREMENT BY 1 MINVALUE 1 CYCLE CACHE 20; +DROP SEQUENCE event_seq; +DROP SEQUENCE IF EXISTS event_seq; +SHOW SEQUENCES; +DESCRIBE SEQUENCE event_seq; +ALTER SEQUENCE event_seq RESTART WITH 100; +``` + +`CREATE SEQUENCE [IF NOT EXISTS] ` accepts these options, in any order: + +| Option | Effect | +|---|---| +| `START [WITH] n` | First value `nextval` returns | +| `INCREMENT [BY] n` | Step between successive values | +| `MINVALUE n` | Lower bound | +| `MAXVALUE n` | Upper bound | +| `CYCLE` / `NO CYCLE` | Wrap to the bound instead of erroring at exhaustion | +| `CACHE n` | Values a node pre-allocates per round-trip to the registry | +| `FORMAT 'template'` | Render template applied to the numeric value | +| `RESET period` | Period after which the counter restarts | +| `GAP_FREE` | Accepted and stored; allocation is not yet serialized or rolled back per transaction | +| `SCOPE name` | Named allocation scope | + +`DROP SEQUENCE [IF EXISTS] ` removes it. `SHOW SEQUENCES` lists every sequence. `DESCRIBE SEQUENCE ` reports one sequence's current state. `ALTER SEQUENCE RESTART [WITH n]` and `ALTER SEQUENCE FORMAT '