diff --git a/.config/nextest.toml b/.config/nextest.toml index 5efbb0c7b..cdf24654e 100644 --- a/.config/nextest.toml +++ b/.config/nextest.toml @@ -147,7 +147,7 @@ slow-timeout = { period = "30s", terminate-after = 8 } # hide the cause rather than fix it. The tail this costs is bounded — roughly a # dozen crash/shutdown tests at about ten seconds each. [[profile.default.overrides]] -filter = 'binary(wal_direct_io) | binary(ilp_client_address) | binary(crash_recovery) | binary(crash_recovery_overlays) | binary(crash_recovery_analytics) | binary(crash_resp_kv_write) | binary(crash_metadata_applier_wedge) | binary(crash_dropped_collection_reclaim) | binary(crash_mid_replay) | binary(crash_checkpoint_corruption) | binary(crash_checkpoint_truncate_window) | binary(crash_refused_write_not_resurrected) | binary(crash_replay_fail_stop) | binary(crash_core_stall) | test(/^cases::startup_failure::/) | test(/^cases::shutdown_in_flight::/) | test(/^cases::shutdown_budget::/) | test(/^cases::shutdown_abort_offender::/) | test(/^cases::shutdown_idempotent::/)' +filter = 'binary(wal_direct_io) | binary(ilp_client_address) | binary(crash_recovery) | binary(crash_recovery_overlays) | binary(crash_recovery_analytics) | binary(crash_resp_kv_write) | binary(crash_metadata_applier_wedge) | binary(crash_dropped_collection_reclaim) | binary(crash_purge_not_resurrected) | binary(crash_mid_replay) | binary(crash_checkpoint_corruption) | binary(crash_checkpoint_truncate_window) | binary(crash_refused_write_not_resurrected) | binary(crash_replay_fail_stop) | binary(crash_core_stall) | binary(crash_replay_stamp) | binary(crash_replay_stamp_calvin) | binary(calvin_hold_liveness) | binary(apply_pipeline_group_independence) | binary(crash_kv_atomic_autocommit) | test(/^cases::startup_failure::/) | test(/^cases::shutdown_in_flight::/) | test(/^cases::shutdown_budget::/) | test(/^cases::shutdown_abort_offender::/) | test(/^cases::shutdown_idempotent::/)' test-group = 'server-process' threads-required = 'num-test-threads' diff --git a/Cargo.lock b/Cargo.lock index c47be165a..466f5ce30 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -4558,6 +4558,7 @@ dependencies = [ "nodedb-query", "nodedb-sql", "nodedb-types", + "rust_decimal", "serde", "thiserror 2.0.20", "zerompk", @@ -4648,6 +4649,7 @@ dependencies = [ name = "nodedb-test-support" version = "0.5.0" dependencies = [ + "async-trait", "base64 0.23.1", "bytes", "futures", diff --git a/docs/kv.md b/docs/kv.md index 18baafdf9..5d857ffc4 100644 --- a/docs/kv.md +++ b/docs/kv.md @@ -155,10 +155,21 @@ SELECT KV_GETSET('session_token', 'player-123', 'new-token-xyz'); **RESP (Redis) equivalents:** `INCR`, `DECR`, `INCRBY`, `DECRBY`, `INCRBYFLOAT`, `GETSET` — all work over the RESP protocol. +**Value shapes:** + +- Raw value (a single `value` column, or RESP `SET`): the value is a byte string. `INCR`/`INCRBY`/`DECR`/`DECRBY` read it as decimal integer text and store the result as decimal text. `INCRBYFLOAT` and `KV_INCR_FLOAT` read decimal text (plain or exponent form, such as `5.0e3`), add exactly, and store the trimmed decimal text: `"0.1"` plus `0.2` stores `"0.3"`, `"3.0"` plus `0` stores `"3"`. Exact addition covers 28 significant digits below 7.9e28. A value outside that range adds in 64-bit float. `SET k 5` then `INCR k` leaves `"6"`. +- Absent key: the counter starts at 0 and is stored as decimal text. +- Typed row (several columns): the first numeric column in key order moves. Every other column stays. + **Error handling:** -- `TYPE_MISMATCH` (SQLSTATE 42846) — INCR on a non-numeric value -- `OVERFLOW` (SQLSTATE 22003) — i64 overflow on INCR +| Condition | RESP reply | SQLSTATE | +|---|---|---| +| Raw value is not a decimal integer in the i64 range | `ERR value is not an integer or out of range` | `22P02` | +| Raw value is not a decimal float | `ERR value is not a valid float` | `22P02` | +| Integer result leaves the i64 range | `ERR increment or decrement would overflow` | `22003` | +| Float result is NaN or infinite | `ERR increment would produce NaN or Infinity` | `22003` | +| Typed row has no numeric column | `WRONGTYPE ...` | `42846` | ## Sorted Indexes (Leaderboards) diff --git a/nodedb-client/src/native/connection/response.rs b/nodedb-client/src/native/connection/response.rs index 894ae949e..15a265a43 100644 --- a/nodedb-client/src/native/connection/response.rs +++ b/nodedb-client/src/native/connection/response.rs @@ -40,10 +40,16 @@ fn error_frame_to_typed( if payload.ndb_code == 0 { return NodeDbError::internal(payload.message.clone()); } - NodeDbError::from_wire( + let error = NodeDbError::from_wire_with_details( nodedb_types::error::ErrorCode(payload.ndb_code), payload.message.clone(), - ) + payload.details.clone(), + ); + // The typed cause, when the server sent one, becomes the error's cause. + match &payload.cause { + Some(cause) => error.with_cause(cause.to_error()), + None => error, + } } pub(super) fn response_to_query_result(resp: NativeResponse) -> NodeDbResult { diff --git a/nodedb-cluster-tests/Cargo.toml b/nodedb-cluster-tests/Cargo.toml index 40e10decf..fd6403f2c 100644 --- a/nodedb-cluster-tests/Cargo.toml +++ b/nodedb-cluster-tests/Cargo.toml @@ -10,6 +10,10 @@ description = "3-node integration test suite for NodeDB. Heavy; runs in its own [lib] path = "src/lib.rs" +[features] +# Arms the in-process fail points the cluster tests park writes at. +failpoints = ["nodedb/failpoints", "nodedb-types/failpoints"] + [dependencies] [dev-dependencies] diff --git a/nodedb-cluster-tests/tests/cluster_common/calvin_test_node.rs b/nodedb-cluster-tests/tests/cluster_common/calvin_test_node.rs index 1827537c2..b421c72a3 100644 --- a/nodedb-cluster-tests/tests/cluster_common/calvin_test_node.rs +++ b/nodedb-cluster-tests/tests/cluster_common/calvin_test_node.rs @@ -27,6 +27,8 @@ //! for shutdown. //! - `add_vshard_sender(vshard_id, sender)` — wire a per-vshard channel //! into the state machine so tests can assert fan-out. +//! - `try_recv_txn(rx)` — read the next sequenced txn a fan-out channel +//! holds. #![allow(dead_code)] // Not every test file uses every helper. @@ -130,24 +132,15 @@ impl CalvinTestNode { /// Register a per-vshard output sender so the state machine can /// fan out sequenced transactions to the receiving test code. - pub fn add_vshard_sender(&self, vshard_id: u32, sender: mpsc::Sender) { - // The state machine now fans out `SchedulerInput`; these tests assert on - // the sequenced-txn stream, so adapt: forward only `Txn` payloads to the - // caller's `SequencedTxn` channel (reservation inputs don't occur here). - let (adapt_tx, mut adapt_rx) = mpsc::channel::(512); - tokio::spawn(async move { - while let Some(input) = adapt_rx.recv().await { - if let SchedulerInput::Txn(txn) = input - && sender.send(txn).await.is_err() - { - break; - } - } - }); + /// + /// The state machine sends into `sender` itself, inside `apply`, before + /// it advances `last_applied_epoch`. A test that waits for the epoch can + /// then read the fan-out with `try_recv_txn` and no further wait. + pub fn add_vshard_sender(&self, vshard_id: u32, sender: mpsc::Sender) { self.state_machine .lock() .unwrap_or_else(|p| p.into_inner()) - .set_vshard_sender(vshard_id, adapt_tx); + .set_vshard_sender(vshard_id, sender); } /// Start the sequencer service epoch-ticker task on this node. @@ -417,3 +410,14 @@ pub async fn wait_for_sequencer_leader( tokio::time::sleep(step).await; } } + +/// The next sequenced txn `rx` holds, skipping any other scheduler input. +/// `None` when the channel holds no txn. +pub fn try_recv_txn(rx: &mut mpsc::Receiver) -> Option { + while let Ok(input) = rx.try_recv() { + if let SchedulerInput::Txn(txn) = input { + return Some(txn); + } + } + None +} diff --git a/nodedb-cluster-tests/tests/cluster_common/mod.rs b/nodedb-cluster-tests/tests/cluster_common/mod.rs index 4e828b2fc..4657be030 100644 --- a/nodedb-cluster-tests/tests/cluster_common/mod.rs +++ b/nodedb-cluster-tests/tests/cluster_common/mod.rs @@ -7,6 +7,6 @@ pub mod rebalancer; pub mod test_node; pub use calvin_test_node::{ - CalvinApplier, CalvinTestNode, spawn_with_sequencer, wait_for_sequencer_leader, + CalvinApplier, CalvinTestNode, spawn_with_sequencer, try_recv_txn, wait_for_sequencer_leader, }; pub use test_node::{NoopApplier, TestNode, test_transport, wait_for}; diff --git a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_normal.rs b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_normal.rs index a052d53c9..4db960948 100644 --- a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_normal.rs +++ b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_normal.rs @@ -19,22 +19,21 @@ use std::time::Duration; use nodedb_cluster::calvin::{ sequencer::{SequencerConfig, new_inbox}, - types::{EngineKeySet, ReadWriteSet, SequencedTxn, SortedVec, TxClass, VersionedReadSet}, -}; -use nodedb_types::{ - TenantId, - id::{DatabaseId, VShardId}, + types::{EngineKeySet, ReadWriteSet, SchedulerInput, SortedVec, TxClass, VersionedReadSet}, }; +use nodedb_types::{TenantId, id::DatabaseId}; use tokio::sync::mpsc; -use super::cluster_common::{spawn_with_sequencer, wait_for_sequencer_leader}; +use super::cluster_common::{spawn_with_sequencer, try_recv_txn, wait_for_sequencer_leader}; /// Find two collection names that hash to distinct vshards. fn two_distinct_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); @@ -48,8 +47,12 @@ fn two_distinct_collections() -> (String, String) { fn make_multishard_txclass() -> (TxClass, u32, u32) { let (col_a, col_b) = two_distinct_collections(); - let va = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_a).as_u32(); - let vb = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_b).as_u32(); + let va = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &col_a) + .vshard() + .as_u32(); + let vb = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &col_b) + .vshard() + .as_u32(); let write_set = ReadWriteSet::new(vec![ EngineKeySet::Document { collection: col_a, @@ -92,11 +95,15 @@ async fn sequencer_normal_path_commit_on_all_replicas() { // Wire per-vshard receivers on every node. let (tx_a, col_b_name) = two_distinct_collections(); - let va = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &tx_a).as_u32(); - let vb = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_b_name).as_u32(); - - let mut vshard_rxs_a: Vec> = Vec::new(); - let mut vshard_rxs_b: Vec> = Vec::new(); + let va = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &tx_a) + .vshard() + .as_u32(); + let vb = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &col_b_name) + .vshard() + .as_u32(); + + let mut vshard_rxs_a: Vec> = Vec::new(); + let mut vshard_rxs_b: Vec> = Vec::new(); for node in &nodes { let (tx_a_ch, rx_a) = mpsc::channel(64); let (tx_b_ch, rx_b) = mpsc::channel(64); @@ -149,8 +156,8 @@ async fn sequencer_normal_path_commit_on_all_replicas() { .zip(vshard_rxs_b.iter_mut()) .enumerate() { - let got_a = rx_a.try_recv().is_ok(); - let got_b = rx_b.try_recv().is_ok(); + let got_a = try_recv_txn(rx_a).is_some(); + let got_b = try_recv_txn(rx_b).is_some(); assert!( got_a || got_b, "node {}: neither vshard receiver got the txn fan-out", diff --git a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_shard_failover.rs b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_shard_failover.rs index fe7ad0979..ca445d371 100644 --- a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_shard_failover.rs +++ b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_shard_failover.rs @@ -43,15 +43,12 @@ use std::time::Duration; use nodedb_cluster::calvin::{ sequencer::{SequencerConfig, new_inbox}, - types::{EngineKeySet, ReadWriteSet, SequencedTxn, SortedVec, TxClass, VersionedReadSet}, -}; -use nodedb_types::{ - TenantId, - id::{DatabaseId, VShardId}, + types::{EngineKeySet, ReadWriteSet, SchedulerInput, SortedVec, TxClass, VersionedReadSet}, }; +use nodedb_types::{TenantId, id::DatabaseId}; use tokio::sync::mpsc; -use super::cluster_common::{spawn_with_sequencer, wait_for_sequencer_leader}; +use super::cluster_common::{spawn_with_sequencer, try_recv_txn, wait_for_sequencer_leader}; // ── Helpers ────────────────────────────────────────────────────────────────── @@ -59,7 +56,9 @@ fn two_distinct_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); @@ -130,11 +129,15 @@ async fn scheduler_catchup_via_raft_log_replay() { // Wire per-vshard receivers on every node so we can verify fan-out. let (col_a, col_b) = two_distinct_collections(); - let va = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_a).as_u32(); - let vb = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_b).as_u32(); + let va = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &col_a) + .vshard() + .as_u32(); + let vb = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &col_b) + .vshard() + .as_u32(); - let mut vshard_rxs_a: Vec> = Vec::new(); - let mut vshard_rxs_b: Vec> = Vec::new(); + let mut vshard_rxs_a: Vec> = Vec::new(); + let mut vshard_rxs_b: Vec> = Vec::new(); for node in &nodes { let (tx_a, rx_a) = mpsc::channel(128); let (tx_b, rx_b) = mpsc::channel(128); @@ -246,7 +249,7 @@ async fn scheduler_catchup_via_raft_log_replay() { // Drain whatever arrived — we care that the routing worked, not the count. let mut total_received = 0usize; for rx in vshard_rxs_a.iter_mut().chain(vshard_rxs_b.iter_mut()) { - while rx.try_recv().is_ok() { + while try_recv_txn(rx).is_some() { total_received += 1; } } diff --git a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_ollp.rs b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_ollp.rs index 1ff200f81..f71e163d5 100644 --- a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_ollp.rs +++ b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_ollp.rs @@ -37,15 +37,12 @@ use std::time::Duration; use nodedb_cluster::calvin::{ sequencer::{SequencerConfig, new_inbox}, - types::{EngineKeySet, ReadWriteSet, SequencedTxn, SortedVec, TxClass, VersionedReadSet}, -}; -use nodedb_types::{ - TenantId, - id::{DatabaseId, VShardId}, + types::{EngineKeySet, ReadWriteSet, SchedulerInput, SortedVec, TxClass, VersionedReadSet}, }; +use nodedb_types::{TenantId, id::DatabaseId}; use tokio::sync::mpsc; -use super::cluster_common::{spawn_with_sequencer, wait_for_sequencer_leader}; +use super::cluster_common::{spawn_with_sequencer, try_recv_txn, wait_for_sequencer_leader}; /// Find two collection names that hash to distinct vshards. /// @@ -54,7 +51,9 @@ fn two_distinct_vshard_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("ollp_col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); @@ -104,14 +103,17 @@ fn make_ollp_tx_class( .expect("valid multi-vshard OLLP TxClass") } -/// Assert: one specific vshard channel received at least one SequencedTxn. +/// Assert: one specific vshard channel received at least one sequenced txn. +/// +/// The state machine sends into the channel inside `apply`, before it +/// advances the epoch the caller waited for, so the txn is already there. fn assert_fan_out_received( - rx: &mut mpsc::Receiver, + rx: &mut mpsc::Receiver, vshard_id: u32, replica_idx: usize, ) { assert!( - rx.try_recv().is_ok(), + try_recv_txn(rx).is_some(), "replica {replica_idx}: vshard {vshard_id} fan-out channel received no txn" ); } @@ -134,13 +136,16 @@ async fn ollp_bulk_update_txclass_admitted_and_fanned_out() { // Find two collections that hash to distinct vshards. let (col_static, col_ollp) = two_distinct_vshard_collections(); - let vs_static = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_static).as_u32(); - let vs_ollp = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_ollp).as_u32(); + let vs_static = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &col_static) + .vshard() + .as_u32(); + let vs_ollp = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &col_ollp) + .vshard() + .as_u32(); // Wire per-vshard fan-out receivers on every replica. - let mut rxs_static: Vec> = Vec::new(); - let mut rxs_ollp: Vec> = Vec::new(); + let mut rxs_static: Vec> = Vec::new(); + let mut rxs_ollp: Vec> = Vec::new(); for node in &nodes { let (tx_s, rx_s) = mpsc::channel(64); let (tx_o, rx_o) = mpsc::channel(64); @@ -191,8 +196,8 @@ async fn ollp_bulk_update_txclass_admitted_and_fanned_out() { // Simulate an OLLP retry: concurrent insert added surrogate 4 to col_ollp. // Re-wire fresh fan-out receivers and re-submit with the corrected set. - let mut retry_rxs_static: Vec> = Vec::new(); - let mut retry_rxs_ollp: Vec> = Vec::new(); + let mut retry_rxs_static: Vec> = Vec::new(); + let mut retry_rxs_ollp: Vec> = Vec::new(); for node in &nodes { let (tx_s, rx_s) = mpsc::channel(64); let (tx_o, rx_o) = mpsc::channel(64); diff --git a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_pgwire.rs b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_pgwire.rs index 2be2ae5a6..cd338f3f6 100644 --- a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_pgwire.rs +++ b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_pgwire.rs @@ -20,15 +20,12 @@ use std::time::Duration; use nodedb_cluster::calvin::{ sequencer::{SequencerConfig, new_inbox}, - types::{EngineKeySet, ReadWriteSet, SequencedTxn, SortedVec, TxClass, VersionedReadSet}, -}; -use nodedb_types::{ - TenantId, - id::{DatabaseId, VShardId}, + types::{EngineKeySet, ReadWriteSet, SchedulerInput, SortedVec, TxClass, VersionedReadSet}, }; +use nodedb_types::{TenantId, id::DatabaseId}; use tokio::sync::mpsc; -use super::cluster_common::{spawn_with_sequencer, wait_for_sequencer_leader}; +use super::cluster_common::{spawn_with_sequencer, try_recv_txn, wait_for_sequencer_leader}; /// Find two collection names that hash to distinct vshards. /// @@ -38,7 +35,9 @@ fn two_distinct_vshard_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("orders_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); @@ -96,11 +95,15 @@ async fn multi_vshard_insert_via_sequencer_admitted_and_replicated() { // Wire per-vshard fan-out receivers on every node. let (col_a, col_b) = two_distinct_vshard_collections(); - let va = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_a).as_u32(); - let vb = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_b).as_u32(); - - let mut fan_out_rxs_a: Vec> = Vec::new(); - let mut fan_out_rxs_b: Vec> = Vec::new(); + let va = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &col_a) + .vshard() + .as_u32(); + let vb = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &col_b) + .vshard() + .as_u32(); + + let mut fan_out_rxs_a: Vec> = Vec::new(); + let mut fan_out_rxs_b: Vec> = Vec::new(); for node in &nodes { let (tx_a, rx_a) = mpsc::channel(64); let (tx_b, rx_b) = mpsc::channel(64); @@ -153,8 +156,8 @@ async fn multi_vshard_insert_via_sequencer_admitted_and_replicated() { .zip(fan_out_rxs_b.iter_mut()) .enumerate() { - let got_a = rx_a.try_recv().is_ok(); - let got_b = rx_b.try_recv().is_ok(); + let got_a = try_recv_txn(rx_a).is_some(); + let got_b = try_recv_txn(rx_b).is_some(); assert!( got_a || got_b, "replica {i}: neither vshard fan-out channel received the txn" diff --git a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_sequencer_failover.rs b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_sequencer_failover.rs index bcae77704..39362247e 100644 --- a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_sequencer_failover.rs +++ b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_sequencer_failover.rs @@ -15,10 +15,7 @@ use nodedb_cluster::calvin::{ sequencer::{SequencerConfig, new_inbox}, types::{EngineKeySet, ReadWriteSet, SortedVec, TxClass, VersionedReadSet}, }; -use nodedb_types::{ - TenantId, - id::{DatabaseId, VShardId}, -}; +use nodedb_types::{TenantId, id::DatabaseId}; use super::cluster_common::{spawn_with_sequencer, wait_for_sequencer_leader}; @@ -26,7 +23,9 @@ fn two_distinct_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); diff --git a/nodedb-cluster-tests/tests/common_suite/cases/assign_surrogate_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/assign_surrogate_cross_node.rs index a631106fb..bff04e519 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/assign_surrogate_cross_node.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/assign_surrogate_cross_node.rs @@ -101,9 +101,8 @@ async fn assign_remote_surrogate_is_authoritative_and_idempotent() { let s1 = assign_surrogate_routed( &coordinator.shared, vshard, - DB, + nodedb_types::CollectionKey::from_bare(DB, &collection), TENANT, - &collection, pk.as_bytes(), TraceId([0u8; 16]), ) @@ -121,9 +120,8 @@ async fn assign_remote_surrogate_is_authoritative_and_idempotent() { let s2 = assign_surrogate_routed( &coordinator.shared, vshard, - DB, + nodedb_types::CollectionKey::from_bare(DB, &collection), TENANT, - &collection, pk.as_bytes(), TraceId([0u8; 16]), ) diff --git a/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_pk_read.rs b/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_pk_read.rs index c49979c32..cfbe7fd9b 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_pk_read.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_pk_read.rs @@ -116,13 +116,15 @@ async fn cross_node_pk_read_from_learner_node_after_calvin_commit() { .position(|n| n.node_id == learner_id) .expect("learner present in cluster"); - // The 4th node joins `col_a`'s group as a non-voting learner: it - // applies the replicated log but never coordinated this transaction, - // so its catalog holds no binding the coordinator minted. + // The 4th node joins `col_a`'s group after it was mounted: it applies + // the replicated log but never coordinated this transaction, so its + // catalog holds no binding the coordinator minted. It joins as a + // learner, and the rebalancer can promote it to a voter at any point, + // so either role holds the premise. It must not lead the group. let status = fx.cluster.nodes[reader].group_status_line(gid_a); assert!( - status.contains("role=Learner"), - "node {learner_id} must join {col_a}'s group {gid_a} as a learner: {status}" + status.contains("role=Learner") || status.contains("role=Follower"), + "node {learner_id} must replicate {col_a}'s group {gid_a} without leading it: {status}" ); // Coordinator is one of the original 3, chosen by the fixture default diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_array_cell_raft_replication.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_array_cell_raft_replication.rs index 78b3a1491..134c037da 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cluster_array_cell_raft_replication.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_array_cell_raft_replication.rs @@ -71,7 +71,11 @@ fn array_surrogate( shared .credentials .catalog() - .get_surrogate_for_pk(DatabaseId::DEFAULT, tenant, ARRAY, coord_bytes) + .get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, ARRAY), + tenant, + coord_bytes, + ) .ok() .flatten() .map(|s| s.as_u32()) diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_remote_cut.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_remote_cut.rs new file mode 100644 index 000000000..532dd1a56 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_remote_cut.rs @@ -0,0 +1,276 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A backup's consistent cut binds a remote source node too. +//! +//! The backup's coordinator snapshots every vShard from the leader of its +//! group. The test parks a write on the collection's group leader only, at +//! the fail gate `funnel::before_dispatch::node::`, between +//! its record and its core. The coordinator is another node, so its own +//! replica applies the write and its own cut passes. The leader must take the +//! same cut before it snapshots: the backup waits until the parked write +//! applies there, and the envelope holds the row. After the purge settles on +//! every node, a restore brings the row back on every node. +//! +//! Requires `--features failpoints`. + +#![cfg(feature = "failpoints")] + +use std::time::Duration; + +use bytes::Bytes; +use futures::{SinkExt, StreamExt}; +use nodedb_types::backup_envelope::{ + DEFAULT_MAX_TOTAL_BYTES, DatabaseDataSection, parse_encrypted, +}; +use nodedb_types::fail_point::{FailAction, FailGuard}; + +use crate::common; +use common::cluster_harness::wait::wait_for; +use common::cluster_harness::{TestCluster, TestClusterNode}; + +/// Fixed test KEK the cluster harness injects into every node. +const TEST_KEK: [u8; 32] = [0x42u8; 32]; + +const TENANT: u64 = 1; +const COLLECTION: &str = "rcut_docs"; + +/// How long the parked write and the backup must stay unfinished. +const PARKED_FOR: Duration = Duration::from_millis(1500); + +fn db_detail(e: &tokio_postgres::Error) -> String { + match e.as_db_error() { + Some(db) => format!("{}: {}", db.code().code(), db.message()), + None => format!("{e}"), + } +} + +async fn drain_backup(client: &tokio_postgres::Client) -> Result, String> { + let stream = client + .copy_out(&format!("COPY (BACKUP TENANT {TENANT}) TO STDOUT")) + .await + .map_err(|e| db_detail(&e))?; + let mut bytes = Vec::new(); + let mut stream = Box::pin(stream); + while let Some(chunk) = stream.next().await { + bytes.extend_from_slice(&chunk.map_err(|e| db_detail(&e))?); + } + Ok(bytes) +} + +async fn push_restore(client: &tokio_postgres::Client, envelope: Vec) -> Result<(), String> { + let sink = client + .copy_in::<_, Bytes>(&format!("COPY tenant_restore({TENANT}) FROM STDIN")) + .await + .map_err(|e| db_detail(&e))?; + let mut sink = Box::pin(sink); + sink.as_mut() + .send(Bytes::from(envelope)) + .await + .map_err(|e| db_detail(&e))?; + sink.as_mut() + .finish() + .await + .map(|_| ()) + .map_err(|e| db_detail(&e)) +} + +/// The leader of `group_id` in `node`'s routing table: the node a backup +/// coordinated there snapshots the group's vShards from. +fn routing_leader(node: &TestClusterNode, group_id: u64) -> u64 { + node.shared + .cluster_routing + .as_ref() + .and_then(|routing| { + routing + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_info(group_id) + .map(|info| info.leader) + }) + .unwrap_or(0) +} + +/// Whether `node` purged `collection`: its catalog row is gone and its WAL +/// tombstone is recorded, so the async purge ran on its Data Plane. +fn purged_on(node: &TestClusterNode, collection: &str) -> bool { + let catalog = node.shared.credentials.catalog(); + let active = matches!( + catalog.get_collection(nodedb_types::DatabaseId::DEFAULT, TENANT, collection), + Ok(Some(c)) if c.is_active + ); + let tombstoned = catalog + .load_wal_tombstones() + .map(|set| { + set.iter() + .any(|(_, tenant, name, lsn)| tenant == TENANT && name == collection && lsn > 0) + }) + .unwrap_or(false); + !active && tombstoned +} + +/// Whether the backup `envelope` holds a KV row of `collection` whose key +/// carries `key`. Each data section wraps one database's snapshot, and a KV +/// table's section key is `"{db}:{tenant}:{collection}"`. +fn envelope_holds_kv_row(envelope: &[u8], collection: &str, key: &[u8]) -> bool { + let parsed = + parse_encrypted(envelope, DEFAULT_MAX_TOTAL_BYTES, &TEST_KEK).expect("parse the envelope"); + let table_key = format!("0:{TENANT}:{collection}"); + parsed + .sections + .iter() + .filter_map(|section| zerompk::from_msgpack::(§ion.body).ok()) + .filter_map(|section| { + zerompk::from_msgpack::(§ion.snapshot).ok() + }) + .flat_map(|snapshot| snapshot.kv_tables) + .filter(|(name, _)| *name == table_key) + .filter_map(|(_, rows)| zerompk::from_msgpack::, Vec, u64)>>(&rows).ok()) + .flatten() + .any(|(row_key, _, _)| row_key.windows(key.len()).any(|window| window == key)) +} + +async fn connect(pg_addr: std::net::SocketAddr) -> tokio_postgres::Client { + let conn_str = format!( + "host={} port={} user=nodedb dbname=default", + pg_addr.ip(), + pg_addr.port() + ); + let (client, connection) = tokio_postgres::connect(&conn_str, tokio_postgres::NoTls) + .await + .expect("connect a second client"); + tokio::spawn(connection); + client +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_backup_waits_for_a_write_held_on_a_remote_source_node() { + let cluster = TestCluster::spawn_three().await.expect("cluster"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {COLLECTION} (key STRING PRIMARY KEY, value STRING) \ + WITH (engine='kv')" + )) + .await + .expect("CREATE COLLECTION"); + + // The coordinator snapshots the collection's vShard from the group leader + // its routing table names. Wait until every node's table names the + // elected leader, so the coordinator picked below names it too. + let group_id = cluster.nodes[0] + .group_id_for_collection(COLLECTION) + .expect("the collection's data group"); + let elected = || { + cluster.nodes[0] + .all_group_leaders() + .into_iter() + .find(|(group, _)| *group == group_id) + .map_or(0, |(_, leader)| leader) + }; + wait_for( + "every routing table names the elected leader of the collection's group", + Duration::from_secs(10), + Duration::from_millis(50), + || { + let leader = elected(); + leader != 0 + && cluster + .nodes + .iter() + .all(|node| routing_leader(node, group_id) == leader) + }, + ) + .await; + let source = elected(); + let coordinator = cluster + .nodes + .iter() + .position(|node| node.node_id != source) + .expect("a node that is not the source"); + + // Park the write on the source node only. + let gate_dir = tempfile::tempdir().expect("gate tempdir"); + let release = gate_dir.path().join("release-source-apply"); + let _gate = FailGuard::install( + &format!("funnel::before_dispatch::node{source}::{COLLECTION}"), + FailAction::WaitForFile(release.clone()), + ); + + let writer = connect(cluster.nodes[coordinator].pg_addr).await; + let insert = tokio::spawn(async move { + writer + .simple_query(&format!( + "INSERT INTO {COLLECTION} (key, value) VALUES ('held', 'x')" + )) + .await + .map(|_| ()) + .map_err(|e| db_detail(&e)) + }); + tokio::time::sleep(PARKED_FOR).await; + + let backup_client = connect(cluster.nodes[coordinator].pg_addr).await; + let backup = tokio::spawn(async move { drain_backup(&backup_client).await }); + tokio::time::sleep(PARKED_FOR).await; + assert!( + !backup.is_finished(), + "the backup snapshotted node {source} before a write held there had its outcome" + ); + + std::fs::write(&release, b"release").expect("release the held apply"); + insert + .await + .expect("insert task") + .unwrap_or_else(|e| panic!("insert: {e}")); + let envelope = backup + .await + .expect("backup task") + .unwrap_or_else(|e| panic!("backup: {e}")); + + assert!( + envelope_holds_kv_row(&envelope, COLLECTION, b"held"), + "the backup missed a write committed before its cut" + ); + + cluster + .exec_ddl_on_any_leader(&format!("DROP COLLECTION {COLLECTION} PURGE")) + .await + .expect("purge the collection"); + // The purge reaches each node's Data Plane asynchronously. A purge that + // lands after the restore would remove the restored row. + wait_for( + "every node purged the collection", + Duration::from_secs(10), + Duration::from_millis(50), + || cluster.nodes.iter().all(|node| purged_on(node, COLLECTION)), + ) + .await; + push_restore(&cluster.nodes[coordinator].client, envelope) + .await + .unwrap_or_else(|e| panic!("restore: {e}")); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + for node in &cluster.nodes { + let rows = node + .client + .simple_query(&format!( + "SELECT value FROM {COLLECTION} WHERE key = 'held'" + )) + .await + .unwrap_or_else(|e| panic!("read the restored row: {}", db_detail(&e))); + let values: Vec = rows + .iter() + .filter_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get(0).map(str::to_owned), + _ => None, + }) + .collect(); + assert_eq!( + values, + vec!["x".to_owned()], + "node {} does not hold the restored row", + node.node_id + ); + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_restore.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_restore.rs index df7195af2..5f23d1dc5 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_restore.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_restore.rs @@ -82,38 +82,6 @@ fn db_detail(e: &tokio_postgres::Error) -> String { } } -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn three_node_backup_gathers_one_section_per_node() { - let cluster = TestCluster::spawn_three().await.expect("cluster"); - - let bytes = drain_backup(0, &cluster, TENANT).await; - let env = parse_envelope(&bytes, DEFAULT_MAX_TOTAL_BYTES, &TEST_KEK).expect("parse envelope"); - assert_eq!( - env.meta.tenant_id, TENANT, - "envelope tenant id should match request" - ); - assert!( - env.meta.source_vshard_count >= 1, - "envelope must record source vshard count, got {}", - env.meta.source_vshard_count - ); - // One section per unique cluster node — three nodes, three sections. - // The orchestrator dedupes by node id (leader + replicas of every group). - assert!( - !env.sections.is_empty() && env.sections.len() <= 3, - "expected 1..=3 sections, got {}", - env.sections.len() - ); - let origins: std::collections::BTreeSet = - env.sections.iter().map(|s| s.origin_node_id).collect(); - assert!( - !origins.contains(&0), - "origin_node_id 0 must not appear in cluster sections: {origins:?}" - ); - - cluster.shutdown().await; -} - #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn three_node_roundtrip_preserves_data() { let cluster = TestCluster::spawn_three().await.expect("cluster"); @@ -313,21 +281,23 @@ async fn backup_watermark_advances_after_writes() { // 3-node spawn cost is acceptable. // ──────────────────────────────────────────────────────────────────── -// Mid-flight node failure during restore fan-out. +// Restore into a group that lost its quorum. // -// The restore orchestrator iterates per-node sub-snapshots via -// `sync_dispatch` (local) or `RaftRpc::ExecuteRequest` (remote) and -// surfaces the first per-node error. Two contracts must hold: +// A restore commits every write through Raft: catalog rows through the +// metadata group, rows through each collection's data group. One dead node +// of three leaves every group a majority, and the restore succeeds. Two dead +// nodes leave no group a majority, and nothing can commit. Two contracts +// must hold: // -// 1. Loud failure — the client receives a structured error that -// names the failing node, never silent partial success. -// 2. Idempotent retry — the engine-level PointPut writes are -// idempotent, so a subsequent retry after the failed node is -// restored converges to the expected state. +// 1. Loud, prompt failure — the client receives a typed error that names +// the group and the nodes it cannot reach, well within the statement +// deadline, never a hang and never a silent success. +// 2. Nothing applied — no group on the surviving node applied an entry of +// the refused restore. // ──────────────────────────────────────────────────────────────────── #[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn restore_surfaces_failing_node_id_on_midflight_failure() { +async fn restore_refuses_a_group_without_quorum_and_applies_nothing() { let mut cluster = TestCluster::spawn_three().await.expect("cluster"); cluster @@ -347,64 +317,55 @@ async fn restore_surfaces_failing_node_id_on_midflight_failure() { .await .unwrap_or_else(|e| panic!("insert f{i}: {}", db_detail(&e))); } + let bytes = drain_backup(0, &cluster, TENANT).await; - // Fault-inject: take down node 2, then pin node 0's routing table - // so it still believes node 2 is the leader for every raft group. - // Without the stale-route pin, quorum-replicated failover hides the - // fault entirely — restore transparently re-routes to a surviving - // replica and succeeds. Pinning forces the restore fan-out to - // attempt dispatch against the dead peer so the structured - // error-naming contract is actually exercised. - let downed_node_id = cluster.nodes[2].node_id; - let downed = cluster.nodes.remove(2); - downed.shutdown().await; - // Wait for the surviving nodes' SWIM/topology subsystem to observe - // the peer death before pinning the stale routes. If we pin before - // SWIM converges, an in-flight SWIM update can land *after* the pin - // and overwrite it with a fresh leader hint, defeating the - // fault-injection. The pin must be the last write to the routing - // table for `downed_node_id`'s groups. + // Take down two of the three nodes: no group keeps a majority. + let mut downed_ids = Vec::new(); + for _ in 0..2 { + let downed = cluster.nodes.remove(1); + downed_ids.push(downed.node_id); + downed.shutdown().await; + } + downed_ids.sort_unstable(); wait_for( - "node 0 marks downed peer as inactive in its topology view", - Duration::from_secs(10), + "the surviving node marks both downed peers inactive", + Duration::from_secs(20), Duration::from_millis(20), - || cluster.nodes[0].active_topology_size() < 3, + || cluster.nodes[0].active_topology_size() == 1, ) .await; - // Back up only once the topology has settled, and before the routes - // are pinned. - // - // The restore path refuses an envelope whose watermark predates the - // destination's last observed write-HLC, and that high-water advances on - // every successful dispatch for the tenant — not only on writes the test - // issues itself. Taking the backup first leaves the node teardown and the - // SWIM convergence window sitting between the envelope and the restore, - // and anything dispatched in there carries the high-water past the - // envelope. The restore then fails the staleness check instead of - // reaching the fan-out, and the assertion below sees the wrong loud - // failure. Capturing after the teardown keeps the envelope dominant; - // pinning afterwards adds nothing, since it only rewrites a local routing - // table. - // - // The backup itself still succeeds with the peer down: routing has failed - // over to the surviving replicas at this point, which is exactly the - // transparent recovery the pin below goes on to defeat. - let bytes = drain_backup(0, &cluster, TENANT).await; - - for group_id in 0..8u64 { - cluster.nodes[0].force_stale_route_for_test(group_id, downed_node_id); - } - - let err = push_restore(0, &cluster, TENANT, bytes) - .await - .expect_err("restore must fail loudly when fan-out targets a dead node"); + let applied_before = cluster.nodes[0].shared.group_watchers().snapshot(); + let started = std::time::Instant::now(); + let err = tokio::time::timeout( + Duration::from_secs(10), + push_restore(0, &cluster, TENANT, bytes), + ) + .await + .expect("a restore into a group without quorum must fail promptly, not hang") + .expect_err("a restore into a group without quorum must fail loudly"); assert!( - err.contains(&format!("node {downed_node_id}")) - || err.contains(&downed_node_id.to_string()), - "restore error must name the failing node id {downed_node_id} so the \ - operator can act; got: {err}" + err.contains("has no reachable quorum"), + "expected the typed quorum refusal, got: {err}" + ); + assert!( + err.contains(&format!("unreachable {downed_ids:?}")), + "the refusal must name the unreachable nodes {downed_ids:?}, got: {err}" + ); + assert!( + err.contains("raft group "), + "the refusal must name the group, got: {err}" + ); + assert!( + started.elapsed() < Duration::from_secs(5), + "the refusal took {:?}; it must not wait out a commit deadline", + started.elapsed() + ); + assert_eq!( + cluster.nodes[0].shared.group_watchers().snapshot(), + applied_before, + "the refused restore applied an entry on the surviving node" ); cluster.shutdown().await; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_restore_databases.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_restore_databases.rs new file mode 100644 index 000000000..eed4c9148 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_restore_databases.rs @@ -0,0 +1,251 @@ +// SPDX-License-Identifier: BUSL-1.1 +//! Cluster BACKUP / RESTORE of a tenant whose collections span databases. +//! +//! Each source node snapshots every database of the tenant after one +//! consistent cut, and each snapshot becomes one data section that names its +//! database. A restore into a FRESH cluster creates the named databases +//! through the metadata group and brings every row back into the database it +//! came from, readable from a node that did not coordinate the restore. +//! +//! The tenant writes the same collection names in every database, with a +//! different row count in each: a row restored into the wrong database +//! changes that database's count. + +use std::collections::{BTreeMap, BTreeSet}; +use std::time::Duration; + +use bytes::Bytes; +use futures::{SinkExt, StreamExt}; +use nodedb_types::backup_envelope::{ + DEFAULT_MAX_TOTAL_BYTES, DatabaseBlob, DatabaseDataSection, SECTION_ORIGIN_CATALOG_ROWS, + SECTION_ORIGIN_DATABASES, parse_encrypted as parse_envelope, +}; + +use crate::common; +use common::cluster_harness::TestCluster; + +/// Fixed test KEK injected into cluster nodes via `cluster_harness`. +const TEST_KEK: [u8; 32] = [0x42u8; 32]; + +const TENANT: u64 = 1; + +/// Every database the tenant writes, with the rows each collection holds +/// there. +const DATABASES: [(&str, usize); 3] = [("default", 2), ("cl_sales", 3), ("cl_ops", 4)]; + +/// Every collection the tenant creates in each database. +const COLLECTIONS: [&str; 3] = ["cl_docs", "cl_kv", "cl_cols"]; + +fn db_detail(e: &tokio_postgres::Error) -> String { + match e.as_db_error() { + Some(db) => format!("{}: {}", db.code().code(), db.message()), + None => format!("{e}"), + } +} + +async fn drain_backup(cluster: &TestCluster) -> Vec { + let stream = cluster.nodes[0] + .client + .copy_out(&format!("COPY (BACKUP TENANT {TENANT}) TO STDOUT")) + .await + .unwrap_or_else(|e| panic!("BACKUP TENANT: {}", db_detail(&e))); + let mut bytes = Vec::new(); + let mut stream = Box::pin(stream); + while let Some(chunk) = stream.next().await { + bytes.extend_from_slice(&chunk.unwrap_or_else(|e| panic!("copy chunk: {}", db_detail(&e)))); + } + bytes +} + +async fn push_restore(cluster: &TestCluster, envelope: Vec) -> Result<(), String> { + let sink = cluster.nodes[0] + .client + .copy_in::<_, Bytes>(&format!("COPY tenant_restore({TENANT}) FROM STDIN")) + .await + .map_err(|e| db_detail(&e))?; + let mut sink = Box::pin(sink); + sink.as_mut() + .send(Bytes::from(envelope)) + .await + .map_err(|e| db_detail(&e))?; + sink.as_mut() + .finish() + .await + .map(|_| ()) + .map_err(|e| db_detail(&e)) +} + +/// Switch every node's harness session to `database`. +async fn use_database(cluster: &TestCluster, database: &str) { + for node in &cluster.nodes { + node.exec(&format!("USE DATABASE {database}")) + .await + .unwrap_or_else(|e| panic!("USE DATABASE {database} on node {}: {e}", node.node_id)); + } +} + +/// Create the named databases, then every collection in every database, +/// and fill each through node 0. Each row's text names its database. +async fn seed(cluster: &TestCluster) { + for (database, _) in DATABASES.iter().skip(1) { + cluster + .exec_ddl_on_any_leader(&format!("CREATE DATABASE {database}")) + .await + .unwrap_or_else(|e| panic!("CREATE DATABASE {database}: {e}")); + } + for (database, rows) in DATABASES { + use_database(cluster, database).await; + for ddl in [ + "CREATE COLLECTION cl_docs (id TEXT PRIMARY KEY, content TEXT) \ + WITH (engine='document_strict')", + "CREATE COLLECTION cl_kv (key STRING PRIMARY KEY, value STRING) WITH (engine='kv')", + "CREATE COLLECTION cl_cols COLUMNS (id TEXT, region TEXT, ts BIGINT) \ + WITH (engine='columnar')", + ] { + cluster + .exec_ddl_on_any_leader(ddl) + .await + .unwrap_or_else(|e| panic!("{ddl} in {database}: {e}")); + } + for i in 0..rows { + for insert in [ + format!("INSERT INTO cl_docs (id, content) VALUES ('k{i}', '{database}-{i}')"), + format!("INSERT INTO cl_kv (key, value) VALUES ('k{i}', '{database}-{i}')"), + format!( + "INSERT INTO cl_cols (id, region, ts) VALUES ('k{i}', '{database}-{i}', {i})" + ), + ] { + cluster.nodes[0] + .client + .simple_query(&insert) + .await + .unwrap_or_else(|e| panic!("{insert} in {database}: {}", db_detail(&e))); + } + } + } + use_database(cluster, "default").await; +} + +/// The first column of every row `sql` returns on node `node_idx`. +async fn first_column(cluster: &TestCluster, node_idx: usize, sql: &str) -> Vec { + let messages = cluster.nodes[node_idx] + .client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql} on node {node_idx}: {}", db_detail(&e))); + messages + .into_iter() + .filter_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get(0).map(str::to_owned), + _ => None, + }) + .collect() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn three_node_backup_gathers_one_section_per_node_and_database() { + let cluster = TestCluster::spawn_three().await.expect("cluster"); + seed(&cluster).await; + + let bytes = drain_backup(&cluster).await; + let env = parse_envelope(&bytes, DEFAULT_MAX_TOTAL_BYTES, &TEST_KEK).expect("parse envelope"); + assert_eq!(env.meta.tenant_id, TENANT); + + // The database section lists every database the tenant writes, by name. + let listed: Vec = env + .sections + .iter() + .filter(|s| s.origin_node_id == SECTION_ORIGIN_DATABASES) + .flat_map(|s| { + zerompk::from_msgpack::>(&s.body).expect("decode databases") + }) + .collect(); + let names: BTreeSet<&str> = listed.iter().map(|b| b.name.as_str()).collect(); + let expected: BTreeSet<&str> = DATABASES.iter().map(|(name, _)| *name).collect(); + assert_eq!(names, expected, "the backup must list every database"); + + // Each source node contributes one data section per database — three + // nodes, so 1..=3 sections per database. Metadata sections carry + // sentinel origins and are not counted. + let mut per_database: BTreeMap> = BTreeMap::new(); + for section in env + .sections + .iter() + .filter(|s| s.origin_node_id < SECTION_ORIGIN_CATALOG_ROWS) + { + assert_ne!(section.origin_node_id, 0, "origin 0 must not appear"); + let data: DatabaseDataSection = + zerompk::from_msgpack(§ion.body).expect("a data section names its database"); + assert!( + per_database + .entry(data.database_id) + .or_default() + .insert(section.origin_node_id), + "node {} snapshots database {} once", + section.origin_node_id, + data.database_id + ); + } + let listed_ids: BTreeSet = listed.iter().map(|b| b.database_id).collect(); + let sectioned_ids: BTreeSet = per_database.keys().copied().collect(); + assert_eq!( + sectioned_ids, listed_ids, + "every listed database must have data sections, and no other" + ); + for (database_id, nodes) in &per_database { + assert!( + (1..=3).contains(&nodes.len()), + "database {database_id}: expected 1..=3 source nodes, got {nodes:?}" + ); + } + + cluster.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn three_node_restore_brings_back_every_database() { + // ── SOURCE cluster A ───────────────────────────────────────────────────── + let cluster_a = TestCluster::spawn_three().await.expect("cluster A"); + seed(&cluster_a).await; + let envelope = drain_backup(&cluster_a).await; + cluster_a.shutdown().await; + + // ── TARGET cluster B: fresh, none of the named databases exist ────────── + let cluster_b = TestCluster::spawn_three().await.expect("cluster B"); + push_restore(&cluster_b, envelope) + .await + .expect("RESTORE into a fresh cluster"); + cluster_b + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + + // Read from node 1: the databases, the catalog rows and the rows reached + // it through replication, not through the coordinator's own apply. + for (database, rows) in DATABASES { + use_database(&cluster_b, database).await; + for collection in COLLECTIONS { + let count = + first_column(&cluster_b, 1, &format!("SELECT COUNT(*) FROM {collection}")).await; + assert_eq!( + count, + vec![rows.to_string()], + "{database}.{collection} must hold exactly its own {rows} rows after the restore" + ); + } + let last = rows - 1; + let expected = vec![format!("{database}-{last}")]; + for sql in [ + format!("SELECT content FROM cl_docs WHERE id = 'k{last}'"), + format!("SELECT value FROM cl_kv WHERE key = 'k{last}'"), + format!("SELECT region FROM cl_cols WHERE id = 'k{last}'"), + ] { + assert_eq!( + first_column(&cluster_b, 1, &sql).await, + expected, + "{sql} in {database} must return the row of {database}" + ); + } + } + + cluster_b.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_crdt_replication.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_crdt_replication.rs index 88898b5b4..35df8bb60 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cluster_crdt_replication.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_crdt_replication.rs @@ -133,7 +133,9 @@ async fn crdt_apply_replicates_and_survives_leader_loss() { // Resolve the collection's data group and its leader from node 0's shared // routing view (same idiom as multi_replica_data_groups). - let vshard = nodedb_cluster::routing::vshard_for_collection(DatabaseId::DEFAULT, COLL); + let vshard = nodedb_cluster::routing::vshard_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, COLL), + ); let (group_id, group_leader) = { let routing = cluster.nodes[0] .shared diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_execute_request.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_execute_request.rs index 3cd22a251..73c924b54 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cluster_execute_request.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_execute_request.rs @@ -42,6 +42,7 @@ fn make_kv_put_request( surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let plan_bytes = plan_wire::encode(&plan).expect("encode plan"); @@ -286,6 +287,7 @@ async fn execute_request_cross_node_dispatch() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); plan_wire::encode(&plan).expect("encode plan") }, diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_documents_restart.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_documents_restart.rs new file mode 100644 index 000000000..57bab4ecc --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_documents_restart.rs @@ -0,0 +1,235 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A cluster RESTORE re-issues document rows, their index entries, and graph +//! edges as replicated writes, so every replica holds them durably. +//! +//! One cluster backs up a schemaless collection with a secondary index and a +//! unique index, a strict collection, a strict `bitemporal=true` collection +//! with two versions of one row, and two edges. A second cluster restores the +//! backup through one node, then every node restarts. Before and after the +//! restart, each node reads its own replica: every row, both index lookups, +//! both versions, and both edges. + +use std::time::Duration; + +use bytes::Bytes; +use futures::{SinkExt, StreamExt}; + +use crate::common; +use common::cluster_harness::{TestCluster, TestClusterNode, read_once_a_leader_exists}; + +const TENANT: u64 = 1; + +const COLLECTIONS: &[&str] = &[ + "CREATE COLLECTION crd_people (id STRING PRIMARY KEY, city STRING, email STRING) \ + WITH (engine='document_schemaless')", + "CREATE COLLECTION crd_accounts (id STRING PRIMARY KEY, owner STRING, balance INT) \ + WITH (engine='document_strict')", + "CREATE COLLECTION crd_ledger (id STRING PRIMARY KEY, value STRING) \ + WITH (engine='document_strict', bitemporal=true)", + "CREATE INDEX ON crd_people (city)", + "CREATE UNIQUE INDEX crd_people_email ON crd_people (email)", +]; + +const WRITES: &[&str] = &[ + "INSERT INTO crd_people (id, city, email) VALUES ('alice', 'paris', 'a@x')", + "INSERT INTO crd_people (id, city, email) VALUES ('bob', 'rome', 'b@x')", + "INSERT INTO crd_people (id, city, email) VALUES ('carol', 'paris', 'c@x')", + "GRAPH INSERT EDGE IN 'crd_people' FROM 'alice' TO 'bob' TYPE 'knows'", + "GRAPH INSERT EDGE IN 'crd_people' FROM 'bob' TO 'carol' TYPE 'knows'", + "INSERT INTO crd_accounts (id, owner, balance) VALUES ('acc1', 'alice', 10)", + "INSERT INTO crd_accounts (id, owner, balance) VALUES ('acc2', 'bob', 20)", + "INSERT INTO crd_ledger (id, value) VALUES ('e1', 'draft')", + "UPDATE crd_ledger SET value = 'final' WHERE id = 'e1'", +]; + +fn db_detail(e: &tokio_postgres::Error) -> String { + match e.as_db_error() { + Some(db) => format!("{}: {}", db.code().code(), db.message()), + None => format!("{e}"), + } +} + +async fn drain_backup(client: &tokio_postgres::Client) -> Vec { + let stream = client + .copy_out(&format!("COPY (BACKUP TENANT {TENANT}) TO STDOUT")) + .await + .unwrap_or_else(|e| panic!("copy_out: {}", db_detail(&e))); + let mut bytes = Vec::new(); + let mut stream = Box::pin(stream); + while let Some(chunk) = stream.next().await { + bytes.extend_from_slice(&chunk.unwrap_or_else(|e| panic!("chunk: {}", db_detail(&e)))); + } + bytes +} + +async fn push_restore(client: &tokio_postgres::Client, envelope: Vec) { + let sink = client + .copy_in::<_, Bytes>(&format!("COPY tenant_restore({TENANT}) FROM STDIN")) + .await + .unwrap_or_else(|e| panic!("copy_in: {}", db_detail(&e))); + let mut sink = Box::pin(sink); + sink.as_mut() + .send(Bytes::from(envelope)) + .await + .unwrap_or_else(|e| panic!("send: {}", db_detail(&e))); + sink.as_mut() + .finish() + .await + .unwrap_or_else(|e| panic!("restore: {}", db_detail(&e))); +} + +/// The first column of every row `sql` returns on `node`, sorted. +async fn column(node: &TestClusterNode, sql: &str) -> Vec { + let messages = read_once_a_leader_exists( + sql, + Duration::from_secs(30), + Duration::from_millis(100), + || node.client.simple_query(sql), + ) + .await; + let mut values: Vec = messages + .iter() + .filter_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get(0).map(str::to_owned), + _ => None, + }) + .collect(); + values.sort(); + values +} + +/// Every text cell `sql` returns on `node`, joined. +async fn text(node: &TestClusterNode, sql: &str) -> String { + let messages = read_once_a_leader_exists( + sql, + Duration::from_secs(30), + Duration::from_millis(100), + || node.client.simple_query(sql), + ) + .await; + messages + .iter() + .filter_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => Some( + (0..row.len()) + .filter_map(|i| row.get(i)) + .collect::(), + ), + _ => None, + }) + .collect() +} + +/// Every restored row, index entry, version and edge reads back from +/// `node`'s own replica. +async fn assert_restored_on(node: &TestClusterNode, stage: &str) { + let id = node.node_id; + node.client + .simple_query("SET default_read_consistency = 'eventual'") + .await + .unwrap_or_else(|e| panic!("node {id}: set eventual reads: {}", db_detail(&e))); + assert_eq!( + column(node, "SELECT id FROM crd_people WHERE city = 'paris'").await, + vec!["alice", "carol"], + "{stage}, node {id}: the secondary index lookup" + ); + assert_eq!( + column(node, "SELECT id FROM crd_people WHERE email = 'b@x'").await, + vec!["bob"], + "{stage}, node {id}: the unique index lookup" + ); + assert_eq!( + column(node, "SELECT balance FROM crd_accounts WHERE id = 'acc2'").await, + vec!["20"], + "{stage}, node {id}: the strict row by primary key" + ); + assert_eq!( + column(node, "SELECT owner FROM crd_accounts").await, + vec!["alice", "bob"], + "{stage}, node {id}: every strict row" + ); + assert_eq!( + column(node, "SELECT value FROM crd_ledger WHERE id = 'e1'").await, + vec!["final"], + "{stage}, node {id}: the bitemporal row's current version" + ); + assert_eq!( + column(node, "SELECT value FROM crd_ledger AS OF SYSTEM TIME NULL").await, + vec!["draft", "final"], + "{stage}, node {id}: the bitemporal row keeps both versions" + ); + let from_alice = text( + node, + "GRAPH NEIGHBORS IN 'crd_people' OF 'alice' LABEL 'knows' DIRECTION out", + ) + .await; + assert!( + from_alice.contains("bob"), + "{stage}, node {id}: the edge alice -> bob, got {from_alice}" + ); + let from_bob = text( + node, + "GRAPH NEIGHBORS IN 'crd_people' OF 'bob' LABEL 'knows' DIRECTION out", + ) + .await; + assert!( + from_bob.contains("carol"), + "{stage}, node {id}: the edge bob -> carol, got {from_bob}" + ); +} + +async fn assert_restored(cluster: &TestCluster, stage: &str) { + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + for node in &cluster.nodes { + assert_restored_on(node, stage).await; + } + let duplicate = cluster.nodes[0] + .exec("INSERT INTO crd_people (id, city, email) VALUES ('dave', 'oslo', 'a@x')") + .await; + assert!( + duplicate.is_err(), + "{stage}: the unique index must refuse a restored row's email" + ); +} + +async fn source_backup() -> Vec { + let source = TestCluster::spawn_three().await.expect("source cluster"); + for sql in COLLECTIONS { + source + .exec_ddl_on_any_leader(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + for sql in WRITES { + source.nodes[0] + .exec(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + source + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + let backup = drain_backup(&source.nodes[0].client).await; + source.shutdown().await; + backup +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn restored_documents_indexes_and_edges_survive_a_full_cluster_restart() { + let backup = source_backup().await; + + let target = TestCluster::spawn_three().await.expect("target cluster"); + push_restore(&target.nodes[1].client, backup).await; + assert_restored(&target, "after the restore").await; + + let target = target + .restart_all() + .await + .unwrap_or_else(|e| panic!("restart every node: {e}")); + assert_restored(&target, "after every node restarted").await; + + target.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_refuses_on_non_replica.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_refuses_on_non_replica.rs new file mode 100644 index 000000000..10042fb46 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_refuses_on_non_replica.rs @@ -0,0 +1,125 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! RESTORE's staleness guard reads the write marks of every data group, not +//! the restoring node's memory. +//! +//! With a replication factor of 1, each data group lives on one node. A write +//! after the backup applies only on that node. A restore issued on another +//! node, which never applied the write, still refuses: the guard asks the +//! group's replica for its newest write. + +use std::time::Duration; + +use bytes::Bytes; +use futures::{SinkExt, StreamExt}; + +use crate::common; +use common::cluster_harness::TestCluster; +use common::cluster_harness::wait::wait_for; + +const TENANT: u64 = 1; +const COLLECTION: &str = "nonreplica_docs"; + +fn db_detail(e: &tokio_postgres::Error) -> String { + match e.as_db_error() { + Some(db) => format!("{}: {}", db.code().code(), db.message()), + None => format!("{e}"), + } +} + +async fn drain_backup(client: &tokio_postgres::Client) -> Vec { + let stream = client + .copy_out(&format!("COPY (BACKUP TENANT {TENANT}) TO STDOUT")) + .await + .unwrap_or_else(|e| panic!("copy_out: {}", db_detail(&e))); + let mut bytes = Vec::new(); + let mut stream = Box::pin(stream); + while let Some(chunk) = stream.next().await { + bytes.extend_from_slice(&chunk.unwrap_or_else(|e| panic!("chunk: {}", db_detail(&e)))); + } + bytes +} + +async fn push_restore(client: &tokio_postgres::Client, envelope: Vec) -> Result<(), String> { + let sink = client + .copy_in::<_, Bytes>(&format!("COPY tenant_restore({TENANT}) FROM STDIN")) + .await + .map_err(|e| db_detail(&e))?; + let mut sink = Box::pin(sink); + sink.as_mut() + .send(Bytes::from(envelope)) + .await + .map_err(|e| db_detail(&e))?; + sink.as_mut() + .finish() + .await + .map(|_| ()) + .map_err(|e| db_detail(&e)) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_restore_on_a_node_that_never_applied_the_write_refuses() { + let cluster = TestCluster::spawn_three_with_replication_factor(1) + .await + .expect("cluster"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {COLLECTION} (key STRING PRIMARY KEY, value STRING) \ + WITH (engine='kv')" + )) + .await + .expect("CREATE COLLECTION"); + let group_id = cluster.nodes[0] + .group_id_for_collection(COLLECTION) + .expect("the collection's data group"); + // Every joiner enters every group as a learner. Placement convergence + // then removes the nodes outside the group's placement, one per tick, + // after a leadership transfer when the leader itself must leave. A + // removed node keeps its mounted replica, so membership decides. + wait_for( + "exactly one node replicates the collection's group", + Duration::from_secs(30), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .filter(|node| node.replicates_data_group(group_id)) + .count() + == 1 + }, + ) + .await; + let restorer = cluster + .nodes + .iter() + .find(|node| !node.replicates_data_group(group_id)) + .expect("a node that does not replicate the group"); + + let backup = drain_backup(&restorer.client).await; + restorer + .client + .simple_query(&format!( + "INSERT INTO {COLLECTION} (key, value) VALUES ('after', 'x')" + )) + .await + .unwrap_or_else(|e| panic!("insert: {}", db_detail(&e))); + assert!( + restorer.shared.tenant_write_mark(TENANT).is_none(), + "the restoring node applied no write of the tenant, so its memory holds no mark" + ); + + let error = push_restore(&restorer.client, backup) + .await + .expect_err("a write after the backup must refuse the restore on every node"); + assert!( + error.contains("restore refused"), + "expected the staleness refusal, got: {error}" + ); + assert!( + error.contains(COLLECTION), + "the refusal must name the collection of the newer write, got: {error}" + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_surrogate_replication.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_surrogate_replication.rs index 15caa51fd..70e4529d5 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cluster_surrogate_replication.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_surrogate_replication.rs @@ -43,9 +43,8 @@ fn surrogate_for_pk( let catalog = shared.credentials.catalog(); catalog .get_surrogate_for_pk( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection), TenantId::new(1), - collection, pk.as_bytes(), ) .ok() diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cross_node_read_occ_abort.rs b/nodedb-cluster-tests/tests/common_suite/cases/cross_node_read_occ_abort.rs index a473802e8..3c3818403 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cross_node_read_occ_abort.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cross_node_read_occ_abort.rs @@ -54,7 +54,7 @@ use crate::common; use std::sync::atomic::Ordering; use std::time::Duration; -use nodedb::types::{DatabaseId, VShardId}; +use nodedb::types::DatabaseId; use tokio_postgres::SimpleQueryMessage; use common::cluster_harness::{TestClusterNode, wait_for, wait_for_async}; @@ -84,14 +84,16 @@ fn admitted_total(node: &TestClusterNode) -> u64 { /// Three `document_schemaless` collection names whose vShard ids are pairwise /// distinct, so a transaction that writes two of them and reads the third is -/// genuinely multi-vShard. Deterministic: `VShardId::from_collection_in_database` +/// genuinely multi-vShard. Deterministic: `VShardId::from_collection` /// is a pure function of the database id + collection-name bytes, so the same /// scan picks the same names every run. fn distinct_vshard_triple() -> (String, String, String) { let mut chosen: Vec<(String, u32)> = Vec::new(); for i in 0u32..1024 { let name = format!("occ_shard_{i}"); - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let v = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if chosen.iter().all(|(_, cv)| *cv != v) { chosen.push((name, v)); if chosen.len() == 3 { diff --git a/nodedb-cluster-tests/tests/common_suite/cases/gateway_execute.rs b/nodedb-cluster-tests/tests/common_suite/cases/gateway_execute.rs index c9ee4a8ab..91c32d7c0 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/gateway_execute.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/gateway_execute.rs @@ -76,6 +76,7 @@ async fn gateway_execute_kv_put_get_single_node() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let put_checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; let put_result = gateway.execute(&ctx, put_checked).await; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/gather_join_in_txn_occ.rs b/nodedb-cluster-tests/tests/common_suite/cases/gather_join_in_txn_occ.rs index 470293ae1..593526524 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/gather_join_in_txn_occ.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/gather_join_in_txn_occ.rs @@ -65,7 +65,7 @@ use crate::common; use std::time::Duration; -use nodedb::types::{DatabaseId, VShardId}; +use nodedb::types::DatabaseId; use tokio_postgres::SimpleQueryMessage; use common::cluster_harness::{TestClusterNode, wait_for, wait_for_async}; @@ -76,13 +76,15 @@ use common::occ_shuffle::{ /// Four collection names whose vShard ids are pairwise distinct, so a transaction /// that reads two of them (the join sides) and writes the other two is genuinely /// multi-vShard on both its read set and its write set. Deterministic: -/// `VShardId::from_collection_in_database` is a pure function of the database id + +/// `VShardId::from_collection` is a pure function of the database id + /// collection-name bytes. fn distinct_vshard_quad() -> (String, String, String, String) { let mut chosen: Vec<(String, u32)> = Vec::new(); for i in 0u32..2048 { let name = format!("gather_join_occ_{i}"); - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let v = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if chosen.iter().all(|(_, cv)| *cv != v) { chosen.push((name, v)); if chosen.len() == 4 { diff --git a/nodedb-cluster-tests/tests/common_suite/cases/hilo_surrogate_uniqueness.rs b/nodedb-cluster-tests/tests/common_suite/cases/hilo_surrogate_uniqueness.rs index 4dfc93625..c72cca998 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/hilo_surrogate_uniqueness.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/hilo_surrogate_uniqueness.rs @@ -63,7 +63,10 @@ fn read_catalog_surrogates( ) -> Vec<(String, u32)> { let catalog = shared.credentials.catalog(); catalog - .scan_surrogates_for_collection(DatabaseId::DEFAULT, TenantId::new(1), collection) + .scan_surrogates_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection), + TenantId::new(1), + ) .unwrap_or_default() .into_iter() .map(|(pk_bytes, surrogate)| { diff --git a/nodedb-cluster-tests/tests/common_suite/cases/http_gateway_migration.rs b/nodedb-cluster-tests/tests/common_suite/cases/http_gateway_migration.rs index ef889768d..286c119a4 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/http_gateway_migration.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/http_gateway_migration.rs @@ -77,6 +77,7 @@ async fn http_gateway_migration_single_node_query() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let put_checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; let put_result = gateway.execute(&ctx, put_checked).await; @@ -156,6 +157,7 @@ async fn http_gateway_migration_cross_node_query() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let put_checked = common::authorize_gateway_plan(&follower.shared, &ctx, put_plan).await; let put_result = gateway.execute(&ctx, put_checked).await; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/kv_atomic_autocommit_replicates.rs b/nodedb-cluster-tests/tests/common_suite/cases/kv_atomic_autocommit_replicates.rs new file mode 100644 index 000000000..881c55de1 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/kv_atomic_autocommit_replicates.rs @@ -0,0 +1,207 @@ +// SPDX-License-Identifier: BUSL-1.1 +//! An autocommit SQL-function write reaches every replica. +//! +//! `KV_INCR` and `CREATE SORTED INDEX` build their `KvOp` by hand instead of +//! planning a statement. In cluster mode each one must be proposed through +//! the data group's Raft log like a planned write, so every replica applies +//! it. A write applied on the receiving node alone exists nowhere else. +//! +//! - `KV_INCR` runs on the counter's data-group leader, and the test kills +//! that leader. A survivor must read the incremented counter: had the write +//! applied on the leader alone, it died with it. +//! - `CREATE SORTED INDEX` builds a tree on the core that owns the rows. A +//! sorted-index read runs on the node that receives it, so a count read on +//! each follower reads that follower's own tree. + +use crate::common; +use common::cluster_harness::TestCluster; + +use std::time::{Duration, Instant}; + +use nodedb::types::DatabaseId; + +const COUNTERS: &str = "repl_kv_ctr"; +const BOARD: &str = "repl_kv_board"; +const INDEX: &str = "repl_kv_board_idx"; + +fn pg_detail(e: &tokio_postgres::Error) -> String { + match e.as_db_error() { + Some(db) => format!("{}: {}", db.code().code(), db.message()), + None => format!("{e}"), + } +} + +/// The first column of the first row `sql` returns, or the error it raised. +async fn first_cell(client: &tokio_postgres::Client, sql: &str) -> Result, String> { + let rows = client.simple_query(sql).await.map_err(|e| pg_detail(&e))?; + Ok(rows.into_iter().find_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(r) => r.get(0).map(str::to_string), + _ => None, + })) +} + +/// Whether a `SORTED_COUNT` read returned a count of three. +fn counts_three(read: &Result, String>) -> bool { + matches!(read, Ok(Some(doc)) if doc.replace(' ', "").contains("\"count\":3")) +} + +/// The leader node id of the data group that owns `collection`. +fn group_leader(cluster: &TestCluster, collection: &str) -> u64 { + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection).vshard(); + let routing = cluster.nodes[0] + .shared + .cluster_routing + .as_ref() + .expect("cluster_routing") + .read() + .unwrap_or_else(|p| p.into_inner()); + let group = routing + .group_for_vshard(vshard.as_u32()) + .expect("the collection's vShard maps to a data group"); + routing + .group_info(group) + .map(|info| info.leader) + .unwrap_or(0) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn an_autocommit_kv_incr_survives_its_leader() { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {COUNTERS} (key TEXT PRIMARY KEY, n INT) WITH (engine='kv')" + )) + .await + .expect("create the counter collection"); + cluster.nodes[0] + .client + .simple_query(&format!( + "INSERT INTO {COUNTERS} (key, n) VALUES ('ctr', 5)" + )) + .await + .unwrap_or_else(|e| panic!("seed the counter: {}", pg_detail(&e))); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + + let leader_id = group_leader(&cluster, COUNTERS); + assert_ne!(leader_id, 0, "the counter's data group has no leader"); + let leader = cluster + .nodes + .iter() + .find(|n| n.node_id == leader_id) + .expect("the leader node is in the cluster"); + let incremented = first_cell( + &leader.client, + &format!("SELECT KV_INCR('{COUNTERS}', 'ctr', 3)"), + ) + .await + .unwrap_or_else(|e| panic!("KV_INCR on the group leader: {e}")); + assert!( + incremented.as_deref().is_some_and(|doc| doc.contains('8')), + "the live KV_INCR returns 5 + 3: {incremented:?}" + ); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + + let mut nodes = cluster.nodes; + let leader_idx = nodes + .iter() + .position(|n| n.node_id == leader_id) + .expect("leader node present"); + nodes.remove(leader_idx).shutdown().await; + + let read = format!("SELECT n FROM {COUNTERS} WHERE key = 'ctr'"); + for node in &nodes { + let deadline = Instant::now() + Duration::from_secs(30); + let mut last = Err(String::from("never read")); + while Instant::now() < deadline { + last = first_cell(&node.client, &read).await; + if matches!(&last, Ok(Some(n)) if n == "8") { + break; + } + tokio::time::sleep(Duration::from_millis(200)).await; + } + assert_eq!( + last, + Ok(Some("8".to_string())), + "survivor node {} must read the counter the autocommit KV_INCR moved; 5 means \ + the increment applied on the killed leader alone", + node.node_id + ); + } + + for node in nodes { + node.shutdown().await; + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_sorted_index_is_built_on_every_replica() { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {BOARD} (k TEXT PRIMARY KEY, score INT) WITH (engine='kv')" + )) + .await + .expect("create the board collection"); + for (key, score) in [("p0", 10), ("p1", 20), ("p2", 30)] { + cluster.nodes[0] + .client + .simple_query(&format!( + "INSERT INTO {BOARD} (k, score) VALUES ('{key}', {score})" + )) + .await + .unwrap_or_else(|e| panic!("insert {key}: {}", pg_detail(&e))); + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + + let leader_id = group_leader(&cluster, BOARD); + assert_ne!(leader_id, 0, "the board's data group has no leader"); + let leader = cluster + .nodes + .iter() + .find(|n| n.node_id == leader_id) + .expect("the leader node is in the cluster"); + leader + .client + .simple_query(&format!( + "CREATE SORTED INDEX {INDEX} ON {BOARD} (score DESC) KEY k" + )) + .await + .unwrap_or_else(|e| panic!("CREATE SORTED INDEX: {}", pg_detail(&e))); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + + let read = format!("SELECT SORTED_COUNT({INDEX})"); + for node in cluster.nodes.iter().filter(|n| n.node_id != leader_id) { + let deadline = Instant::now() + Duration::from_secs(30); + let mut last = Err(String::from("never read")); + while Instant::now() < deadline { + last = first_cell(&node.client, &read).await; + if counts_three(&last) { + break; + } + tokio::time::sleep(Duration::from_millis(200)).await; + } + assert!( + counts_three(&last), + "follower node {} must count every row in its own replica of the index tree; \ + a missing tree means the registration applied on the receiving node alone \ + (last read: {last:?})", + node.node_id + ); + } + + for node in cluster.nodes { + node.shutdown().await; + } +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/listeners_gateway_smoke.rs b/nodedb-cluster-tests/tests/common_suite/cases/listeners_gateway_smoke.rs index 05c97b55c..74c3a0104 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/listeners_gateway_smoke.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/listeners_gateway_smoke.rs @@ -78,6 +78,7 @@ async fn pgwire_gateway_smoke_cache_hit() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; gateway.execute(&ctx, checked).await.expect("gateway Put"); @@ -148,6 +149,7 @@ async fn http_gateway_smoke_cache_hit() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; gateway.execute(&ctx, checked).await.expect("gateway Put"); @@ -212,6 +214,7 @@ async fn resp_gateway_smoke_cache_hit() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; gateway.execute(&ctx, checked).await.expect("gateway Put"); @@ -279,6 +282,7 @@ async fn ilp_gateway_smoke_cache_hit() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; gateway.execute(&ctx, checked).await.expect("gateway Put"); @@ -343,6 +347,7 @@ async fn native_gateway_smoke_cache_hit() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; gateway.execute(&ctx, checked).await.expect("gateway Put"); diff --git a/nodedb-cluster-tests/tests/common_suite/cases/listeners_typed_not_leader.rs b/nodedb-cluster-tests/tests/common_suite/cases/listeners_typed_not_leader.rs index 54bb5ae6c..9ab4585ae 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/listeners_typed_not_leader.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/listeners_typed_not_leader.rs @@ -179,6 +179,7 @@ async fn pgwire_not_leader_retry_uses_shared_gateway() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let ctx = test_ctx(); let checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; @@ -244,6 +245,7 @@ async fn http_not_leader_gateway_error_mapping() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let ctx = test_ctx(); let checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; @@ -315,6 +317,7 @@ async fn resp_not_leader_gateway_error_mapping() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let ctx = test_ctx(); let checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; @@ -449,6 +452,7 @@ async fn native_not_leader_gateway_error_mapping() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let ctx = test_ctx(); let checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; @@ -459,7 +463,8 @@ async fn native_not_leader_gateway_error_mapping() { assert_eq!(node.not_leader_retry_count(), 0); - // Error-mapping proof: GatewayErrorMap::to_native maps NotLeader to code 40. + // Error-mapping proof: GatewayErrorMap::to_native maps NotLeader to the + // public NOT_LEADER code. let not_leader = Error::NotLeader { vshard_id: VShardId::new(0), leader_node: 1, @@ -467,8 +472,9 @@ async fn native_not_leader_gateway_error_mapping() { }; let (native_code, _native_msg) = GatewayErrorMap::to_native(¬_leader); assert_eq!( - native_code, 10, - "NotLeader must map to native error code 10 (CODE_NOT_LEADER)" + native_code, + nodedb::ErrorCode::NOT_LEADER, + "NotLeader must map to the public NOT_LEADER code" ); node.shutdown().await; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_cross_core.rs b/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_cross_core.rs index e443d69c8..41663c576 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_cross_core.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_cross_core.rs @@ -27,7 +27,7 @@ use common::cluster_harness::{TestClusterNode, wait_for}; use std::time::Duration; -use nodedb::types::{DatabaseId, VShardId}; +use nodedb::types::DatabaseId; use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; const SOURCE: &str = "xc_entries"; @@ -50,8 +50,8 @@ fn pg_detail(e: &tokio_postgres::Error) -> String { #[test] fn source_and_target_home_to_different_vshards() { assert_ne!( - VShardId::from_collection_in_database(DatabaseId::DEFAULT, SOURCE), - VShardId::from_collection_in_database(DatabaseId::DEFAULT, TARGET), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, SOURCE).vshard(), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, TARGET).vshard(), "this file tests the CROSS-SHARD path; '{SOURCE}' and '{TARGET}' must not be co-resident" ); } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_cross_shard.rs b/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_cross_shard.rs index 0eed4c862..d860d3442 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_cross_shard.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_cross_shard.rs @@ -19,7 +19,7 @@ use common::cluster_harness::TestCluster; use std::time::Duration; -use nodedb::types::{DatabaseId, VShardId}; +use nodedb::types::DatabaseId; /// Source and target, chosen for readability rather than for their hashes — the /// homing assertion below is what makes the choice meaningful. @@ -84,8 +84,8 @@ async fn declare_binding(cluster: &TestCluster) { /// vShard, so every balance below travels on its own task. #[test] fn source_and_target_home_to_different_vshards() { - let source = VShardId::from_collection_in_database(DatabaseId::DEFAULT, SOURCE); - let target = VShardId::from_collection_in_database(DatabaseId::DEFAULT, TARGET); + let source = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, SOURCE).vshard(); + let target = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, TARGET).vshard(); assert_ne!( source, target, "this file tests the CROSS-SHARD path; '{SOURCE}' and '{TARGET}' must not be co-resident" diff --git a/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_replication.rs b/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_replication.rs index 599afe0e8..9a5864bcc 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_replication.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_replication.rs @@ -26,7 +26,7 @@ use common::cluster_harness::TestCluster; use std::time::Duration; -use nodedb::types::{DatabaseId, VShardId}; +use nodedb::types::DatabaseId; /// Cross-shard fixture: source and target hash to different vShards. const XS_SOURCE: &str = "rep_entries"; @@ -204,8 +204,8 @@ async fn assert_every_replica_agrees( #[test] fn coresident_fixture_shares_one_vshard() { assert_eq!( - VShardId::from_collection_in_database(DatabaseId::DEFAULT, CO_SOURCE), - VShardId::from_collection_in_database(DatabaseId::DEFAULT, CO_TARGET), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, CO_SOURCE).vshard(), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, CO_TARGET).vshard(), "this fixture must exercise the fold that runs inside the source write's own \ transaction" ); @@ -215,8 +215,8 @@ fn coresident_fixture_shares_one_vshard() { #[test] fn replication_fixture_is_cross_shard() { assert_ne!( - VShardId::from_collection_in_database(DatabaseId::DEFAULT, XS_SOURCE), - VShardId::from_collection_in_database(DatabaseId::DEFAULT, XS_TARGET), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, XS_SOURCE).vshard(), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, XS_TARGET).vshard(), "this fixture must exercise the replicated cross-shard balance write" ); } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/mod.rs b/nodedb-cluster-tests/tests/common_suite/cases/mod.rs index bd7e362ef..9fc78978a 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/mod.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/mod.rs @@ -25,7 +25,9 @@ mod catalog_put_if_absent; mod checkpoint_cross_node; mod cluster_array; mod cluster_array_cell_raft_replication; +mod cluster_backup_remote_cut; mod cluster_backup_restore; +mod cluster_backup_restore_databases; mod cluster_backup_restore_engines; mod cluster_cdc_publish_once; mod cluster_collection_hard_delete; @@ -34,6 +36,8 @@ mod cluster_epoch_self_fence; mod cluster_execute_request; mod cluster_partition_strategy_replication; mod cluster_post_apply_follower_dispatch; +mod cluster_restore_documents_restart; +mod cluster_restore_refuses_on_non_replica; mod cluster_surrogate_replication; mod column_stats_cross_node; mod constraint_delivery; @@ -55,6 +59,7 @@ mod http_gateway_migration; mod ilp_gateway_migration; mod install_snapshot_crdt_constraints_cluster; mod install_snapshot_e2e_cluster; +mod kv_atomic_autocommit_replicates; mod learner_cleanup; mod linearizable_read_leadership; mod listeners_gateway_smoke; @@ -69,6 +74,7 @@ mod node_labels_replicate_to_followers; mod pgwire_gateway_migration; mod planner_local_only; mod prepared_cache_invalidation; +mod proposal_committed_twice_applies_once; mod resp_gateway_migration; mod retention_policy_cross_node; mod scope_quota_cross_node; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/multi_replica_data_groups.rs b/nodedb-cluster-tests/tests/common_suite/cases/multi_replica_data_groups.rs index 5e4f7f45a..77e6464dd 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/multi_replica_data_groups.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/multi_replica_data_groups.rs @@ -121,7 +121,9 @@ async fn data_group_is_multi_replica_and_survives_leader_loss() { .await; // Resolve the collection's data group. - let vshard = nodedb_cluster::routing::vshard_for_collection(DatabaseId::DEFAULT, COLL); + let vshard = nodedb_cluster::routing::vshard_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, COLL), + ); let group_id = { let routing = cluster.nodes[0] .shared @@ -213,3 +215,246 @@ async fn data_group_is_multi_replica_and_survives_leader_loss() { node.shutdown().await; } } + +const TXN_COLL: &str = "mr_txn_commit"; + +/// Rows every replica holds once the explicit transaction commits. +const TXN_EXPECTED: [(&str, &str); 4] = [ + ("seed-0", "updated"), + ("seed-1", "seed"), + ("txn-0", "inserted-0"), + ("txn-1", "inserted-1"), +]; + +/// Leader of `group_id` as `node`'s routing table records it, or `0`. +fn routing_leader(node: &common::cluster_harness::TestClusterNode, group_id: u64) -> u64 { + let routing = node + .shared + .cluster_routing + .as_ref() + .expect("cluster_routing") + .read() + .unwrap_or_else(|p| p.into_inner()); + routing.group_info(group_id).map(|i| i.leader).unwrap_or(0) +} + +/// True when `node` leads `group_id` by its own Raft state. +fn leads(node: &common::cluster_harness::TestClusterNode, group_id: u64) -> bool { + node.all_group_leaders().contains(&(group_id, node.node_id)) +} + +/// Index of the node that leads `group_id` by its own Raft state and by every +/// node's routing table. +async fn group_leader_index( + nodes: &[common::cluster_harness::TestClusterNode], + group_id: u64, +) -> usize { + let deadline = Instant::now() + Duration::from_secs(20); + loop { + let found = nodes.iter().position(|n| { + leads(n, group_id) + && nodes + .iter() + .all(|m| routing_leader(m, group_id) == n.node_id) + }); + if let Some(idx) = found { + return idx; + } + if Instant::now() >= deadline { + panic!("data group {group_id} has no leader agreed by every node within 20s"); + } + tokio::time::sleep(Duration::from_millis(100)).await; + } +} + +/// This node's local document entries for `TXN_COLL`, keyed by storage key. +async fn local_txn_documents( + node: &common::cluster_harness::TestClusterNode, +) -> std::collections::BTreeMap> { + let bytes = node + .create_tenant_snapshot(nodedb_types::TenantId::new(1)) + .await; + assert!( + !bytes.is_empty(), + "node {} returned an empty tenant snapshot", + node.node_id + ); + let snapshot: nodedb::types::TenantDataSnapshot = + zerompk::from_msgpack(&bytes).expect("decode TenantDataSnapshot"); + let marker = format!(":{TXN_COLL}:"); + snapshot + .documents + .into_iter() + .filter(|(key, _)| key.contains(&marker)) + .collect() +} + +/// `(id, payload)` rows of `TXN_COLL` served through `client`, sorted by id. +async fn served_txn_rows(client: &tokio_postgres::Client) -> Result, String> { + let msgs = client + .simple_query(&format!("SELECT id, payload FROM {TXN_COLL}")) + .await + .map_err(|e| pg_detail(&e))?; + let mut rows: Vec<(String, String)> = msgs + .iter() + .filter_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(r) => Some(( + r.get("id").unwrap_or_default().to_owned(), + r.get("payload").unwrap_or_default().to_owned(), + )), + _ => None, + }) + .collect(); + rows.sort(); + Ok(rows) +} + +/// Spawn 3 nodes, seed `TXN_COLL`, and commit one single-shard explicit +/// transaction on the node that leads the collection's data group. Returns +/// the cluster, the group id, and the leader's index. +async fn commit_single_shard_txn_on_group_leader() -> (TestCluster, u64, usize) { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {TXN_COLL} WITH (engine='document_schemaless')" + )) + .await + .expect("CREATE COLLECTION"); + for id in ["seed-0", "seed-1"] { + cluster.nodes[0] + .client + .simple_query(&format!( + "INSERT INTO {TXN_COLL} (id, payload) VALUES ('{id}', 'seed')" + )) + .await + .unwrap_or_else(|e| panic!("seed {id}: {}", pg_detail(&e))); + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + + let group_id = cluster.nodes[0] + .group_id_for_collection(TXN_COLL) + .expect("collection vshard mapped to a group"); + let leader_idx = group_leader_index(&cluster.nodes, group_id).await; + let leader = &cluster.nodes[leader_idx]; + + // One vShard, run on its leader: the commit takes the local single-shard path. + leader + .client + .simple_query(&format!( + "BEGIN; \ + INSERT INTO {TXN_COLL} (id, payload) VALUES ('txn-0', 'inserted-0'); \ + INSERT INTO {TXN_COLL} (id, payload) VALUES ('txn-1', 'inserted-1'); \ + UPDATE {TXN_COLL} SET payload = 'updated' WHERE id = 'seed-0'; \ + COMMIT" + )) + .await + .unwrap_or_else(|e| panic!("single-shard COMMIT on the leader: {}", pg_detail(&e))); + assert!( + leads(leader, group_id), + "node {} lost leadership of group {group_id} during the commit; \ + the commit did not run on the group leader", + leader.node_id + ); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + (cluster, group_id, leader_idx) +} + +/// A single-shard explicit transaction committed on the data-group leader +/// reaches every replica's local state. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn single_shard_txn_committed_on_group_leader_reaches_every_replica() { + let (cluster, _group_id, leader_idx) = commit_single_shard_txn_on_group_leader().await; + let leader = &cluster.nodes[leader_idx]; + + let expected: Vec<(String, String)> = TXN_EXPECTED + .iter() + .map(|(id, p)| ((*id).to_owned(), (*p).to_owned())) + .collect(); + let served = served_txn_rows(&leader.client) + .await + .expect("read committed rows on the leader"); + assert_eq!(served, expected, "the leader must serve the committed rows"); + let leader_docs = local_txn_documents(leader).await; + assert_eq!( + leader_docs.len(), + TXN_EXPECTED.len(), + "the leader must hold every committed row locally" + ); + + for follower in cluster.nodes.iter().filter(|n| n.node_id != leader.node_id) { + let deadline = Instant::now() + Duration::from_secs(15); + let follower_docs = loop { + let docs = local_txn_documents(follower).await; + if docs == leader_docs || Instant::now() >= deadline { + break docs; + } + tokio::time::sleep(Duration::from_millis(200)).await; + }; + let missing: Vec<&String> = leader_docs + .keys() + .filter(|k| follower_docs.get(*k) != leader_docs.get(*k)) + .collect(); + assert!( + missing.is_empty() && follower_docs.len() == leader_docs.len(), + "follower {} local state differs from leader {} after the committed \ + transaction; rows missing or different on the follower: {missing:?}", + follower.node_id, + leader.node_id + ); + } + + for node in cluster.nodes { + node.shutdown().await; + } +} + +/// A single-shard explicit transaction committed on the data-group leader +/// survives the loss of that leader. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn single_shard_txn_committed_on_group_leader_survives_leader_loss() { + let (cluster, group_id, leader_idx) = commit_single_shard_txn_on_group_leader().await; + + let mut nodes = cluster.nodes; + nodes.remove(leader_idx).shutdown().await; + + // A survivor takes over the data group. + let deadline = Instant::now() + Duration::from_secs(20); + while !nodes.iter().any(|n| leads(n, group_id)) { + if Instant::now() >= deadline { + panic!("no survivor took over data group {group_id} within 20s"); + } + tokio::time::sleep(Duration::from_millis(100)).await; + } + + let expected: Vec<(String, String)> = TXN_EXPECTED + .iter() + .map(|(id, p)| ((*id).to_owned(), (*p).to_owned())) + .collect(); + for node in &nodes { + let deadline = Instant::now() + Duration::from_secs(20); + let served = loop { + match served_txn_rows(&node.client).await { + Ok(rows) => break rows, + Err(e) if Instant::now() >= deadline => { + panic!("survivor {} could not serve rows: {e}", node.node_id) + } + Err(_) => tokio::time::sleep(Duration::from_millis(150)).await, + } + }; + assert_eq!( + served, expected, + "survivor {} must serve the committed transaction after the leader is lost", + node.node_id + ); + } + + for node in nodes { + node.shutdown().await; + } +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/native_gateway_migration.rs b/nodedb-cluster-tests/tests/common_suite/cases/native_gateway_migration.rs index b9c9a820c..ecb671784 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/native_gateway_migration.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/native_gateway_migration.rs @@ -7,8 +7,8 @@ //! assert rows returned. //! 2. **Cross-node SELECT** — 3-node cluster, gateway on follower routes a //! KV GET to the leaseholder; asserts success. -//! 3. **Typed error → native code** — trigger `CollectionNotFound`, assert the -//! native error code matches `GatewayErrorMap::to_native` mapping (code 40). +//! 3. **Typed error → native code** — map each error variant through +//! `GatewayErrorMap::to_native` and assert its stable `nodedb_types` code. use crate::common; @@ -22,6 +22,7 @@ use nodedb::control::gateway::core::QueryContext; use nodedb::types::{RequestId, TenantId, VShardId}; use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; use nodedb_types::QualifiedCollection; +use nodedb_types::error::ErrorCode; use common::cluster_harness::{TestCluster, TestClusterNode}; @@ -77,6 +78,7 @@ async fn native_gateway_migration_single_node_select() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let put_checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; gateway @@ -147,6 +149,7 @@ async fn native_gateway_migration_cross_node_select() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let put_checked = common::authorize_gateway_plan(&cluster.nodes[0].shared, &ctx, put_plan).await; @@ -186,21 +189,17 @@ async fn native_gateway_migration_cross_node_select() { // Test 3: Typed error → native code mapping // --------------------------------------------------------------------------- // -// `GatewayErrorMap::to_native` maps each error variant to a numeric code. -// The migrated `direct_ops.rs` and `sql_gateway.rs` call this mapper. -// These tests verify the codes align with the constants defined in error_map.rs. +// `GatewayErrorMap::to_native` returns the stable `nodedb_types` error code +// and the message the native error frame carries for each error variant. #[test] -fn native_gateway_error_collection_not_found_is_code_40() { +fn native_gateway_error_collection_not_found_code() { let err = Error::CollectionNotFound { tenant_id: TenantId::new(0), collection: "missing_native_col".into(), }; let (code, msg) = GatewayErrorMap::to_native(&err); - assert_eq!( - code, 40, - "CollectionNotFound should map to code 40, got {code}" - ); + assert_eq!(code, ErrorCode::COLLECTION_NOT_FOUND, "got {code}"); assert!( msg.contains("missing_native_col"), "error message should name the collection: {msg}" @@ -208,14 +207,14 @@ fn native_gateway_error_collection_not_found_is_code_40() { } #[test] -fn native_gateway_error_not_leader_is_code_10() { +fn native_gateway_error_not_leader_code() { let err = Error::NotLeader { vshard_id: VShardId::new(1), leader_node: 2, leader_addr: "10.0.0.1:9000".into(), }; let (code, msg) = GatewayErrorMap::to_native(&err); - assert_eq!(code, 10, "NotLeader should map to code 10, got {code}"); + assert_eq!(code, ErrorCode::NOT_LEADER, "got {code}"); assert!( msg.contains("hint:"), "not-leader message should contain hint: {msg}" @@ -223,50 +222,31 @@ fn native_gateway_error_not_leader_is_code_10() { } #[test] -fn native_gateway_error_deadline_is_code_20() { +fn native_gateway_error_deadline_code() { let err = Error::DeadlineExceeded { request_id: RequestId::new(1), }; let (code, _msg) = GatewayErrorMap::to_native(&err); - assert_eq!( - code, 20, - "DeadlineExceeded should map to code 20, got {code}" - ); + assert_eq!(code, ErrorCode::DEADLINE_EXCEEDED, "got {code}"); } #[test] -fn native_gateway_error_schema_changed_is_code_30() { - let err = Error::RetryableSchemaChanged { - descriptor: "users".into(), - }; - let (code, msg) = GatewayErrorMap::to_native(&err); - assert_eq!( - code, 30, - "RetryableSchemaChanged should map to code 30, got {code}" - ); - assert!( - msg.contains("users"), - "message should name descriptor: {msg}" - ); -} - -#[test] -fn native_gateway_error_authz_is_code_50() { +fn native_gateway_error_authz_code() { let err = Error::RejectedAuthz { tenant_id: TenantId::new(0), resource: "secret".into(), }; let (code, _msg) = GatewayErrorMap::to_native(&err); - assert_eq!(code, 50, "RejectedAuthz should map to code 50, got {code}"); + assert_eq!(code, ErrorCode::AUTHORIZATION_DENIED, "got {code}"); } #[test] -fn native_gateway_error_bad_request_is_code_60() { +fn native_gateway_error_bad_request_code() { let err = Error::BadRequest { detail: "invalid plan".into(), }; let (code, msg) = GatewayErrorMap::to_native(&err); - assert_eq!(code, 60, "BadRequest should map to code 60, got {code}"); + assert_eq!(code, ErrorCode::BAD_REQUEST, "got {code}"); assert!( msg.contains("invalid plan"), "message should contain detail: {msg}" @@ -274,24 +254,21 @@ fn native_gateway_error_bad_request_is_code_60() { } #[test] -fn native_gateway_error_constraint_is_code_70() { +fn native_gateway_error_constraint_code() { let err = Error::RejectedConstraint { detail: "unique violation".into(), constraint: "pk".into(), collection: "orders".into(), }; let (code, _msg) = GatewayErrorMap::to_native(&err); - assert_eq!( - code, 70, - "RejectedConstraint should map to code 70, got {code}" - ); + assert_eq!(code, ErrorCode::CONSTRAINT_VIOLATION, "got {code}"); } #[test] -fn native_gateway_error_internal_is_code_99() { +fn native_gateway_error_internal_code() { let err = Error::Internal { detail: "unexpected state".into(), }; let (code, _msg) = GatewayErrorMap::to_native(&err); - assert_eq!(code, 99, "Internal should map to code 99, got {code}"); + assert_eq!(code, ErrorCode::INTERNAL, "got {code}"); } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/native_gather_join_in_txn_occ.rs b/nodedb-cluster-tests/tests/common_suite/cases/native_gather_join_in_txn_occ.rs index 8fd90192d..a77327353 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/native_gather_join_in_txn_occ.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/native_gather_join_in_txn_occ.rs @@ -33,7 +33,7 @@ use crate::common; use std::time::Duration; -use nodedb::types::{DatabaseId, VShardId}; +use nodedb::types::DatabaseId; use nodedb_client::NativeClient; use nodedb_client::native::pool::PoolConfig; use nodedb_types::error::NodeDbError; @@ -48,13 +48,15 @@ const SERIALIZATION_ABORT: &str = "could not serialize access due to concurrent /// Four collection names whose vShard ids are pairwise distinct, so a transaction /// that reads two of them (the join sides) and writes the other two is genuinely /// multi-vShard on both its read set and its write set. Deterministic: -/// `VShardId::from_collection_in_database` is a pure function of the database id + +/// `VShardId::from_collection` is a pure function of the database id + /// collection-name bytes. fn distinct_vshard_quad() -> (String, String, String, String) { let mut chosen: Vec<(String, u32)> = Vec::new(); for i in 0u32..2048 { let name = format!("native_gather_join_occ_{i}"); - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let v = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if chosen.iter().all(|(_, cv)| *cv != v) { chosen.push((name, v)); if chosen.len() == 4 { diff --git a/nodedb-cluster-tests/tests/common_suite/cases/pgwire_gateway_migration.rs b/nodedb-cluster-tests/tests/common_suite/cases/pgwire_gateway_migration.rs index d02c841b5..8a3b46cdb 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/pgwire_gateway_migration.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/pgwire_gateway_migration.rs @@ -191,6 +191,7 @@ async fn pgwire_gateway_migration_plan_cache_hits() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let put_checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; gateway diff --git a/nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs b/nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs new file mode 100644 index 000000000..cd1a628f3 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs @@ -0,0 +1,167 @@ +// SPDX-License-Identifier: BUSL-1.1 +//! A proposal whose bytes commit twice applies once on every replica. +//! +//! A proposer re-proposes the same entry bytes after `RetryableLeaderChange`. +//! When the first copy also committed, the data group's log holds two copies +//! with one `idempotency_key`. The apply loop recognises the second copy by +//! that key and skips it. +//! +//! The test commits the same `KV_INCR` entry twice through the group leader's +//! raw proposer, which is exactly the log a double commit leaves behind. A +//! delta applied twice moves the counter twice, so every replica must read +//! the counter moved once. + +use crate::common; +use common::cluster_harness::TestCluster; + +use std::time::{Duration, Instant}; + +use nodedb::control::wal_replication::{ReplicableWrite, to_replicated_entry}; +use nodedb::types::{DatabaseId, TenantId}; +use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; + +const COLL: &str = "dup_proposal_ctr"; +const TENANT: u64 = 1; + +fn pg_detail(e: &tokio_postgres::Error) -> String { + match e.as_db_error() { + Some(db) => format!("{}: {}", db.code().code(), db.message()), + None => format!("{e}"), + } +} + +async fn counter_on(client: &tokio_postgres::Client) -> Option { + let rows = client + .simple_query(&format!("SELECT n FROM {COLL} WHERE key = 'ctr'")) + .await + .unwrap_or_else(|e| panic!("read counter: {}", pg_detail(&e))); + rows.into_iter().find_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(r) => r.get(0).map(str::to_string), + _ => None, + }) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_proposal_committed_twice_moves_the_counter_once() { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {COLL} (key TEXT PRIMARY KEY, n INT) WITH (engine='kv')" + )) + .await + .expect("create the counter collection"); + cluster.nodes[0] + .client + .simple_query(&format!("INSERT INTO {COLL} (key, n) VALUES ('ctr', 5)")) + .await + .unwrap_or_else(|e| panic!("seed the counter: {}", pg_detail(&e))); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, COLL).vshard(); + let deadline = Instant::now() + Duration::from_secs(20); + let mut committed = 0; + let mut entry_bytes: Option> = None; + while committed < 2 { + assert!( + Instant::now() < deadline, + "could not commit both copies of the proposal through the group leader" + ); + let leader_id = { + let routing = cluster.nodes[0] + .shared + .cluster_routing + .as_ref() + .expect("cluster_routing") + .read() + .unwrap_or_else(|p| p.into_inner()); + let group = routing + .group_for_vshard(vshard.as_u32()) + .expect("the counter's vShard maps to a data group"); + routing + .group_info(group) + .map(|info| info.leader) + .unwrap_or(0) + }; + let Some(leader) = cluster.nodes.iter().find(|n| n.node_id == leader_id) else { + tokio::time::sleep(Duration::from_millis(100)).await; + continue; + }; + // One entry, built once: both copies carry the same bytes and so the + // same idempotency key, like a re-proposal after a leader change. + let bytes = match &entry_bytes { + Some(bytes) => bytes.clone(), + None => { + let surrogate = leader + .shared + .surrogate_assigner + .assign( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, COLL), + TenantId::new(TENANT), + b"ctr", + ) + .expect("the seeded key has a surrogate"); + let plan = PhysicalPlan::Kv(KvOp::Incr { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, COLL), + key: b"ctr".to_vec(), + delta: 3, + ttl_ms: 0, + surrogate, + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + // The key is seeded before this proposal, so the shape an + // absent key takes never applies. + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, + }); + let write = + ReplicableWrite::decide_for_replication(&plan).expect("KV_INCR replicates"); + let entry = + to_replicated_entry(TenantId::new(TENANT), DatabaseId::DEFAULT, vshard, &write) + .expect("encode the proposal") + .expect("KV_INCR encodes to a replicated entry"); + let bytes = entry.to_bytes(); + entry_bytes = Some(bytes.clone()); + bytes + } + }; + let proposer = leader + .shared + .raft_proposer + .get() + .expect("the leader has a raft proposer"); + match proposer(vshard.as_u32(), bytes) { + Ok(_) => committed += 1, + Err(_) => tokio::time::sleep(Duration::from_millis(100)).await, + } + } + + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + + for node in &cluster.nodes { + let deadline = Instant::now() + Duration::from_secs(15); + let mut last = None; + while Instant::now() < deadline { + last = counter_on(&node.client).await; + if last.as_deref() != Some("5") { + break; + } + tokio::time::sleep(Duration::from_millis(100)).await; + } + assert_eq!( + last.as_deref(), + Some("8"), + "node {} must hold the counter moved once by the proposal committed twice \ + (5 + 3); 11 means the second copy applied again", + node.node_id + ); + } + + for node in cluster.nodes { + node.shutdown().await; + } +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/resp_gateway_migration.rs b/nodedb-cluster-tests/tests/common_suite/cases/resp_gateway_migration.rs index 68bfecdae..2a77900cc 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/resp_gateway_migration.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/resp_gateway_migration.rs @@ -74,6 +74,7 @@ async fn resp_gateway_migration_single_node_set_get() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let put_checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; let put_result = gateway.execute(&ctx, put_checked).await; @@ -146,6 +147,7 @@ async fn resp_gateway_migration_cross_node_get() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let put_checked = common::authorize_gateway_plan(&cluster.nodes[0].shared, &ctx, put_plan).await; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_aggregate_in_txn_occ.rs b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_aggregate_in_txn_occ.rs index 0623c8712..3ee2bdb35 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_aggregate_in_txn_occ.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_aggregate_in_txn_occ.rs @@ -57,7 +57,7 @@ use crate::common; use std::time::Duration; -use nodedb::types::{DatabaseId, VShardId}; +use nodedb::types::DatabaseId; use common::cluster_harness::{TestClusterNode, wait_for, wait_for_async}; use common::occ_shuffle::{ @@ -66,13 +66,15 @@ use common::occ_shuffle::{ /// Three `metrics`/`w1`/`w2` collection names whose vShard ids are pairwise /// distinct, so a transaction that writes two of them and reads the third is -/// genuinely multi-vShard. Deterministic: `VShardId::from_collection_in_database` +/// genuinely multi-vShard. Deterministic: `VShardId::from_collection` /// is a pure function of the database id + collection-name bytes. fn distinct_vshard_triple() -> (String, String, String) { let mut chosen: Vec<(String, u32)> = Vec::new(); for i in 0u32..1024 { let name = format!("shuffle_occ_{i}"); - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let v = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if chosen.iter().all(|(_, cv)| *cv != v) { chosen.push((name, v)); if chosen.len() == 3 { diff --git a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_cost_model.rs b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_cost_model.rs index 8d3055ec3..3c24f1c9d 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_cost_model.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_cost_model.rs @@ -67,8 +67,14 @@ async fn cost_model_auto_selects_shuffle_from_analyze_stats() { const LEFT: &str = "orders"; const RIGHT: &str = "customers"; assert_ne!( - vshard_for_collection(DatabaseId::DEFAULT, LEFT), - vshard_for_collection(DatabaseId::DEFAULT, RIGHT), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + LEFT + )), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + RIGHT + )), "test collections must hash to different vShards to exercise cross-node shuffle" ); diff --git a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_end_to_end.rs b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_end_to_end.rs index 9939abf38..e2e29d62e 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_end_to_end.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_end_to_end.rs @@ -66,8 +66,14 @@ async fn distributed_shuffle_join_matches_inner_join() { const LEFT: &str = "orders"; const RIGHT: &str = "customers"; assert_ne!( - vshard_for_collection(DatabaseId::DEFAULT, LEFT), - vshard_for_collection(DatabaseId::DEFAULT, RIGHT), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + LEFT + )), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + RIGHT + )), "test collections must hash to different vShards to exercise cross-node shuffle" ); @@ -230,8 +236,14 @@ async fn a_shuffle_join_renders_a_time_key_as_the_stored_instant() { const LEFT: &str = "ts_shuffle_events"; const RIGHT: &str = "ts_shuffle_hosts"; assert_ne!( - vshard_for_collection(DatabaseId::DEFAULT, LEFT), - vshard_for_collection(DatabaseId::DEFAULT, RIGHT), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + LEFT + )), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + RIGHT + )), "test collections must hash to different vShards to exercise cross-node shuffle" ); diff --git a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_in_txn_occ.rs b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_in_txn_occ.rs index 86ae24058..4a96ac84c 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_in_txn_occ.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_in_txn_occ.rs @@ -57,7 +57,7 @@ use crate::common; use std::time::Duration; -use nodedb::types::{DatabaseId, VShardId}; +use nodedb::types::DatabaseId; use common::cluster_harness::{TestClusterNode, wait_for, wait_for_async}; use common::occ_shuffle::{ @@ -67,13 +67,15 @@ use common::occ_shuffle::{ /// Four collection names whose vShard ids are pairwise distinct, so a transaction /// that reads two of them (the join sides) and writes the other two is genuinely /// multi-vShard on both its read set and its write set. Deterministic: -/// `VShardId::from_collection_in_database` is a pure function of the database id + +/// `VShardId::from_collection` is a pure function of the database id + /// collection-name bytes. fn distinct_vshard_quad() -> (String, String, String, String) { let mut chosen: Vec<(String, u32)> = Vec::new(); for i in 0u32..2048 { let name = format!("shuffle_join_occ_{i}"); - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let v = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if chosen.iter().all(|(_, cv)| *cv != v) { chosen.push((name, v)); if chosen.len() == 4 { diff --git a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_graph_txn.rs b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_graph_txn.rs index 05fb8e06c..babc516ea 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_graph_txn.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_graph_txn.rs @@ -220,8 +220,10 @@ async fn calvin_commit_publishes_control_changes_at_participant_lsns() { let second = (0..4096) .map(|i| format!("sncgtx_cdc_b_{i}")) .find(|candidate| { - VShardId::from_collection_in_database(nodedb_types::DatabaseId::DEFAULT, candidate) - != VShardId::from_collection_in_database(nodedb_types::DatabaseId::DEFAULT, first) + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, candidate) + .vshard() + != nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, first) + .vshard() }) .expect("collection on a distinct vShard"); for collection in [first, second.as_str()] { diff --git a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_hot_key_reservation.rs b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_hot_key_reservation.rs index ed3354781..3c0c447d5 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_hot_key_reservation.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_hot_key_reservation.rs @@ -19,7 +19,7 @@ use std::time::{Duration, Instant}; use nodedb::control::cluster::calvin::scheduler::lock::LockKey; use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; -use nodedb_types::id::{DatabaseId, VShardId}; +use nodedb_types::id::DatabaseId; use common::cluster_harness::{TestClusterNode, wait_for}; @@ -41,7 +41,9 @@ fn sequencer_leader(node: &TestClusterNode) -> u64 { fn other_vshard_collection(exclude_vshard: u32) -> String { for i in 0u32..4096 { let name = format!("hkr_other_{i}"); - if VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32() + if nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32() != exclude_vshard { return name; @@ -55,7 +57,8 @@ fn other_vshard_collection(exclude_vshard: u32) -> String { fn reservation_count(node: &TestClusterNode, vshard: u32, key: &LockKey) -> usize { let managers = node .shared - .calvin_lock_managers + .calvin + .lock_managers .lock() .unwrap_or_else(|p| p.into_inner()); let Some(lm) = managers.get(&vshard) else { @@ -80,7 +83,9 @@ async fn hot_key_read_reservation_installs_self_upgrades_and_releases() { .await; let hot_coll = "hkr_hot_kv"; - let hot_vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, hot_coll).as_u32(); + let hot_vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, hot_coll) + .vshard() + .as_u32(); let other_coll = other_vshard_collection(hot_vshard); node.client @@ -112,6 +117,7 @@ async fn hot_key_read_reservation_installs_self_upgrades_and_releases() { { let mut table = node .shared + .calvin .hot_key_table .lock() .unwrap_or_else(|p| p.into_inner()); diff --git a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_two_phase.rs b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_two_phase.rs index 9bc0d3913..8c7653806 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_two_phase.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_two_phase.rs @@ -7,7 +7,7 @@ //! //! The staged buffer + verdict-driven flush/drop live on the `!Send` Data-Plane //! core, so this asserts the flush FIRED via the node-global -//! `calvin_counters.commits_flushed` counter (incremented once per staged apply the +//! `calvin.counters.commits_flushed` counter (incremented once per staged apply the //! per-vShard scheduler resolved to commit) plus the functional proof that the //! flushed write is visible. @@ -38,7 +38,8 @@ fn sequencer_leader(node: &TestClusterNode) -> u64 { /// flushing their commit-pending buffer to base. fn commits_flushed(node: &TestClusterNode) -> u64 { node.shared - .calvin_counters + .calvin + .counters .commits_flushed .load(Ordering::Relaxed) } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_write_versions.rs b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_write_versions.rs index aeb3cbfc9..b4ee53f00 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_write_versions.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_write_versions.rs @@ -6,7 +6,7 @@ //! //! The version index itself lives on the `!Send` Data-Plane core and its //! readers are test-only, so this asserts the recording FIRED via the -//! node-global `calvin_counters.write_versions_recorded` counter, which the per-vShard +//! node-global `calvin.counters.write_versions_recorded` counter, which the per-vShard //! scheduler increments once per committed Calvin apply for which it dispatched //! a write-version record op (at the CalvinApplied WAL LSN). The counter is the //! standalone-observable proof that a cross-shard-committed write now advances @@ -40,7 +40,8 @@ fn sequencer_leader(node: &TestClusterNode) -> u64 { /// the per-core write-version index. fn write_versions_recorded(node: &TestClusterNode) -> u64 { node.shared - .calvin_counters + .calvin + .counters .write_versions_recorded .load(Ordering::Relaxed) } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/sync_constraint_version_fence.rs b/nodedb-cluster-tests/tests/common_suite/cases/sync_constraint_version_fence.rs index 1a1be059d..b1d3c7df6 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/sync_constraint_version_fence.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/sync_constraint_version_fence.rs @@ -99,11 +99,11 @@ async fn read_crdt_doc( } // A missing document is terminal, not transient: the CRDT read path // answers `NotFound`, which the `crdt_state` scalar function surfaces - // as an internal error whose message carries "NotFound". That is - // exactly the "not imported" state the fence assertion expects. + // as SQLSTATE `02000` (no_data). That is exactly the "not imported" + // state the fence assertion expects. Err(e) if e.as_db_error() - .is_some_and(|d| d.message().contains("NotFound")) => + .is_some_and(|d| d.code() == &tokio_postgres::error::SqlState::NO_DATA) => { return Ok(None); } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/sync_failover.rs b/nodedb-cluster-tests/tests/common_suite/cases/sync_failover.rs index 606a4ed84..30c8d8043 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/sync_failover.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/sync_failover.rs @@ -245,7 +245,7 @@ async fn cluster_sync_columnar_dedup_survives_failover() { const COLL: &str = "csync_failover"; const PRODUCER: u64 = 7777; let tenant = TenantId::new(0); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, COLL); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, COLL).vshard(); cluster .exec_ddl_on_any_leader(&format!( diff --git a/nodedb-cluster-tests/tests/common_suite/cases/sync_peer_id_collision.rs b/nodedb-cluster-tests/tests/common_suite/cases/sync_peer_id_collision.rs index 0eeca7bb6..21269ed21 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/sync_peer_id_collision.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/sync_peer_id_collision.rs @@ -113,7 +113,7 @@ async fn row_is_readable( } Err(e) if e.as_db_error() - .is_some_and(|d| d.message().contains("NotFound")) => + .is_some_and(|d| d.code() == &tokio_postgres::error::SqlState::NO_DATA) => { return Ok(false); } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/sync_retryable_delta_refusal.rs b/nodedb-cluster-tests/tests/common_suite/cases/sync_retryable_delta_refusal.rs index 61f781b44..b918206c7 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/sync_retryable_delta_refusal.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/sync_retryable_delta_refusal.rs @@ -106,7 +106,7 @@ async fn read_crdt_doc( } Err(e) if e.as_db_error() - .is_some_and(|d| d.message().contains("NotFound")) => + .is_some_and(|d| d.code() == &tokio_postgres::error::SqlState::NO_DATA) => { return Ok(None); } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/vshard_names.rs b/nodedb-cluster-tests/tests/common_suite/cases/vshard_names.rs index 155e91073..201508c74 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/vshard_names.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/vshard_names.rs @@ -2,7 +2,7 @@ //! Deterministic name picking for cross-vShard tests. //! -//! vShards are per collection (`VShardId::from_collection_in_database`) and +//! vShards are per collection (`VShardId::from_collection`) and //! per graph endpoint key (`VShardId::from_key`). Both are pure functions of //! their input bytes, so a test can pick names that land on distinct vShards //! without probing the cluster. @@ -16,10 +16,12 @@ const MAX_TRIES: u32 = 512; /// used verbatim; `second` is `{second_prefix}_{i}` for the lowest `i` that /// hashes away from `first`. pub fn distinct_vshard_collections(first: &str, second_prefix: &str) -> (String, String) { - let first_vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, first); + let first_vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, first).vshard(); for i in 0..MAX_TRIES { let second = format!("{second_prefix}_{i}"); - if VShardId::from_collection_in_database(DatabaseId::DEFAULT, &second) != first_vshard { + if nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &second).vshard() + != first_vshard + { return (first.to_owned(), second); } } @@ -33,7 +35,8 @@ pub fn distinct_vshard_collections(first: &str, second_prefix: &str) -> (String, /// `collection`'s own vShard, so an implicit edge task homed on the key is /// dispatched to a different vShard than the document write. pub fn key_on_other_vshard(collection: &str, prefix: &str) -> String { - let coll_vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, collection); + let coll_vshard = + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection).vshard(); for i in 0..MAX_TRIES { let key = format!("{prefix}_{i}"); if VShardId::from_key(key.as_bytes()) != coll_vshard { diff --git a/nodedb-cluster-tests/tests/dml_suite/cases/sql_cluster_cross_node_dml.rs b/nodedb-cluster-tests/tests/dml_suite/cases/sql_cluster_cross_node_dml.rs index c37a5423f..388047548 100644 --- a/nodedb-cluster-tests/tests/dml_suite/cases/sql_cluster_cross_node_dml.rs +++ b/nodedb-cluster-tests/tests/dml_suite/cases/sql_cluster_cross_node_dml.rs @@ -45,6 +45,10 @@ mod graph_traverse_reverse_cross_node; mod join_cross_node; #[path = "../../sql_cluster_cross_node_dml_tests/native_implicit_edge_delete_cross_node.rs"] mod native_implicit_edge_delete_cross_node; +#[path = "../../sql_cluster_cross_node_dml_tests/permission_tree_cross_node.rs"] +mod permission_tree_cross_node; +#[path = "../../sql_cluster_cross_node_dml_tests/permission_tree_lease_partition.rs"] +mod permission_tree_lease_partition; #[path = "../../sql_cluster_cross_node_dml_tests/schema_objects.rs"] mod schema_objects; #[path = "../../sql_cluster_cross_node_dml_tests/select_remote_stream_cross_node.rs"] diff --git a/nodedb-cluster-tests/tests/misc_suite/cases/calvin_sequencer_starvation.rs b/nodedb-cluster-tests/tests/misc_suite/cases/calvin_sequencer_starvation.rs index e541310bb..716ced9b6 100644 --- a/nodedb-cluster-tests/tests/misc_suite/cases/calvin_sequencer_starvation.rs +++ b/nodedb-cluster-tests/tests/misc_suite/cases/calvin_sequencer_starvation.rs @@ -26,16 +26,15 @@ use nodedb_cluster::calvin::sequencer::validator::validate_batch; use nodedb_cluster::calvin::types::{ EngineKeySet, ReadWriteSet, SortedVec, TxClass, VersionedReadSet, }; -use nodedb_types::{ - TenantId, - id::{DatabaseId, VShardId}, -}; +use nodedb_types::{TenantId, id::DatabaseId}; fn find_two_distinct_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); diff --git a/nodedb-cluster-tests/tests/misc_suite/cases/cluster_triggers.rs b/nodedb-cluster-tests/tests/misc_suite/cases/cluster_triggers.rs index 799b46536..5ef85e352 100644 --- a/nodedb-cluster-tests/tests/misc_suite/cases/cluster_triggers.rs +++ b/nodedb-cluster-tests/tests/misc_suite/cases/cluster_triggers.rs @@ -183,6 +183,7 @@ fn event_source_preserved_through_write_event() { op: WriteOp::Insert, row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("doc-1")), lsn: Lsn::new(100), + record: None, tenant_id: TenantId::new(1), vshard_id: VShardId::new(0), source: EventSource::User, diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/auth_objects.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/auth_objects.rs index aa19f6621..2e83ce568 100644 --- a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/auth_objects.rs +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/auth_objects.rs @@ -11,7 +11,7 @@ async fn user_create_visible_on_every_node() { let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); cluster - .exec_ddl_on_any_leader("CREATE USER alice WITH PASSWORD 'sekret123' ROLE read_write") + .exec_ddl_on_any_leader("CREATE USER alice WITH PASSWORD 'sekret123' ROLE readwrite") .await .expect("create user"); @@ -76,38 +76,43 @@ async fn role_create_visible_on_every_node() { async fn alter_user_role_replicates() { let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); - cluster - .exec_ddl_on_any_leader("CREATE USER bob WITH PASSWORD 'initial-pass' ROLE read_only") - .await - .expect("create user"); + for sql in [ + "CREATE ROLE auditor", + "CREATE USER bob WITH PASSWORD 'initial-pass' ROLE readonly", + ] { + cluster + .exec_ddl_on_any_leader(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } wait_for( - "all 3 nodes see bob with read_only role", + "all 3 nodes see bob with the readonly role", Duration::from_secs(10), Duration::from_millis(50), || { cluster .nodes .iter() - .all(|n| n.user_has_role("bob", "read_only")) + .all(|n| n.user_has_role("bob", "readonly")) }, ) .await; cluster - .exec_ddl_on_any_leader("ALTER USER bob SET ROLE read_write") + .exec_ddl_on_any_leader("ALTER USER bob SET ROLE auditor") .await .expect("alter user set role"); wait_for( - "all 3 nodes see bob with read_write role", + "all 3 nodes see bob with the custom auditor role", Duration::from_secs(10), Duration::from_millis(50), || { cluster .nodes .iter() - .all(|n| n.user_has_role("bob", "read_write")) + .all(|n| n.user_has_role("bob", "auditor")) }, ) .await; @@ -115,6 +120,98 @@ async fn alter_user_role_replicates() { cluster.shutdown().await; } +/// A role name that is neither built in nor defined is refused on every +/// entry point, on every node, with SQLSTATE 42704. No node ends up with a +/// user holding it. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn an_undefined_role_is_refused_on_every_node() { + let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); + + cluster + .exec_ddl_on_any_leader("CREATE USER carol WITH PASSWORD 'carol-pass-1' ROLE readonly") + .await + .expect("create carol"); + + for node in &cluster.nodes { + for sql in [ + "CREATE USER dave WITH PASSWORD 'dave-pass-1' ROLE read_write", + "ALTER USER carol SET ROLE read_write", + "GRANT ROLE read_write TO carol", + ] { + let error = node + .exec(sql) + .await + .expect_err("an undefined role must be refused"); + assert!( + error.contains("42704") && error.contains("read_write"), + "node {}: {sql}: expected 42704 naming the role, got {error}", + node.node_id + ); + } + } + for node in &cluster.nodes { + assert!( + !node.has_active_user("dave"), + "node {} created dave", + node.node_id + ); + assert!( + node.user_has_role("carol", "readonly") && !node.user_has_role("carol", "read_write"), + "node {} changed carol's roles", + node.node_id + ); + } + + cluster.shutdown().await; +} + +/// A role a user holds is not dropped, as PostgreSQL refuses: the drop +/// fails with SQLSTATE 2BP01 naming the user. Once no user holds it, the +/// drop succeeds on every node. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_held_role_is_not_dropped() { + let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); + + for sql in [ + "CREATE ROLE reviewer", + "CREATE USER erin WITH PASSWORD 'erin-pass-1' ROLE reviewer", + ] { + cluster + .exec_ddl_on_any_leader(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + + let error = cluster.nodes[0] + .exec("DROP ROLE reviewer") + .await + .expect_err("a held role must not be dropped"); + assert!( + error.contains("2BP01") && error.contains("erin"), + "expected 2BP01 naming the holder, got {error}" + ); + assert!( + cluster.nodes.iter().all(|n| n.has_role("reviewer")), + "the refused drop removed the role on some node" + ); + + for sql in ["ALTER USER erin SET ROLE readonly", "DROP ROLE reviewer"] { + cluster + .exec_ddl_on_any_leader(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + wait_for( + "all 3 nodes no longer see the dropped role", + Duration::from_secs(10), + Duration::from_millis(50), + || cluster.nodes.iter().all(|n| !n.has_role("reviewer")), + ) + .await; + + cluster.shutdown().await; +} + #[tokio::test(flavor = "multi_thread", worker_threads = 6)] async fn api_key_create_and_revoke_replicates() { let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/join_cross_node.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/join_cross_node.rs index 452708b45..94a8a8388 100644 --- a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/join_cross_node.rs +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/join_cross_node.rs @@ -53,8 +53,14 @@ async fn cross_node_join_returns_all_matches() { const FACT: &str = "fact"; const DIM: &str = "dim"; assert_ne!( - vshard_for_collection(DatabaseId::DEFAULT, FACT), - vshard_for_collection(DatabaseId::DEFAULT, DIM), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + FACT + )), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + DIM + )), "test collections must hash to different vShards to exercise cross-node join" ); @@ -177,8 +183,14 @@ async fn cross_node_join_compares_time_keys_in_one_unit() { const EVENTS: &str = "ts_events"; const FEATURES: &str = "ts_features"; assert_ne!( - vshard_for_collection(DatabaseId::DEFAULT, EVENTS), - vshard_for_collection(DatabaseId::DEFAULT, FEATURES), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + EVENTS + )), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + FEATURES + )), "test collections must hash to different vShards to exercise cross-node join" ); diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/permission_tree_cross_node.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/permission_tree_cross_node.rs new file mode 100644 index 000000000..e77268c93 --- /dev/null +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/permission_tree_cross_node.rs @@ -0,0 +1,201 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A permission-tree revoke acknowledged on one node binds the next +//! statement on every other node. +//! +//! A grant is a row in the tree's permission table. Node B learns of it when +//! its own replica applies the write and its Event Plane updates its +//! permission cache. Node A acknowledges the write only after every node +//! holding an authorization lease reported that coverage, or its lease +//! expired. So B's first statement after the acknowledgement either sees the +//! change or is refused with a retryable error. It never plans against the +//! grant before the revoke. +//! +//! The test repeats grant and revoke several rounds and reads on every other +//! node right after each acknowledgement, with no wait in between. + +use std::time::{Duration, Instant}; + +use crate::common::cluster_harness::TestCluster; + +const PROBE_USER: &str = "ptx_probe"; +const SELECT_DOCS: &str = "SELECT id FROM ptx_docs ORDER BY id"; +const ROUNDS: usize = 5; + +/// SQLSTATE of a statement refused because this node's authorization state +/// is behind. The client retries it. +const AUTHORIZATION_BEHIND: &str = "55P03"; + +/// The outcome of one probe read. +#[derive(Debug, PartialEq)] +enum Read { + Rows(Vec), + Refused, +} + +async fn probe_read(client: &tokio_postgres::Client) -> Read { + match client.simple_query(SELECT_DOCS).await { + Ok(messages) => Read::Rows( + messages + .into_iter() + .filter_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => { + Some(row.get(0).unwrap_or("").to_string()) + } + _ => None, + }) + .collect(), + ), + Err(error) => { + let code = error.code().map(|code| code.code().to_string()); + assert_eq!( + code.as_deref(), + Some(AUTHORIZATION_BEHIND), + "a probe read failed with an error other than a retryable refusal: {error:?}" + ); + Read::Refused + } + } +} + +/// Wait until `probe` reads exactly `expected`. A retryable refusal and a +/// replica still catching up are retried. A denial fails at once, in +/// [`probe_read`]. +async fn await_rows(probe: &tokio_postgres::Client, expected: &[&str], what: &str) { + let expected: Vec = expected.iter().map(|row| row.to_string()).collect(); + let deadline = Instant::now() + Duration::from_secs(10); + loop { + let read = probe_read(probe).await; + if read == Read::Rows(expected.clone()) { + return; + } + assert!( + Instant::now() < deadline, + "{what}: last read {read:?}, expected {expected:?}" + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } +} + +async fn connect_probe( + pg_addr: std::net::SocketAddr, +) -> (tokio_postgres::Client, tokio::task::JoinHandle<()>) { + let conn_str = format!( + "host={} port={} user={PROBE_USER} dbname=default", + pg_addr.ip(), + pg_addr.port() + ); + let (client, connection) = tokio_postgres::connect(&conn_str, tokio_postgres::NoTls) + .await + .unwrap_or_else(|e| panic!("connect as {PROBE_USER} to {pg_addr}: {e}")); + let handle = tokio::spawn(async move { + let _ = connection.await; + }); + (client, handle) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn revoke_on_one_node_binds_the_next_statement_on_every_other_node() { + let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); + + for sql in [ + "CREATE COLLECTION ptx_docs (id TEXT PRIMARY KEY, title TEXT) \ + WITH (engine='document_strict')", + "CREATE COLLECTION ptx_grants", + "CREATE ROLE ptx_role", + // `readwrite` is the built-in role that grants Read on every + // collection. An unknown name such as `read_write` is a custom role + // with no permissions, and every read would be denied. + "CREATE USER ptx_probe WITH PASSWORD 'ptx-probe-password' ROLE readwrite", + "GRANT ROLE ptx_role TO ptx_probe", + ] { + cluster + .exec_ddl_on_any_leader(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + let writer = &cluster.nodes[0]; + for sql in [ + "INSERT INTO ptx_docs (id, title) VALUES ('d1', 'Doc One')", + "INSERT INTO ptx_docs (id, title) VALUES ('d2', 'Doc Two')", + ] { + writer + .exec(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + // The probe's base Read is in effect on every node, the writer included, + // before the tree narrows it. A denial here is a setup error, not a + // tree result. + for (index, node) in cluster.nodes.iter().enumerate() { + let (probe, handle) = connect_probe(node.pg_addr).await; + await_rows( + &probe, + &["d1", "d2"], + &format!("node {index}: base Read before the tree"), + ) + .await; + drop(probe); + handle.abort(); + } + + cluster + .exec_ddl_on_any_leader( + "ALTER COLLECTION ptx_docs SET PERMISSION_TREE = '{\ + \"resource_column\":\"id\",\ + \"graph_index\":\"ptx_docs_tree\",\ + \"permission_table\":\"ptx_grants\"\ + }'", + ) + .await + .expect("set permission tree"); + + let mut probes = Vec::new(); + for node in &cluster.nodes[1..] { + probes.push(connect_probe(node.pg_addr).await); + } + + // Reads that planned rather than being refused. A run of refusals only + // proves nothing was served stale, so the test also requires answers. + let mut answered = 0usize; + for round in 0..ROUNDS { + writer + .exec( + "INSERT INTO ptx_grants (resource_id, grantee, level, inherited) \ + VALUES ('d1', 'ptx_role', 'viewer', false)", + ) + .await + .unwrap_or_else(|e| panic!("round {round}: grant: {e}")); + for (index, (probe, _)) in probes.iter().enumerate() { + let read = probe_read(probe).await; + answered += usize::from(read != Read::Refused); + assert!( + read == Read::Rows(vec!["d1".to_string()]) || read == Read::Refused, + "round {round}: node {} planned against the state before the grant: {read:?}", + index + 1 + ); + } + + writer + .exec("DELETE FROM ptx_grants WHERE resource_id = 'd1' AND grantee = 'ptx_role'") + .await + .unwrap_or_else(|e| panic!("round {round}: revoke: {e}")); + for (index, (probe, _)) in probes.iter().enumerate() { + let read = probe_read(probe).await; + answered += usize::from(read != Read::Refused); + assert!( + read == Read::Rows(Vec::new()) || read == Read::Refused, + "round {round}: node {} served d1 after the revoke was acknowledged: {read:?}", + index + 1 + ); + } + } + + assert!(answered > 0, "every probe read was refused"); + + for (probe, handle) in probes { + drop(probe); + handle.abort(); + } + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/permission_tree_lease_partition.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/permission_tree_lease_partition.rs new file mode 100644 index 000000000..4136d21b0 --- /dev/null +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/permission_tree_lease_partition.rs @@ -0,0 +1,224 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A partitioned node loses its authorization lease, and a revoke +//! acknowledged meanwhile never plans on it. +//! +//! Node B is severed from the other two nodes. It cannot renew its lease, so +//! the revoke written on another node is acknowledged once B's lease lapses. +//! From then on B refuses every permission-checked statement with a +//! retryable error. After the partition heals, B renews only once its own +//! state covers the revoke, so its first answer shows the revoke. + +use std::time::{Duration, Instant}; + +use crate::common::cluster_harness::TestCluster; + +const SELECT_DOCS: &str = "SELECT id FROM ptl_docs ORDER BY id"; +const AUTHORIZATION_BEHIND: &str = "55P03"; + +#[derive(Debug, PartialEq)] +enum Read { + Rows(Vec), + Refused, +} + +async fn probe_read(client: &tokio_postgres::Client) -> Read { + match client.simple_query(SELECT_DOCS).await { + Ok(messages) => Read::Rows( + messages + .into_iter() + .filter_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => { + Some(row.get(0).unwrap_or("").to_string()) + } + _ => None, + }) + .collect(), + ), + Err(error) => { + assert_eq!( + error.code().map(|code| code.code().to_string()).as_deref(), + Some(AUTHORIZATION_BEHIND), + "a probe read failed with an error other than a retryable refusal: {error:?}" + ); + Read::Refused + } + } +} + +async fn connect_probe( + pg_addr: std::net::SocketAddr, +) -> (tokio_postgres::Client, tokio::task::JoinHandle<()>) { + let conn_str = format!( + "host={} port={} user=ptl_probe dbname=default", + pg_addr.ip(), + pg_addr.port() + ); + let (client, connection) = tokio_postgres::connect(&conn_str, tokio_postgres::NoTls) + .await + .unwrap_or_else(|e| panic!("connect as ptl_probe to {pg_addr}: {e}")); + let handle = tokio::spawn(async move { + let _ = connection.await; + }); + (client, handle) +} + +/// Sever node `b` from every other node, both ways, or heal it. +fn partition(cluster: &TestCluster, b: usize, severed: bool) { + let b_id = cluster.nodes[b].node_id; + let b_transport = cluster.nodes[b] + .shared + .cluster_transport + .as_ref() + .expect("cluster transport"); + for (index, node) in cluster.nodes.iter().enumerate() { + if index == b { + continue; + } + let transport = node + .shared + .cluster_transport + .as_ref() + .expect("cluster transport"); + if severed { + transport.sever(b_id); + b_transport.sever(node.node_id); + } else { + transport.heal(b_id); + b_transport.heal(node.node_id); + } + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_partitioned_node_refuses_until_it_covers_the_revoke() { + let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); + + for sql in [ + "CREATE COLLECTION ptl_docs (id TEXT PRIMARY KEY, title TEXT) \ + WITH (engine='document_strict')", + "CREATE COLLECTION ptl_grants", + "CREATE ROLE ptl_role", + // `readwrite` is the built-in role that grants Read on every + // collection. An unknown name such as `read_write` is a custom role + // with no permissions, and every read would be denied. + "CREATE USER ptl_probe WITH PASSWORD 'ptl-probe-password' ROLE readwrite", + "GRANT ROLE ptl_role TO ptl_probe", + ] { + cluster + .exec_ddl_on_any_leader(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + + // B leads neither the metadata group nor the group homing the grants, + // so both keep a quorum while B is cut off. + let grants_group = cluster.nodes[0] + .group_id_for_collection("ptl_grants") + .expect("grants group"); + let metadata_leader = cluster.nodes[0].metadata_group_leader(); + let grants_leader = cluster.nodes[0] + .all_group_leaders() + .into_iter() + .find(|(group, _)| *group == grants_group) + .map(|(_, leader)| leader) + .unwrap_or(0); + let b = cluster + .nodes + .iter() + .position(|node| node.node_id != metadata_leader && node.node_id != grants_leader) + .expect("a node that leads neither group"); + let writer = &cluster.nodes[(b + 1) % cluster.nodes.len()]; + + writer + .exec("INSERT INTO ptl_docs (id, title) VALUES ('d1', 'Doc One')") + .await + .expect("insert d1"); + + // The probe's base Read is in effect on every node, B included, before + // the tree narrows it. A denial here is a setup error, not a tree result. + for (index, node) in cluster.nodes.iter().enumerate() { + let (probe, handle) = connect_probe(node.pg_addr).await; + let deadline = Instant::now() + Duration::from_secs(10); + loop { + let read = probe_read(&probe).await; + if read == Read::Rows(vec!["d1".to_string()]) { + break; + } + assert!( + Instant::now() < deadline, + "node {index}: base Read before the tree: last read {read:?}" + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } + drop(probe); + handle.abort(); + } + + cluster + .exec_ddl_on_any_leader( + "ALTER COLLECTION ptl_docs SET PERMISSION_TREE = '{\ + \"resource_column\":\"id\",\ + \"graph_index\":\"ptl_docs_tree\",\ + \"permission_table\":\"ptl_grants\"\ + }'", + ) + .await + .expect("set permission tree"); + writer + .exec( + "INSERT INTO ptl_grants (resource_id, grantee, level, inherited) \ + VALUES ('d1', 'ptl_role', 'viewer', false)", + ) + .await + .expect("grant d1"); + + let (probe, probe_handle) = connect_probe(cluster.nodes[b].pg_addr).await; + let deadline = Instant::now() + Duration::from_secs(10); + loop { + match probe_read(&probe).await { + Read::Rows(rows) if rows == vec!["d1".to_string()] => break, + Read::Rows(rows) => panic!("node B served {rows:?} after the grant was acknowledged"), + Read::Refused => {} + } + assert!(Instant::now() < deadline, "node B never served the grant"); + tokio::time::sleep(Duration::from_millis(50)).await; + } + + partition(&cluster, b, true); + writer + .exec("DELETE FROM ptl_grants WHERE resource_id = 'd1' AND grantee = 'ptl_role'") + .await + .expect("the revoke is acknowledged once B's lease lapses"); + + // B's lease has lapsed: it refuses rather than plan against the grant. + for _ in 0..5 { + assert_eq!( + probe_read(&probe).await, + Read::Refused, + "the partitioned node planned without a lease" + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } + + partition(&cluster, b, false); + let deadline = Instant::now() + Duration::from_secs(20); + loop { + match probe_read(&probe).await { + Read::Rows(rows) => { + assert!(rows.is_empty(), "node B served the revoked grant: {rows:?}"); + break; + } + Read::Refused => {} + } + assert!( + Instant::now() < deadline, + "node B never renewed after the partition healed" + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } + + drop(probe); + probe_handle.abort(); + cluster.shutdown().await; +} diff --git a/nodedb-cluster/src/array_routing.rs b/nodedb-cluster/src/array_routing.rs index 1cd89d73a..0085835e2 100644 --- a/nodedb-cluster/src/array_routing.rs +++ b/nodedb-cluster/src/array_routing.rs @@ -36,7 +36,7 @@ pub const VSHARD_COUNT: u32 = 1024; /// /// Array-specific name-only fallback used by `array_sync` paths that route /// before a coordinate or tile extent is known. Uses the same DJB -/// multiply-31 hash as `VShardId::from_collection_in_database`, but without +/// multiply-31 hash as `VShardId::from_collection`, but without /// the database scope — array-sync messages carry their own scoping in the /// op-log header and route per-array by name. pub fn array_vshard_for_name(array_name: &str) -> u32 { diff --git a/nodedb-cluster/src/calvin/applied_acks.rs b/nodedb-cluster/src/calvin/applied_acks.rs new file mode 100644 index 000000000..ff9ad7893 --- /dev/null +++ b/nodedb-cluster/src/calvin/applied_acks.rs @@ -0,0 +1,87 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Completion acks this node applied from the sequencer log, with their +//! Raft index. +//! +//! A participant's scheduler on the vShard leader proposes a `CompletionAck` +//! once it applied a transaction. Every other replica of that vShard applies +//! the transaction in its own time. A node that serves authorization state +//! from its replicas needs to know, for each ack in the log, whether its own +//! replica applied the transaction too. The sequencer state machine records +//! each ack here as it applies it. The host crate drains the log and settles +//! each ack against its local schedulers. +//! +//! The log records nothing until the host enables it, so a node that never +//! drains it holds no entries. + +use std::collections::VecDeque; +use std::sync::Mutex; + +use super::completion::TxnId; + +/// One `CompletionAck` applied from the sequencer log. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct AppliedCompletionAck { + /// Raft index of the ack in the sequencer group. + pub index: u64, + pub txn: TxnId, + pub vshard_id: u32, +} + +/// Applied acks not yet drained by the host. `None` until enabled. +#[derive(Debug, Default)] +pub struct AppliedAckLog { + acks: Mutex>>, +} + +impl AppliedAckLog { + /// Start recording applied acks. + pub fn enable(&self) { + let mut acks = self.acks.lock().unwrap_or_else(|p| p.into_inner()); + if acks.is_none() { + *acks = Some(VecDeque::new()); + } + } + + /// Record an ack the sequencer state machine applied at `index`. + pub fn record(&self, ack: AppliedCompletionAck) { + if let Some(acks) = self.acks.lock().unwrap_or_else(|p| p.into_inner()).as_mut() { + acks.push_back(ack); + } + } + + /// Take every recorded ack, in log order. + pub fn drain(&self) -> Vec { + self.acks + .lock() + .unwrap_or_else(|p| p.into_inner()) + .as_mut() + .map(|acks| acks.drain(..).collect()) + .unwrap_or_default() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn ack(index: u64) -> AppliedCompletionAck { + AppliedCompletionAck { + index, + txn: TxnId::new(index, 0), + vshard_id: 3, + } + } + + #[test] + fn nothing_is_recorded_until_enabled() { + let log = AppliedAckLog::default(); + log.record(ack(1)); + assert!(log.drain().is_empty()); + log.enable(); + log.record(ack(2)); + log.record(ack(3)); + assert_eq!(log.drain(), vec![ack(2), ack(3)]); + assert!(log.drain().is_empty()); + } +} diff --git a/nodedb-cluster/src/calvin/completion.rs b/nodedb-cluster/src/calvin/completion.rs index 05d42629a..4f27a7532 100644 --- a/nodedb-cluster/src/calvin/completion.rs +++ b/nodedb-cluster/src/calvin/completion.rs @@ -73,6 +73,24 @@ impl VerdictOutcome { } } +/// What this node's registry has applied for one participant vShard of a txn. +/// +/// A scheduler reads it to learn whether a sequencer entry it proposed has +/// been applied here. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ParticipantProgress { + /// A `Vote` or `AbortVote` from this vShard is in the tally. + pub voted: bool, + /// A `CompletionAck` from this vShard is recorded. + pub acked: bool, + /// The txn's global verdict is stored. + pub has_verdict: bool, + /// An `OllpMismatch` for the txn is recorded. + pub mismatched: bool, + /// A `TxnRoutingFailed` for the txn is recorded and not yet delivered. + pub routing_failed: bool, +} + pub(crate) struct PendingCompletion { /// `pub(crate)`: also read/written by the vote/verdict-tally methods in /// `completion_verdict.rs` (a sibling module in the same crate). @@ -176,6 +194,8 @@ pub struct CalvinCompletionRegistry { /// the signal into a `SequencerEntry::Verdict` proposal. `pub(crate)`: also /// used by `note_vote` in `completion_verdict.rs`. pub(crate) verdict_tx: mpsc::Sender<(TxnId, VerdictOutcome)>, + /// Completion acks this node applied, with their Raft index. + pub applied_acks: super::applied_acks::AppliedAckLog, } impl CalvinCompletionRegistry { @@ -186,6 +206,7 @@ impl CalvinCompletionRegistry { Arc::new(Self { inner: Mutex::new(Inner::default()), verdict_tx, + applied_acks: super::applied_acks::AppliedAckLog::default(), }) } @@ -384,6 +405,25 @@ impl CalvinCompletionRegistry { } } + /// What this registry holds for participant `vshard` of `txn`. + /// + /// `None` means no entry exists for `txn`. That is either a txn this node + /// never seeded, or one whose outcome already fired and evicted its entry. + pub fn participant_progress(&self, txn: TxnId, vshard: u32) -> Option { + self.inner + .lock() + .unwrap_or_else(|p| p.into_inner()) + .completions + .get(&txn) + .map(|entry| ParticipantProgress { + voted: entry.votes.contains_key(&vshard), + acked: entry.acked_vshards.contains(&vshard), + has_verdict: entry.verdict.is_some(), + mismatched: entry.mismatched, + routing_failed: entry.routing_failed.is_some(), + }) + } + /// Test-only: returns the number of pending completion entries. /// Used to verify entries are removed once all acks arrive (no leak). #[cfg(test)] @@ -711,4 +751,45 @@ mod tests { "entry must be evicted once mismatch is signalled" ); } + + #[tokio::test] + async fn participant_progress_is_none_before_any_entry_exists() { + let reg = CalvinCompletionRegistry::new_detached(); + assert_eq!(reg.participant_progress(TxnId::new(40, 0), 1), None); + } + + #[tokio::test] + async fn participant_progress_reports_each_applied_signal_for_its_vshard() { + let reg = CalvinCompletionRegistry::new_detached(); + let txn = TxnId::new(40, 1); + reg.seed_expected(txn, 2); + let empty = reg.participant_progress(txn, 1).expect("seeded entry"); + assert!(!empty.voted && !empty.acked && !empty.has_verdict); + assert!(!empty.mismatched && !empty.routing_failed); + + reg.note_vote(txn, 1, ParticipantVote::Commit); + reg.note_completion_ack(txn, 1); + let own = reg.participant_progress(txn, 1).expect("entry"); + assert!(own.voted && own.acked); + let peer = reg.participant_progress(txn, 2).expect("entry"); + assert!( + !peer.voted && !peer.acked, + "another vShard's signals do not count" + ); + + reg.note_verdict(txn, VerdictOutcome::Commit); + reg.note_ollp_mismatch(txn); + reg.note_routing_failed(txn, "unroutable".to_string()); + let txn_wide = reg.participant_progress(txn, 2).expect("entry"); + assert!(txn_wide.has_verdict && txn_wide.mismatched && txn_wide.routing_failed); + } + + #[tokio::test] + async fn participant_progress_is_none_once_the_outcome_fired() { + let reg = CalvinCompletionRegistry::new_detached(); + let txn = TxnId::new(40, 2); + let _rx = reg.register_completion(txn, 1); + reg.note_completion_ack(txn, 1); + assert_eq!(reg.participant_progress(txn, 1), None); + } } diff --git a/nodedb-cluster/src/calvin/mod.rs b/nodedb-cluster/src/calvin/mod.rs index 49ae97404..659690779 100644 --- a/nodedb-cluster/src/calvin/mod.rs +++ b/nodedb-cluster/src/calvin/mod.rs @@ -1,12 +1,15 @@ // SPDX-License-Identifier: BUSL-1.1 +pub mod applied_acks; pub mod completion; mod completion_verdict; pub mod sequencer; pub mod types; +pub use applied_acks::{AppliedAckLog, AppliedCompletionAck}; pub use completion::{ - AttemptOutcome, CalvinCompletionRegistry, ParticipantVote, TxnId, VerdictOutcome, + AttemptOutcome, CalvinCompletionRegistry, ParticipantProgress, ParticipantVote, TxnId, + VerdictOutcome, }; pub use completion_verdict::VerdictSignal; pub use sequencer::{ diff --git a/nodedb-cluster/src/calvin/sequencer/entry.rs b/nodedb-cluster/src/calvin/sequencer/entry.rs index 318539a40..66b232ff6 100644 --- a/nodedb-cluster/src/calvin/sequencer/entry.rs +++ b/nodedb-cluster/src/calvin/sequencer/entry.rs @@ -133,6 +133,12 @@ pub enum SequencerEntry { position: u32, reason: AbortReason, }, + /// A backup's consistent-cut marker, carrying the backup's watermark + /// `hlc`. Every replica fans it out to each of its vShard schedulers in + /// log order. A scheduler reports the marker once every transaction + /// delivered to it before the marker finished, and gives every + /// transaction delivered after it a commit HLC above `hlc`. + CutMarker { hlc: u64 }, } #[cfg(test)] @@ -141,14 +147,16 @@ mod tests { use crate::calvin::types::{EngineKeySet, ReadWriteSet, SequencedTxn, SortedVec, TxClass}; use nodedb_types::{ TenantId, - id::{DatabaseId, VShardId}, + id::{CollectionKey, DatabaseId}, }; fn find_two_distinct_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); diff --git a/nodedb-cluster/src/calvin/sequencer/inbox.rs b/nodedb-cluster/src/calvin/sequencer/inbox.rs index 9b2d0a7db..630518de9 100644 --- a/nodedb-cluster/src/calvin/sequencer/inbox.rs +++ b/nodedb-cluster/src/calvin/sequencer/inbox.rs @@ -355,7 +355,7 @@ pub fn new_inbox(capacity: usize, config: &SequencerConfig) -> (Inbox, InboxRece mod tests { use super::*; use crate::calvin::types::{EngineKeySet, ReadWriteSet, SortedVec, TxClass}; - use nodedb_types::id::{DatabaseId, VShardId}; + use nodedb_types::id::{CollectionKey, DatabaseId}; fn default_config() -> SequencerConfig { SequencerConfig::default() @@ -365,7 +365,9 @@ mod tests { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); diff --git a/nodedb-cluster/src/calvin/sequencer/replay.rs b/nodedb-cluster/src/calvin/sequencer/replay.rs index c20e3bfb9..79d40796e 100644 --- a/nodedb-cluster/src/calvin/sequencer/replay.rs +++ b/nodedb-cluster/src/calvin/sequencer/replay.rs @@ -56,6 +56,7 @@ impl SequencerStateMachine { /// computed identically to the live [`SequencerStateMachine::apply`] path. /// * `ReserveRead` targeting `vshard_id` → [`SchedulerInput::Reserve`]. /// * `ReleaseReservation` targeting `vshard_id` → [`SchedulerInput::Release`]. + /// * `CutMarker` → [`SchedulerInput::CutMarker`], for every vShard. /// * All other variants carry no per-vShard scheduler input. /// /// Entries are emitted in Raft-log order (and, within an epoch batch, in @@ -81,6 +82,11 @@ impl SequencerStateMachine { // No-op entry (newly elected leader heartbeat). continue; } + if crate::conf_change::ConfChange::is_conf_change(&entry.data) { + // A membership change of the sequencer group, applied by the + // Raft layer; it carries no sequencer input. + continue; + } let seq_entry: SequencerEntry = match zerompk::from_msgpack(&entry.data) { Ok(e) => e, Err(err) => { @@ -98,9 +104,25 @@ impl SequencerStateMachine { continue; } // Re-derive participating_vshards (skipped during serialization) - // exactly as the live apply path does before fan-out. + // exactly as the live apply path does before fan-out. The live + // path skips an entry whose participants cannot be derived and + // files the report, so replay skips it too. + let mut underivable = None; for txn in &mut batch.txns { - txn.tx_class.restore_derived(); + if let Err(err) = txn.tx_class.restore_derived() { + underivable = Some(err); + break; + } + } + if let Some(err) = underivable { + tracing::warn!( + raft_index = entry.index, + epoch = batch.epoch, + error = %err, + "calvin replay: epoch batch carries a transaction with \ + underivable participants; skipping" + ); + continue; } // Shared with the live `apply` EpochBatch arm via @@ -140,6 +162,11 @@ impl SequencerStateMachine { } if vshard == vshard_id => { result.push(SchedulerInput::Release { owner, reason }); } + // A cut marker reaches every vShard, exactly as the live + // `CutMarker` arm fans it out. + SequencerEntry::CutMarker { hlc } => { + result.push(SchedulerInput::CutMarker { hlc }); + } // Reservation entries for a different vShard carry nothing for us. SequencerEntry::ReserveRead { .. } => {} SequencerEntry::ReleaseReservation { .. } => {} @@ -170,7 +197,7 @@ mod tests { }; use nodedb_types::{ TenantId, - id::{DatabaseId, VShardId}, + id::{CollectionKey, DatabaseId}, }; use std::collections::HashMap; use tokio::sync::mpsc; @@ -179,7 +206,9 @@ mod tests { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); @@ -193,8 +222,12 @@ mod tests { fn make_batch_with_two_vshards() -> (EpochBatch, u32, u32) { let (col_a, col_b) = find_two_distinct_collections(); - let real_va = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_a).as_u32(); - let real_vb = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_b).as_u32(); + let real_va = CollectionKey::from_bare(DatabaseId::DEFAULT, &col_a) + .vshard() + .as_u32(); + let real_vb = CollectionKey::from_bare(DatabaseId::DEFAULT, &col_b) + .vshard() + .as_u32(); let write_set = ReadWriteSet::new(vec![ EngineKeySet::Document { collection: col_a, diff --git a/nodedb-cluster/src/calvin/sequencer/service/core.rs b/nodedb-cluster/src/calvin/sequencer/service/core.rs index 297289909..3bd1bc201 100644 --- a/nodedb-cluster/src/calvin/sequencer/service/core.rs +++ b/nodedb-cluster/src/calvin/sequencer/service/core.rs @@ -509,6 +509,7 @@ fn entry_txn_count(entry: &SequencerEntry) -> usize { SequencerEntry::AbortVerdict { .. } => 0, SequencerEntry::ReserveRead { .. } => 0, SequencerEntry::ReleaseReservation { .. } => 0, + SequencerEntry::CutMarker { .. } => 0, } } @@ -531,14 +532,16 @@ mod tests { use crate::routing::RoutingTable; use nodedb_types::{ TenantId, - id::{DatabaseId, VShardId}, + id::{CollectionKey, DatabaseId}, }; fn find_two_distinct_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); diff --git a/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs b/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs index b7d239d50..4ff93a135 100644 --- a/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs +++ b/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs @@ -70,6 +70,12 @@ impl SequencerStateMachine { // committed at `index` regardless, so it is a safe replay upper bound. self.last_committed_index = index; + // A membership change of the sequencer group commits in its log too. + // It is no sequencer entry, and the Raft layer applied it already. + if crate::conf_change::ConfChange::is_conf_change(data) { + return; + } + let entry: SequencerEntry = match zerompk::from_msgpack(data) { Ok(e) => e, Err(err) => { @@ -82,8 +88,25 @@ impl SequencerStateMachine { SequencerEntry::EpochBatch { mut batch } => { // Re-derive the participating_vshards field which is skipped // during serialization (it is computed from write_set collection names). + // A class whose participants cannot be derived makes the entry + // as unusable as one that fails to decode, so it is skipped the + // same way. for txn in &mut batch.txns { - txn.tx_class.restore_derived(); + if let Err(err) = txn.tx_class.restore_derived() { + error!( + epoch = batch.epoch, + raft_index = index, + error = %err, + "sequencer state machine: epoch batch carries a transaction \ + with underivable participants; skipping entry" + ); + crate::diag::sequencer_participants_underivable( + batch.epoch, + index, + &err.to_string(), + ); + return; + } } // A halted state machine has already diverged from the log; @@ -268,8 +291,15 @@ impl SequencerStateMachine { position, vshard_id, } => { + let txn = crate::calvin::TxnId::new(epoch, position); + self.completion_registry.note_completion_ack(txn, vshard_id); self.completion_registry - .note_completion_ack(crate::calvin::TxnId::new(epoch, position), vshard_id); + .applied_acks + .record(crate::calvin::AppliedCompletionAck { + index, + txn, + vshard_id, + }); } // Broadcast the OLLP predicate-mismatch signal to ALL replicas so the // coordinator's registry fires wherever it lives (including remote nodes). @@ -387,6 +417,34 @@ impl SequencerStateMachine { } } } + // Fan a backup's cut marker out to every vShard scheduler this + // node hosts. Same `try_send` discipline as `ReserveRead`: a + // dropped marker is recovered by the scheduler's catch-up drain, + // which replays it in log order. + SequencerEntry::CutMarker { hlc } => { + for (&vshard, sender) in &self.vshard_senders { + match sender.try_send(SchedulerInput::CutMarker { hlc }) { + Ok(()) => {} + Err(mpsc::error::TrySendError::Full(_)) => { + warn!( + vshard, + hlc, + "sequencer apply: vshard channel full (backpressure); \ + dropping cut marker" + ); + self.record_catch_up(vshard, index); + } + Err(mpsc::error::TrySendError::Closed(_)) => { + warn!( + vshard, + "sequencer apply: vshard sender gone; \ + scheduler may have exited (cut marker)" + ); + self.record_catch_up(vshard, index); + } + } + } + } // Fan a reservation release out to its owning vShard's scheduler. // Same `try_send` discipline as `ReserveRead`. SequencerEntry::ReleaseReservation { @@ -434,14 +492,16 @@ mod tests { }; use nodedb_types::{ TenantId, - id::{DatabaseId, VShardId}, + id::{CollectionKey, DatabaseId}, }; fn find_two_distinct_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); @@ -460,8 +520,12 @@ mod tests { // We'll use find_two_distinct_collections and use whatever vshards they hash to. let (col_a, col_b) = find_two_distinct_collections(); let _ = (vshard_a, vshard_b); // actual vshard ids come from the collection hash - let real_va = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_a).as_u32(); - let real_vb = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_b).as_u32(); + let real_va = CollectionKey::from_bare(DatabaseId::DEFAULT, &col_a) + .vshard() + .as_u32(); + let real_vb = CollectionKey::from_bare(DatabaseId::DEFAULT, &col_b) + .vshard() + .as_u32(); let write_set = ReadWriteSet::new(vec![ EngineKeySet::Document { collection: col_a, diff --git a/nodedb-cluster/src/calvin/sequencer/validator.rs b/nodedb-cluster/src/calvin/sequencer/validator.rs index 1c5691acb..188fe6fc2 100644 --- a/nodedb-cluster/src/calvin/sequencer/validator.rs +++ b/nodedb-cluster/src/calvin/sequencer/validator.rs @@ -388,14 +388,16 @@ mod tests { use crate::calvin::types::{EngineKeySet, ReadWriteSet, SortedVec, TxClass}; use nodedb_types::{ TenantId, - id::{DatabaseId, VShardId}, + id::{CollectionKey, DatabaseId}, }; fn find_two_distinct_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); diff --git a/nodedb-cluster/src/calvin/types/scheduler_input.rs b/nodedb-cluster/src/calvin/types/scheduler_input.rs index 1ceb6a37c..4e0e7be50 100644 --- a/nodedb-cluster/src/calvin/types/scheduler_input.rs +++ b/nodedb-cluster/src/calvin/types/scheduler_input.rs @@ -29,4 +29,8 @@ pub enum SchedulerInput { owner: TxnIdWire, reason: ReleaseReason, }, + /// A backup's consistent-cut marker carrying its watermark `hlc`. Every + /// transaction delivered before it must finish before the scheduler + /// reports it; every transaction delivered after it commits above `hlc`. + CutMarker { hlc: u64 }, } diff --git a/nodedb-cluster/src/calvin/types/sequencer.rs b/nodedb-cluster/src/calvin/types/sequencer.rs index 65d8a8607..238a3106a 100644 --- a/nodedb-cluster/src/calvin/types/sequencer.rs +++ b/nodedb-cluster/src/calvin/types/sequencer.rs @@ -107,7 +107,7 @@ pub struct EpochBatch { #[cfg(test)] mod tests { use nodedb_types::TenantId; - use nodedb_types::id::{DatabaseId, VShardId}; + use nodedb_types::id::{CollectionKey, DatabaseId}; use super::super::primitives::{EngineKeySet, SortedVec, VersionedReadSet}; use super::super::transaction::ReadWriteSet; @@ -133,7 +133,9 @@ mod tests { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); @@ -170,7 +172,7 @@ mod tests { }; let bytes = zerompk::to_msgpack_vec(&st).unwrap(); let mut decoded: SequencedTxn = zerompk::from_msgpack(&bytes).unwrap(); - decoded.tx_class.restore_derived(); + decoded.tx_class.restore_derived().expect("restore derived"); assert_eq!(st.epoch, decoded.epoch); assert_eq!(st.position, decoded.position); assert_eq!(st.epoch_system_ms, decoded.epoch_system_ms); @@ -205,7 +207,7 @@ mod tests { let bytes = zerompk::to_msgpack_vec(&batch).unwrap(); let mut decoded: EpochBatch = zerompk::from_msgpack(&bytes).unwrap(); for txn in &mut decoded.txns { - txn.tx_class.restore_derived(); + txn.tx_class.restore_derived().expect("restore derived"); } assert_eq!(batch.epoch, decoded.epoch); assert_eq!(batch.epoch_system_ms, decoded.epoch_system_ms); diff --git a/nodedb-cluster/src/calvin/types/transaction.rs b/nodedb-cluster/src/calvin/types/transaction.rs index e3f0d9770..bd681251e 100644 --- a/nodedb-cluster/src/calvin/types/transaction.rs +++ b/nodedb-cluster/src/calvin/types/transaction.rs @@ -6,7 +6,7 @@ //! representation submitted to the sequencer. use nodedb_types::TenantId; -use nodedb_types::id::{DatabaseId, VShardId}; +use nodedb_types::id::{CollectionKey, DatabaseId, VShardId}; use serde::{Deserialize, Serialize}; use crate::error::CalvinError; @@ -58,12 +58,20 @@ impl ReadWriteSet { /// This derivation is re-run on decode rather than serialized, so the /// serialized bytes remain deterministic regardless of how `VShardId` /// is computed. - pub fn participating_vshards(&self) -> Vec { + pub fn participating_vshards(&self) -> Result, CalvinError> { self.participating_vshards_in_database(DatabaseId::DEFAULT) } /// Derive participants using database-scoped collection homes. - pub fn participating_vshards_in_database(&self, database_id: DatabaseId) -> Vec { + /// + /// Key-set collection names are database-qualified, because the Data + /// Plane reads storage by them. Each one is de-qualified into a + /// [`CollectionKey`] before hashing, so the participant set matches the + /// vShard every other path homes the collection to. + pub fn participating_vshards_in_database( + &self, + database_id: DatabaseId, + ) -> Result, CalvinError> { let mut seen = std::collections::HashSet::new(); let mut result = Vec::new(); for engine_set in &self.0 { @@ -80,7 +88,8 @@ impl ReadWriteSet { | EngineKeySet::Vector { .. } | EngineKeySet::Kv { .. } => { let vshard = - VShardId::from_collection_in_database(database_id, engine_set.collection()); + CollectionKey::from_qualified_str(database_id, engine_set.collection())? + .vshard(); if seen.insert(vshard.as_u32()) { result.push(vshard); } @@ -88,7 +97,7 @@ impl ReadWriteSet { } } result.sort_by_key(|v| v.as_u32()); - result + Ok(result) } } @@ -304,7 +313,7 @@ impl TxClass { if write_set.is_empty() { return Err(CalvinError::EmptyWriteSet); } - let mut participating_vshards = write_set.participating_vshards_in_database(database_id); + let mut participating_vshards = write_set.participating_vshards_in_database(database_id)?; let min_participants = if allow_single_vshard { 1 } else { 2 }; // The participant FLOOR is computed from the WRITE set ONLY, and BEFORE // the read-set union below: a txn that writes a single shard but reads N @@ -323,7 +332,7 @@ impl TxClass { // `new_checked` and `restore_derived` — `participating_vshards` is // `#[serde(skip)]` and re-derived on decode, so an encoded and a decoded // `TxClass` would disagree on their participant set if the two diverged. - for v in read_set.participating_vshards_in_database(database_id) { + for v in read_set.participating_vshards_in_database(database_id)? { if !participating_vshards .iter() .any(|e| e.as_u32() == v.as_u32()) @@ -393,17 +402,20 @@ impl TxClass { /// Re-derive fields skipped during serialization. /// /// Call this immediately after deserializing a `TxClass` that came off - /// the wire or out of the Raft log. - pub fn restore_derived(&mut self) { + /// the wire or out of the Raft log. Fails when a key-set collection name + /// lacks the qualifier of `database_id`. Construction rejects such a + /// class, so a failure here means the decoded bytes are not a class any + /// constructor built. + pub fn restore_derived(&mut self) -> Result<(), CalvinError> { let mut vshards = self .write_set - .participating_vshards_in_database(self.database_id); + .participating_vshards_in_database(self.database_id)?; // Union the read set's participating vShards — MUST match `new_checked`'s // union exactly so a decoded `TxClass` derives the identical participant // set the encoder computed (participants are not serialized). for v in self .read_set - .participating_vshards_in_database(self.database_id) + .participating_vshards_in_database(self.database_id)? { if !vshards.iter().any(|e| e.as_u32() == v.as_u32()) { vshards.push(v); @@ -419,6 +431,7 @@ impl TxClass { // Final stable sort after all unions — lockstep with `new_checked`. vshards.sort_by_key(|v| v.as_u32()); self.participating_vshards = vshards; + Ok(()) } /// Set the lock-table owner id propagated to `SequencedTxn.lock_owner`. @@ -462,7 +475,9 @@ mod tests { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); @@ -491,14 +506,14 @@ mod tests { #[test] fn read_write_set_participating_vshards_distinct() { let ws = multi_vshard_write_set(); - let vshards = ws.participating_vshards(); + let vshards = ws.participating_vshards().expect("participants"); assert!(vshards.len() >= 2, "expected at least 2 distinct vShards"); } #[test] fn read_write_set_participating_vshards_sorted() { let ws = multi_vshard_write_set(); - let vshards = ws.participating_vshards(); + let vshards = ws.participating_vshards().expect("participants"); let ids: Vec = vshards.iter().map(|v| v.as_u32()).collect(); let mut sorted = ids.clone(); sorted.sort(); @@ -509,7 +524,7 @@ mod tests { fn read_write_set_same_collection_counted_once() { // Two EngineKeySets for the same collection: still one vshard. let ws = ReadWriteSet::new(vec![doc_set("users", vec![1]), vec_set("users", vec![1])]); - let vshards = ws.participating_vshards(); + let vshards = ws.participating_vshards().expect("participants"); assert_eq!(vshards.len(), 1); } @@ -611,7 +626,7 @@ mod tests { let first = sonic_rs::to_vec(&tc).unwrap(); let mut restored: TxClass = sonic_rs::from_slice(&first).unwrap(); - restored.restore_derived(); + restored.restore_derived().expect("restore derived"); let second = sonic_rs::to_vec(&restored).unwrap(); assert_eq!(first, second); @@ -624,7 +639,7 @@ mod tests { let tc = make_tx_class(multi_vshard_write_set()); let bytes = zerompk::to_msgpack_vec(&tc).unwrap(); let mut decoded: TxClass = zerompk::from_msgpack(&bytes).unwrap(); - decoded.restore_derived(); + decoded.restore_derived().expect("restore derived"); assert_eq!(tc.tenant_id, decoded.tenant_id); assert_eq!(tc.plans, decoded.plans); assert_eq!(tc.write_set, decoded.write_set); @@ -639,13 +654,19 @@ mod tests { // Pick a vshard id that's different from col_a and col_b. let passive_vshard_id = { - let a = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_a).as_u32(); - let b = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_b).as_u32(); + let a = CollectionKey::from_bare(DatabaseId::DEFAULT, &col_a) + .vshard() + .as_u32(); + let b = CollectionKey::from_bare(DatabaseId::DEFAULT, &col_b) + .vshard() + .as_u32(); // Find one that differs from both. let mut candidate = 9999u32; for i in 0u32..64 { let name = format!("passive_col_{i}"); - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let v = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if v != a && v != b { candidate = v; break; @@ -730,7 +751,7 @@ mod tests { let bytes = zerompk::to_msgpack_vec(&tx).expect("encode TxClass"); let mut decoded: TxClass = zerompk::from_msgpack(&bytes).expect("decode TxClass"); - decoded.restore_derived(); + decoded.restore_derived().expect("restore derived"); // Every read_lsn and the Point/Predicate distinction survive exactly. assert_eq!(decoded.versioned_reads, reads); @@ -771,7 +792,7 @@ mod tests { .expect("valid TxClass"); let bytes = zerompk::to_msgpack_vec(&tx).expect("encode"); let mut decoded: TxClass = zerompk::from_msgpack(&bytes).expect("decode"); - decoded.restore_derived(); + decoded.restore_derived().expect("restore derived"); assert_eq!(decoded.database_id, DatabaseId::new(9)); assert_eq!(decoded.participating_vshards(), tx.participating_vshards()); } @@ -796,7 +817,7 @@ mod tests { let bytes = zerompk::to_msgpack_vec(&legacy).expect("encode legacy"); let mut decoded: TxClass = zerompk::from_msgpack(&bytes).expect("decode legacy as TxClass"); - decoded.restore_derived(); + decoded.restore_derived().expect("restore derived"); assert!(decoded.versioned_reads.is_empty()); assert!(decoded.dependent_reads.is_none()); @@ -832,7 +853,9 @@ mod tests { // Pick a collection name whose collection-homed vShard differs from // both endpoint homes, to prove routing ignores the collection. - let coll_v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, "follows").as_u32(); + let coll_v = CollectionKey::from_bare(DatabaseId::DEFAULT, "follows") + .vshard() + .as_u32(); let ws = ReadWriteSet::new(vec![EngineKeySet::Edge { collection: "follows".to_owned(), @@ -842,6 +865,7 @@ mod tests { let mut got: Vec = ws .participating_vshards() + .expect("participants derive") .iter() .map(|v| v.as_u32()) .collect(); @@ -867,8 +891,9 @@ mod tests { collection: "users".to_owned(), surrogates: SortedVec::new(vec![7u32]), }]); - let want_vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "users").as_u32(); + let want_vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, "users") + .vshard() + .as_u32(); // Strict path still rejects. let strict = TxClass::new( @@ -914,7 +939,9 @@ mod tests { let mut first: Option<(String, u32)> = None; for i in 0u32..2048 { let name = format!("coll_{i}"); - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let v = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != v { return (fname.clone(), name); @@ -933,8 +960,12 @@ mod tests { // write-only floor still passes via the single-vshard opt-in), and the // union is reproduced identically on decode. let (wcoll, rcoll) = two_distinct_vshard_collections(); - let wv = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &wcoll).as_u32(); - let rv = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &rcoll).as_u32(); + let wv = CollectionKey::from_bare(DatabaseId::DEFAULT, &wcoll) + .vshard() + .as_u32(); + let rv = CollectionKey::from_bare(DatabaseId::DEFAULT, &rcoll) + .vshard() + .as_u32(); assert_ne!(wv, rv); let write_set = ReadWriteSet::new(vec![EngineKeySet::Document { @@ -975,7 +1006,7 @@ mod tests { // set (participants are `#[serde(skip)]`, re-derived in lockstep). let bytes = zerompk::to_msgpack_vec(&tx).expect("encode"); let mut decoded: TxClass = zerompk::from_msgpack(&bytes).expect("decode"); - decoded.restore_derived(); + decoded.restore_derived().expect("restore derived"); assert_eq!( tx.participating_vshards(), decoded.participating_vshards(), @@ -1020,9 +1051,27 @@ mod tests { }]); let got: Vec = ws .participating_vshards() + .expect("participants derive") .iter() .map(|v| v.as_u32()) .collect(); assert_eq!(got, vec![only]); } + + #[test] + fn participating_vshards_dequalify_named_database_collections() { + let db = DatabaseId::new(1024); + let qualified = nodedb_types::QualifiedCollection::new(db, "users"); + let ws = ReadWriteSet::new(vec![doc_set(qualified.as_str(), vec![1])]); + let vshards = ws + .participating_vshards_in_database(db) + .expect("participants"); + assert_eq!( + vshards, + vec![CollectionKey::from_bare(db, "users").vshard()] + ); + + let unqualified = ReadWriteSet::new(vec![doc_set("users", vec![1])]); + assert!(unqualified.participating_vshards_in_database(db).is_err()); + } } diff --git a/nodedb-cluster/src/circuit_breaker.rs b/nodedb-cluster/src/circuit_breaker.rs index bbffd3f1f..a38d24208 100644 --- a/nodedb-cluster/src/circuit_breaker.rs +++ b/nodedb-cluster/src/circuit_breaker.rs @@ -251,10 +251,51 @@ impl RetryPolicy { /// Determine if an error is retryable. /// - /// Only transport errors (connection failures, timeouts) are retried. - /// Codec errors, circuit-open errors, and application errors are not. + /// Only transport errors (connection failures) are retried. Codec errors, + /// circuit-open errors, shard timeouts, and application errors are not. pub fn is_retryable(err: &ClusterError) -> bool { - matches!(err, ClusterError::Transport { .. }) + match err { + ClusterError::Transport { .. } => true, + // A timed-out send may still have reached the peer, so resending + // it can apply the request twice. + ClusterError::ShardTimeout { .. } => false, + ClusterError::Raft(_) + | ClusterError::VShardNotMapped { .. } + | ClusterError::GroupNotFound { .. } + | ClusterError::LearnerNotCaughtUp { .. } + | ClusterError::MigrationInProgress { .. } + | ClusterError::MigrationPauseBudgetExceeded { .. } + | ClusterError::NodeUnreachable { .. } + | ClusterError::GhostNotFound { .. } + | ClusterError::StreamTerminal { .. } + | ClusterError::Storage { .. } + | ClusterError::DataPlane { .. } + | ClusterError::Codec { .. } + | ClusterError::UnsupportedWireVersion { .. } + | ClusterError::CircuitOpen { .. } + | ClusterError::JoinGroupDisappeared { .. } + | ClusterError::JoinCommitTimeout { .. } + | ClusterError::ReadIndexNotLeader { .. } + | ClusterError::ReadIndexTimeout { .. } + | ClusterError::Config { .. } + | ClusterError::MigrationCheckpoint(_) + | ClusterError::MigrationRecovery(_) + | ClusterError::WrongOwner { .. } + | ClusterError::Calvin(_) + | ClusterError::SnapshotCrcMismatch { .. } + | ClusterError::SnapshotOffsetRegression { .. } + | ClusterError::PartialSnapshotCorrupt { .. } + | ClusterError::PartialSnapshotCleanupFailed { .. } + | ClusterError::SnapshotApplyFailed { .. } + | ClusterError::Mirror(_) + | ClusterError::BspBarrier(_) + | ClusterError::VectorGather(_) + | ClusterError::SpatialGather(_) + | ClusterError::Bm25Gather(_) + | ClusterError::TsGather(_) + | ClusterError::RemoteUntyped { .. } + | ClusterError::ShardExecution { .. } => false, + } } } diff --git a/nodedb-cluster/src/diag/context.rs b/nodedb-cluster/src/diag/context.rs index cb1fb4e37..9ea6e171a 100644 --- a/nodedb-cluster/src/diag/context.rs +++ b/nodedb-cluster/src/diag/context.rs @@ -140,6 +140,41 @@ impl DomainContext for SequencerBackpressureDrop<'_> { } } +/// An epoch batch carries a transaction whose participant set cannot be +/// derived: a key-set collection name lacks the qualifier of the +/// transaction's database. No constructor builds such a class. +pub(super) struct SequencerParticipantsUnderivable<'a> { + pub epoch: u64, + pub raft_index: u64, + pub detail: &'a str, +} + +impl DomainContext for SequencerParticipantsUnderivable<'_> { + fn domain_kind(&self) -> &'static str { + "nodedb_cluster.sequencer_participants_underivable" + } + + fn grouping_key(&self) -> String { + // One bug: a class reached the log without passing construction. The + // epoch, index, and offending name are the occurrence. + "underivable".to_owned() + } + + fn to_json(&self) -> Value { + json!({ + "epoch": self.epoch, + "raft_index": self.raft_index, + "detail": self.detail, + "why_fatal": "the batch is skipped like an undecodable entry, so none of its \ + transactions is fanned out and their completion waiters never \ + resolve until their own deadlines elapse", + "operator_action": "find the proposer that built this transaction class without \ + a TxClass constructor, or the corruption that altered its \ + key-set collection names", + }) + } +} + #[cfg(test)] mod tests { use super::*; diff --git a/nodedb-cluster/src/diag/inert.rs b/nodedb-cluster/src/diag/inert.rs index 772595c95..9086ca417 100644 --- a/nodedb-cluster/src/diag/inert.rs +++ b/nodedb-cluster/src/diag/inert.rs @@ -25,3 +25,6 @@ pub fn sequencer_backpressure_drop( _drops: &[(u32, &'static str)], ) { } + +#[inline] +pub fn sequencer_participants_underivable(_epoch: u64, _raft_index: u64, _detail: &str) {} diff --git a/nodedb-cluster/src/diag/mod.rs b/nodedb-cluster/src/diag/mod.rs index 4bd095d86..301dde73b 100644 --- a/nodedb-cluster/src/diag/mod.rs +++ b/nodedb-cluster/src/diag/mod.rs @@ -30,7 +30,11 @@ mod recording; mod inert; #[cfg(all(feature = "diagnostics", not(target_arch = "wasm32")))] -pub use recording::{sequencer_backpressure_drop, sequencer_epoch_gap}; +pub use recording::{ + sequencer_backpressure_drop, sequencer_epoch_gap, sequencer_participants_underivable, +}; #[cfg(not(all(feature = "diagnostics", not(target_arch = "wasm32"))))] -pub use inert::{sequencer_backpressure_drop, sequencer_epoch_gap}; +pub use inert::{ + sequencer_backpressure_drop, sequencer_epoch_gap, sequencer_participants_underivable, +}; diff --git a/nodedb-cluster/src/diag/recording.rs b/nodedb-cluster/src/diag/recording.rs index 48684ee20..22b341cad 100644 --- a/nodedb-cluster/src/diag/recording.rs +++ b/nodedb-cluster/src/diag/recording.rs @@ -75,3 +75,23 @@ pub fn sequencer_backpressure_drop(epoch: u64, dropped_count: u64, drops: &[(u32 .with_backtrace() .emit(); } + +/// Report an epoch batch skipped because one of its transactions has an +/// underivable participant set. +/// +/// Called from the one site that detects it: `apply`'s `restore_derived` +/// pass over a decoded epoch batch. +pub fn sequencer_participants_underivable(epoch: u64, raft_index: u64, detail: &str) { + let ctx = context::SequencerParticipantsUnderivable { + epoch, + raft_index, + detail, + }; + let _ = Capture::new( + EventKind::InvariantViolation, + "Calvin sequencer: epoch batch carries a transaction with underivable participants", + ) + .domain(&ctx) + .with_backtrace() + .emit(); +} diff --git a/nodedb-cluster/src/distributed_array/scatter.rs b/nodedb-cluster/src/distributed_array/scatter.rs index 547b479f0..fbde79528 100644 --- a/nodedb-cluster/src/distributed_array/scatter.rs +++ b/nodedb-cluster/src/distributed_array/scatter.rs @@ -81,16 +81,53 @@ use super::rpc::ShardRpcDispatch; /// answers `WrongOwner` until the coordinator's routing table catches up (the /// single retry in `call_with_wrong_owner_retry` already re-reads the live /// table). Counting it as a liveness failure would open the shared breaker and -/// then fast-fail healthy shards' slice/put/agg/delete with `CircuitOpen`. Only -/// `WrongOwner` is excluded — every genuine transport/timeout/unreachable error -/// still counts, mirroring `RetryPolicy::is_retryable`'s conservative policy. +/// then fast-fail healthy shards' slice/put/agg/delete with `CircuitOpen`. +/// `WrongOwner` and a typed verdict from a shard that answered are excluded. +/// Every genuine transport/timeout/unreachable error still counts, mirroring +/// `RetryPolicy::is_retryable`'s conservative policy. fn counts_against_breaker(err: &ClusterError) -> bool { match err { ClusterError::WrongOwner { .. } => false, + // A typed verdict comes from a healthy shard that answered. + ClusterError::DataPlane { .. } + | ClusterError::ShardExecution { .. } + | ClusterError::StreamTerminal { .. } => false, // An unresponsive peer is exactly what the breaker exists to shed // load from, so a shard timeout counts like any other liveness failure. ClusterError::ShardTimeout { .. } => true, - _ => true, + ClusterError::Raft(_) + | ClusterError::VShardNotMapped { .. } + | ClusterError::GroupNotFound { .. } + | ClusterError::LearnerNotCaughtUp { .. } + | ClusterError::MigrationInProgress { .. } + | ClusterError::MigrationPauseBudgetExceeded { .. } + | ClusterError::NodeUnreachable { .. } + | ClusterError::GhostNotFound { .. } + | ClusterError::Transport { .. } + | ClusterError::Storage { .. } + | ClusterError::Codec { .. } + | ClusterError::UnsupportedWireVersion { .. } + | ClusterError::CircuitOpen { .. } + | ClusterError::JoinGroupDisappeared { .. } + | ClusterError::JoinCommitTimeout { .. } + | ClusterError::ReadIndexNotLeader { .. } + | ClusterError::ReadIndexTimeout { .. } + | ClusterError::Config { .. } + | ClusterError::MigrationCheckpoint(_) + | ClusterError::MigrationRecovery(_) + | ClusterError::Calvin(_) + | ClusterError::SnapshotCrcMismatch { .. } + | ClusterError::SnapshotOffsetRegression { .. } + | ClusterError::PartialSnapshotCorrupt { .. } + | ClusterError::PartialSnapshotCleanupFailed { .. } + | ClusterError::SnapshotApplyFailed { .. } + | ClusterError::Mirror(_) + | ClusterError::BspBarrier(_) + | ClusterError::VectorGather(_) + | ClusterError::SpatialGather(_) + | ClusterError::Bm25Gather(_) + | ClusterError::TsGather(_) + | ClusterError::RemoteUntyped { .. } => true, } } diff --git a/nodedb-cluster/src/error.rs b/nodedb-cluster/src/error.rs index 4650576f9..2aa8b43b7 100644 --- a/nodedb-cluster/src/error.rs +++ b/nodedb-cluster/src/error.rs @@ -16,6 +16,11 @@ pub enum CalvinError { )] SingleVshardTxn { vshard: u32 }, + /// A key-set collection name lacks the qualifier of the transaction's + /// database, so its vShard cannot be derived. + #[error("calvin key set: {0}")] + CollectionKey(#[from] nodedb_types::CollectionKeyError), + /// A sequencer-layer error. See [`crate::calvin::sequencer::error::SequencerError`] /// for the full variant set. #[error("sequencer error: {0}")] @@ -106,13 +111,23 @@ pub enum ClusterError { /// `dispatch_remote_stream`). `detail` is the `Debug` rendering for logs. #[error("streaming execution terminal error: {detail}")] StreamTerminal { - error: crate::rpc_codec::TypedClusterError, + error: Box, detail: String, }, #[error("storage error: {detail}")] Storage { detail: String }, + /// A shard's Data Plane refused the request with a typed verdict. + /// + /// The code crosses the node hop verbatim as `RaftRpc::VShardRefusal`, so + /// the coordinator renders the SQLSTATE a single-node execution renders. + /// The message uses the code's `Debug` form for logs only. + #[error("data plane refused the request: {code:?}")] + DataPlane { + code: crate::rpc_codec::DataPlaneErrorCode, + }, + #[error("codec error: {detail}")] Codec { detail: String }, @@ -213,4 +228,20 @@ pub enum ClusterError { #[error("timeseries gather error: {0}")] TsGather(#[from] crate::distributed_timeseries::TsGatherError), + + /// A remote node answered with an error whose type has no wire mirror. + /// `detail` is that error's message. + #[error("remote error: {detail}")] + RemoteUntyped { detail: String }, + + /// A shard's local execution failed with a classified error. + /// + /// `error` is the typed wire form of that error, so the coordinator + /// rebuilds the error and renders the SQLSTATE a single-node execution + /// renders. `detail` is the message with the shard's context, for logs. + #[error("shard execution error: {detail}")] + ShardExecution { + error: Box, + detail: String, + }, } diff --git a/nodedb-cluster/src/lib.rs b/nodedb-cluster/src/lib.rs index e17427a5a..0464926ea 100644 --- a/nodedb-cluster/src/lib.rs +++ b/nodedb-cluster/src/lib.rs @@ -108,8 +108,8 @@ pub use migration_executor::{ }; pub use multi_raft::{GroupStatus, MultiRaft}; pub use raft_loop::{ - AssignRemoteSurrogate, CalvinSubmit, CalvinSubmitInbox, CommitApplier, RaftLoop, - ReleaseReservation, ReserveRead, ShuffleAggregator, ShuffleConsumer, ShuffleProducer, + AssignRemoteSurrogate, AuthLeaseService, CalvinSubmit, CalvinSubmitInbox, CommitApplier, + RaftLoop, ReleaseReservation, ReserveRead, ShuffleAggregator, ShuffleConsumer, ShuffleProducer, ShuffleReceiver, SnapshotApplier, SnapshotBuilder, SnapshotQuarantineHook, VShardEnvelopeHandler, }; @@ -126,13 +126,14 @@ pub use rebalancer::{ pub use routing::RoutingTable; pub use routing_liveness::{NodeIdResolver, RoutingLivenessHook}; pub use rpc_codec::{ - AssignSurrogateRequest, AssignSurrogateResponse, DataPlaneErrorCode, JoinKeyPair, MacKey, - PartNodeEntry, RaftRpc, ReleaseReservationRequest, ReleaseReservationResponse, - ReserveReadRequest, ReserveReadResponse, ShuffleAggregateConsumeRequest, - ShuffleAggregateConsumeResponse, ShuffleConsumeRequest, ShuffleConsumeResponse, - ShuffleProduceRequest, ShuffleProduceResponse, ShufflePushChunk, ShufflePushEnd, - ShufflePushRequest, SortKey, SubmitCalvinInboxRequest, SubmitCalvinInboxResponse, - SubmitCalvinTxnRequest, SubmitCalvinTxnResponse, TypedClusterError, + AssignSurrogateRequest, AssignSurrogateResponse, AuthBarrierOutcome, AuthBarrierRequest, + AuthBarrierResponse, AuthLeaseRenewOutcome, AuthLeaseRenewRequest, AuthLeaseRenewResponse, + DataPlaneErrorCode, GroupCoverage, JoinKeyPair, MacKey, PartNodeEntry, RaftRpc, + ReleaseReservationRequest, ReleaseReservationResponse, ReserveReadRequest, ReserveReadResponse, + ShuffleAggregateConsumeRequest, ShuffleAggregateConsumeResponse, ShuffleConsumeRequest, + ShuffleConsumeResponse, ShuffleProduceRequest, ShuffleProduceResponse, ShufflePushChunk, + ShufflePushEnd, ShufflePushRequest, SortKey, SubmitCalvinInboxRequest, + SubmitCalvinInboxResponse, SubmitCalvinTxnRequest, SubmitCalvinTxnResponse, TypedClusterError, }; pub use topology::{ClusterTopology, NodeInfo, NodeState}; pub use transport::{ diff --git a/nodedb-cluster/src/multi_raft/core.rs b/nodedb-cluster/src/multi_raft/core.rs index bef7d79ae..d8c201b96 100644 --- a/nodedb-cluster/src/multi_raft/core.rs +++ b/nodedb-cluster/src/multi_raft/core.rs @@ -1,6 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 -//! `MultiRaft` struct, constructors, group lifecycle, tick, observability. +//! `MultiRaft` struct, constructors, group lifecycle and tick. use std::collections::HashMap; use std::path::PathBuf; @@ -16,40 +16,6 @@ use crate::error::{ClusterError, Result}; use crate::raft_storage::RedbLogStorage; use crate::routing::RoutingTable; -/// Snapshot of a single Raft group's state for observability. -#[derive(Debug, Clone, serde::Serialize)] -pub struct GroupStatus { - pub group_id: u64, - /// Role as a human-readable string ("Leader", "Follower", "Candidate", "Learner"). - pub role: String, - pub leader_id: u64, - pub term: u64, - pub commit_index: u64, - pub last_applied: u64, - pub last_log_index: u64, - /// Highest log index covered by the latest compacted snapshot. - /// Advances when the group's log is compacted past the start (gated - /// by `RaftConfig::log_compaction_threshold`). A non-zero value - /// means entries at or below it are no longer in the log and a - /// lagging peer below this index can only be caught up via - /// `InstallSnapshot`, never `AppendEntries`. - pub snapshot_index: u64, - pub member_count: usize, - pub learner_count: usize, - pub vshard_count: usize, -} - -/// Membership snapshot for a hosted Raft group. -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct GroupMembership { - pub group_id: u64, - pub leader_id: u64, - /// Voting members, including this node when it is a voter. - pub voters: Vec, - /// Non-voting learners, including this node when it is a learner. - pub learners: Vec, -} - /// Multi-Raft coordinator managing multiple Raft groups on a single node. /// /// This coordinator: @@ -157,6 +123,17 @@ impl MultiRaft { self } + /// The shortest election timeout a group on this node waits before it + /// campaigns. + pub fn election_timeout_min(&self) -> Duration { + self.election_timeout_min + } + + /// How often a leader on this node sends heartbeats. + pub fn heartbeat_interval(&self) -> Duration { + self.heartbeat_interval + } + /// Configure the auto-compaction threshold for every group created on /// this node. `None` disables auto-compaction (the default). See /// [`RaftConfig::log_compaction_threshold`]. @@ -299,207 +276,10 @@ impl MultiRaft { ids } - /// Snapshot the actual Raft membership rather than the vShard routing view. - pub fn group_membership(&self, group_id: u64) -> Option { - let node = self.groups.get(&group_id)?; - let mut voters = node.voters().to_vec(); - let mut learners = node.learners().to_vec(); - match node.role() { - nodedb_raft::NodeRole::Learner => learners.push(self.node_id), - nodedb_raft::NodeRole::Observer => {} - _ => voters.push(self.node_id), - } - voters.sort_unstable(); - voters.dedup(); - learners.sort_unstable(); - learners.dedup(); - Some(GroupMembership { - group_id, - leader_id: node.leader_id(), - voters, - learners, - }) - } - /// Mutable access to the underlying Raft groups (for testing / bootstrap). pub fn groups_mut(&mut self) -> &mut HashMap> { &mut self.groups } - - /// Snapshot of all Raft group states for observability. - pub fn group_statuses(&self) -> Vec { - let mut statuses = Vec::with_capacity(self.groups.len()); - for (&group_id, node) in &self.groups { - let vshard_count = self - .routing - .read() - .unwrap_or_else(|p| p.into_inner()) - .vshards_for_group(group_id) - .len(); - let self_is_voter = !matches!( - node.role(), - nodedb_raft::NodeRole::Learner | nodedb_raft::NodeRole::Observer - ); - - statuses.push(GroupStatus { - group_id, - role: format!("{:?}", node.role()), - leader_id: node.leader_id(), - term: node.current_term(), - commit_index: node.commit_index(), - last_applied: node.last_applied(), - last_log_index: node.last_log_index(), - snapshot_index: node.log_snapshot_index(), - member_count: node.voters().len() + usize::from(self_is_voter), - learner_count: node.learners().len() - + usize::from(node.role() == nodedb_raft::NodeRole::Learner), - vshard_count, - }); - } - statuses.sort_by_key(|s| s.group_id); - statuses - } - - /// Get the leader for a given vShard (from local group state). - pub fn leader_for_vshard(&self, vshard_id: u32) -> Result> { - let group_id = self - .routing - .read() - .unwrap_or_else(|p| p.into_inner()) - .group_for_vshard(vshard_id)?; - let node = self - .groups - .get(&group_id) - .ok_or(ClusterError::GroupNotFound { group_id })?; - let lid = node.leader_id(); - Ok(if lid == 0 { None } else { Some(lid) }) - } - - /// Whether THIS node is currently the leader of the data-group that owns - /// `vshard_id`. - /// - /// Maps the vshard to its Raft group via the routing table and reuses the - /// existing local leader-role check — no new election. Returns `false` when - /// the vshard has no group mapping or this node is a follower/learner for - /// the owning group. Used by the Calvin scheduler to stamp the per-node, - /// non-replicated `is_group_leader` dispatch flag so the OLLP optimistic-lock - /// verification runs only on the leader while every replica applies the same - /// predicted write-set (determinism). - pub fn vshard_role_is_leader(&self, vshard_id: u32) -> bool { - match self - .routing - .read() - .unwrap_or_else(|p| p.into_inner()) - .group_for_vshard(vshard_id) - { - Ok(group_id) => self.is_group_leader(group_id), - Err(_) => false, - } - } - - /// Propose a command to the Raft group that owns the given vShard. - /// - /// Returns `(group_id, log_index)` on success. - pub fn propose(&mut self, vshard_id: u32, data: Vec) -> Result<(u64, u64)> { - let group_id = self - .routing - .read() - .unwrap_or_else(|p| p.into_inner()) - .group_for_vshard(vshard_id)?; - let node = self - .groups - .get_mut(&group_id) - .ok_or(ClusterError::GroupNotFound { group_id })?; - let log_index = node.propose(data)?; - Ok((group_id, log_index)) - } - - /// Returns `true` if this node is currently the leader of `group_id`. - /// - /// Returns `false` when the group does not exist on this node or when the - /// node is a follower, candidate, or learner in the group. - pub fn is_group_leader(&self, group_id: u64) -> bool { - use nodedb_raft::state::NodeRole; - self.groups - .get(&group_id) - .map(|n| n.role() == NodeRole::Leader) - .unwrap_or(false) - } - - /// Propose a command directly to a specific Raft group (e.g. the - /// metadata group, which has no vShard mapping). - /// - /// Returns the committed log index on success. - pub fn propose_to_group(&mut self, group_id: u64, data: Vec) -> Result { - let node = self - .groups - .get_mut(&group_id) - .ok_or(ClusterError::GroupNotFound { group_id })?; - Ok(node.propose(data)?) - } - - /// Read committed log entries for a Raft group in the inclusive index - /// range `[lo, hi]`. - /// - /// `hi` is clamped to the group's `commit_index` so callers that pass - /// `u64::MAX` never read uncommitted entries. - /// - /// Used by the Calvin scheduler's rebuild path to replay sequenced - /// transactions from the sequencer Raft log after a restart. - /// - /// Returns `Err(ClusterError::Raft(RaftError::LogCompacted))` if `lo` - /// has been compacted into a snapshot (caller must install a snapshot - /// instead of replaying from log). - pub fn read_committed_entries( - &self, - group_id: u64, - lo: u64, - hi: u64, - ) -> Result> { - let node = self - .groups - .get(&group_id) - .ok_or(ClusterError::GroupNotFound { group_id })?; - let entries = node.log_entries_range(lo, hi)?; - Ok(entries.to_vec()) - } - - /// The lowest committed index still available in `group_id`'s retained log - /// (`snapshot_index + 1`), or `None` when the group is absent on this node. - /// - /// Used to arm a Calvin scheduler catch-up from the earliest replayable - /// sequencer index so its drain reads exactly the retained log and never - /// faults on a compacted range. - pub fn first_available_index(&self, group_id: u64) -> Option { - self.groups - .get(&group_id) - .map(|n| n.first_available_index()) - } - - /// Auto-compact a group's log if its configured threshold has been - /// reached, given the DATA-PLANE applied watermark `applied_index`. - /// - /// `applied_index` MUST be the index the data-plane state machine has - /// durably applied to (NOT raft's commit index). Compacting past an - /// unapplied index would let the `SnapshotBuilder` serialize - /// incomplete state and corrupt a lagging follower's snapshot. - /// - /// No-op (returns `Ok(false)`) when the group is absent on this node, - /// the threshold is `None`, or the retained-entry count is below the - /// threshold. Returns `Ok(true)` when a compaction was performed. - pub fn maybe_compact_group(&mut self, group_id: u64, applied_index: u64) -> Result { - // Defer compaction while a snapshot transfer for this group is in - // flight: advancing the snapshot boundary mid-transfer would corrupt - // the catching-up peer. The apply loop retries on the next applied - // entry, so the watermark still advances once the transfer completes. - if self.in_flight_snapshots.is_active(group_id) { - return Ok(false); - } - let Some(node) = self.groups.get_mut(&group_id) else { - return Ok(false); - }; - Ok(node.maybe_compact_log(applied_index)?) - } } // Re-export LogEntry so callers of `read_committed_entries` can name the type. @@ -508,6 +288,7 @@ pub use nodedb_raft::LogEntry; #[cfg(test)] mod tests { use super::*; + use crate::multi_raft::status::GroupMembership; use std::time::Instant; #[test] diff --git a/nodedb-cluster/src/multi_raft/mod.rs b/nodedb-cluster/src/multi_raft/mod.rs index e05a41916..e2c3ddc55 100644 --- a/nodedb-cluster/src/multi_raft/mod.rs +++ b/nodedb-cluster/src/multi_raft/mod.rs @@ -6,8 +6,10 @@ //! //! Split across files: //! - [`core`]: struct, constructors, group lifecycle (`add_group`, -//! `add_group_as_learner`), tick pipeline, routing accessors, -//! observability (`group_statuses`). +//! `add_group_as_learner`), tick pipeline, routing accessors. +//! - [`status`]: observability snapshots (`group_statuses`, +//! `group_membership`). +//! - [`proposals`]: leadership checks, proposals and log access. //! - [`rpc_dispatch`]: inbound RPC routing to the correct group and the //! corresponding response handlers. //! - [`conf_change`]: `propose_conf_change` / `apply_conf_change` with @@ -19,7 +21,10 @@ pub mod conf_change; pub mod core; pub mod membership; +pub mod proposals; pub mod read_index; pub mod rpc_dispatch; +pub mod status; -pub use core::{GroupStatus, MultiRaft, MultiRaftReady}; +pub use core::{MultiRaft, MultiRaftReady}; +pub use status::{GroupMembership, GroupStatus}; diff --git a/nodedb-cluster/src/multi_raft/proposals.rs b/nodedb-cluster/src/multi_raft/proposals.rs new file mode 100644 index 000000000..5dc69b9c0 --- /dev/null +++ b/nodedb-cluster/src/multi_raft/proposals.rs @@ -0,0 +1,149 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Leadership checks, proposals and log access on hosted Raft groups. + +use crate::error::{ClusterError, Result}; +use crate::multi_raft::core::MultiRaft; + +impl MultiRaft { + /// Get the leader for a given vShard (from local group state). + pub fn leader_for_vshard(&self, vshard_id: u32) -> Result> { + let group_id = self + .routing + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard(vshard_id)?; + let node = self + .groups + .get(&group_id) + .ok_or(ClusterError::GroupNotFound { group_id })?; + let lid = node.leader_id(); + Ok(if lid == 0 { None } else { Some(lid) }) + } + + /// Whether THIS node is currently the leader of the data-group that owns + /// `vshard_id`. + /// + /// Maps the vshard to its Raft group via the routing table and reuses the + /// existing local leader-role check — no new election. Returns `false` when + /// the vshard has no group mapping or this node is a follower/learner for + /// the owning group. Used by the Calvin scheduler to stamp the per-node, + /// non-replicated `is_group_leader` dispatch flag so the OLLP optimistic-lock + /// verification runs only on the leader while every replica applies the same + /// predicted write-set (determinism). + pub fn vshard_role_is_leader(&self, vshard_id: u32) -> bool { + match self + .routing + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard(vshard_id) + { + Ok(group_id) => self.is_group_leader(group_id), + Err(_) => false, + } + } + + /// Propose a command to the Raft group that owns the given vShard. + /// + /// Returns `(group_id, log_index)` on success. + pub fn propose(&mut self, vshard_id: u32, data: Vec) -> Result<(u64, u64)> { + let group_id = self + .routing + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard(vshard_id)?; + let node = self + .groups + .get_mut(&group_id) + .ok_or(ClusterError::GroupNotFound { group_id })?; + let log_index = node.propose(data)?; + Ok((group_id, log_index)) + } + + /// Returns `true` if this node is currently the leader of `group_id`. + /// + /// Returns `false` when the group does not exist on this node or when the + /// node is a follower, candidate, or learner in the group. + pub fn is_group_leader(&self, group_id: u64) -> bool { + use nodedb_raft::state::NodeRole; + self.groups + .get(&group_id) + .map(|n| n.role() == NodeRole::Leader) + .unwrap_or(false) + } + + /// Propose a command directly to a specific Raft group (e.g. the + /// metadata group, which has no vShard mapping). + /// + /// Returns the committed log index on success. + pub fn propose_to_group(&mut self, group_id: u64, data: Vec) -> Result { + let node = self + .groups + .get_mut(&group_id) + .ok_or(ClusterError::GroupNotFound { group_id })?; + Ok(node.propose(data)?) + } + + /// Read committed log entries for a Raft group in the inclusive index + /// range `[lo, hi]`. + /// + /// `hi` is clamped to the group's `commit_index` so callers that pass + /// `u64::MAX` never read uncommitted entries. + /// + /// Used by the Calvin scheduler's rebuild path to replay sequenced + /// transactions from the sequencer Raft log after a restart. + /// + /// Returns `Err(ClusterError::Raft(RaftError::LogCompacted))` if `lo` + /// has been compacted into a snapshot (caller must install a snapshot + /// instead of replaying from log). + pub fn read_committed_entries( + &self, + group_id: u64, + lo: u64, + hi: u64, + ) -> Result> { + let node = self + .groups + .get(&group_id) + .ok_or(ClusterError::GroupNotFound { group_id })?; + let entries = node.log_entries_range(lo, hi)?; + Ok(entries.to_vec()) + } + + /// The lowest committed index still available in `group_id`'s retained log + /// (`snapshot_index + 1`), or `None` when the group is absent on this node. + /// + /// Used to arm a Calvin scheduler catch-up from the earliest replayable + /// sequencer index so its drain reads exactly the retained log and never + /// faults on a compacted range. + pub fn first_available_index(&self, group_id: u64) -> Option { + self.groups + .get(&group_id) + .map(|n| n.first_available_index()) + } + + /// Auto-compact a group's log if its configured threshold has been + /// reached, given the DATA-PLANE applied watermark `applied_index`. + /// + /// `applied_index` MUST be the index the data-plane state machine has + /// durably applied to (NOT raft's commit index). Compacting past an + /// unapplied index would let the `SnapshotBuilder` serialize + /// incomplete state and corrupt a lagging follower's snapshot. + /// + /// No-op (returns `Ok(false)`) when the group is absent on this node, + /// the threshold is `None`, or the retained-entry count is below the + /// threshold. Returns `Ok(true)` when a compaction was performed. + pub fn maybe_compact_group(&mut self, group_id: u64, applied_index: u64) -> Result { + // Defer compaction while a snapshot transfer for this group is in + // flight: advancing the snapshot boundary mid-transfer would corrupt + // the catching-up peer. The apply loop retries on the next applied + // entry, so the watermark still advances once the transfer completes. + if self.in_flight_snapshots.is_active(group_id) { + return Ok(false); + } + let Some(node) = self.groups.get_mut(&group_id) else { + return Ok(false); + }; + Ok(node.maybe_compact_log(applied_index)?) + } +} diff --git a/nodedb-cluster/src/multi_raft/status.rs b/nodedb-cluster/src/multi_raft/status.rs new file mode 100644 index 000000000..081fbff1e --- /dev/null +++ b/nodedb-cluster/src/multi_raft/status.rs @@ -0,0 +1,97 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Observability snapshots of the Raft groups hosted on this node. + +use crate::multi_raft::core::MultiRaft; + +/// Snapshot of a single Raft group's state for observability. +#[derive(Debug, Clone, serde::Serialize)] +pub struct GroupStatus { + pub group_id: u64, + /// Role as a human-readable string ("Leader", "Follower", "Candidate", "Learner"). + pub role: String, + pub leader_id: u64, + pub term: u64, + pub commit_index: u64, + pub last_applied: u64, + pub last_log_index: u64, + /// Highest log index covered by the latest compacted snapshot. + /// Advances when the group's log is compacted past the start (gated + /// by `RaftConfig::log_compaction_threshold`). A non-zero value + /// means entries at or below it are no longer in the log and a + /// lagging peer below this index can only be caught up via + /// `InstallSnapshot`, never `AppendEntries`. + pub snapshot_index: u64, + pub member_count: usize, + pub learner_count: usize, + pub vshard_count: usize, +} + +/// Membership snapshot for a hosted Raft group. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct GroupMembership { + pub group_id: u64, + pub leader_id: u64, + /// Voting members, including this node when it is a voter. + pub voters: Vec, + /// Non-voting learners, including this node when it is a learner. + pub learners: Vec, +} + +impl MultiRaft { + /// Snapshot the actual Raft membership rather than the vShard routing view. + pub fn group_membership(&self, group_id: u64) -> Option { + let node = self.groups.get(&group_id)?; + let mut voters = node.voters().to_vec(); + let mut learners = node.learners().to_vec(); + match node.role() { + nodedb_raft::NodeRole::Learner => learners.push(self.node_id), + nodedb_raft::NodeRole::Observer => {} + _ => voters.push(self.node_id), + } + voters.sort_unstable(); + voters.dedup(); + learners.sort_unstable(); + learners.dedup(); + Some(GroupMembership { + group_id, + leader_id: node.leader_id(), + voters, + learners, + }) + } + + /// Snapshot of all Raft group states for observability. + pub fn group_statuses(&self) -> Vec { + let mut statuses = Vec::with_capacity(self.groups.len()); + for (&group_id, node) in &self.groups { + let vshard_count = self + .routing + .read() + .unwrap_or_else(|p| p.into_inner()) + .vshards_for_group(group_id) + .len(); + let self_is_voter = !matches!( + node.role(), + nodedb_raft::NodeRole::Learner | nodedb_raft::NodeRole::Observer + ); + + statuses.push(GroupStatus { + group_id, + role: format!("{:?}", node.role()), + leader_id: node.leader_id(), + term: node.current_term(), + commit_index: node.commit_index(), + last_applied: node.last_applied(), + last_log_index: node.last_log_index(), + snapshot_index: node.log_snapshot_index(), + member_count: node.voters().len() + usize::from(self_is_voter), + learner_count: node.learners().len() + + usize::from(node.role() == nodedb_raft::NodeRole::Learner), + vshard_count, + }); + } + statuses.sort_by_key(|s| s.group_id); + statuses + } +} diff --git a/nodedb-cluster/src/raft_loop/auth_lease.rs b/nodedb-cluster/src/raft_loop/auth_lease.rs new file mode 100644 index 000000000..be6e87e6f --- /dev/null +++ b/nodedb-cluster/src/raft_loop/auth_lease.rs @@ -0,0 +1,37 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Answer authorization lease renewals and barriers through the host hook. + +use crate::error::Result; +use crate::forward::PlanExecutor; +use crate::rpc_codec::{ + AuthBarrierOutcome, AuthBarrierRequest, AuthBarrierResponse, AuthLeaseRenewOutcome, + AuthLeaseRenewRequest, AuthLeaseRenewResponse, RaftRpc, +}; + +use super::loop_core::{CommitApplier, RaftLoop}; + +impl RaftLoop { + pub(super) async fn handle_auth_lease_renew_rpc( + &self, + req: AuthLeaseRenewRequest, + ) -> Result { + let response = match &self.auth_lease { + Some(service) => service.renew(req).await, + None => AuthLeaseRenewResponse { + outcome: AuthLeaseRenewOutcome::NotLeader { leader_hint: None }, + }, + }; + Ok(RaftRpc::AuthLeaseRenewResponse(response)) + } + + pub(super) async fn handle_auth_barrier_rpc(&self, req: AuthBarrierRequest) -> Result { + let response = match &self.auth_lease { + Some(service) => service.barrier(req).await, + None => AuthBarrierResponse { + outcome: AuthBarrierOutcome::NotLeader { leader_hint: None }, + }, + }; + Ok(RaftRpc::AuthBarrierResponse(response)) + } +} diff --git a/nodedb-cluster/src/raft_loop/auth_lease_hook.rs b/nodedb-cluster/src/raft_loop/auth_lease_hook.rs new file mode 100644 index 000000000..5772c2de8 --- /dev/null +++ b/nodedb-cluster/src/raft_loop/auth_lease_hook.rs @@ -0,0 +1,23 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Hook for the authorization lease service. +//! +//! `nodedb-cluster` cannot depend on `nodedb` (circular). The lease table, +//! the coverage rules and the barrier wait live in `nodedb` behind this +//! `Send + Sync` hook. The transport calls it when a lease renewal or an +//! authorization barrier reaches this node. Cluster-only tests leave the +//! `RaftLoop` field `None`, and such a request is answered `NotLeader`. + +use crate::rpc_codec::{ + AuthBarrierRequest, AuthBarrierResponse, AuthLeaseRenewRequest, AuthLeaseRenewResponse, +}; + +#[async_trait::async_trait] +pub trait AuthLeaseService: Send + Sync + 'static { + /// Grant or withhold the sender's lease from its coverage report. + async fn renew(&self, req: AuthLeaseRenewRequest) -> AuthLeaseRenewResponse; + + /// Hold the answer until no lease holder can plan against state older + /// than the request's targets. + async fn barrier(&self, req: AuthBarrierRequest) -> AuthBarrierResponse; +} diff --git a/nodedb-cluster/src/raft_loop/builder.rs b/nodedb-cluster/src/raft_loop/builder.rs index db681ea80..e3729119a 100644 --- a/nodedb-cluster/src/raft_loop/builder.rs +++ b/nodedb-cluster/src/raft_loop/builder.rs @@ -54,6 +54,7 @@ impl RaftLoop { calvin_submit_inbox: self.calvin_submit_inbox, reserve_read: self.reserve_read, release_reservation: self.release_reservation, + auth_lease: self.auth_lease, snapshot_builder: self.snapshot_builder, snapshot_applier: self.snapshot_applier, partial_snapshots: self.partial_snapshots, @@ -152,6 +153,17 @@ impl RaftLoop { self } + /// Attach the authorization lease service (builder chain). Lease + /// renewals and authorization barriers reaching this node are answered + /// through it. + pub fn with_auth_lease( + mut self, + service: Arc, + ) -> Self { + self.auth_lease = Some(service); + self + } + /// Attach the routed Calvin-submit hook (Cv1, builder chain). /// /// The supplied implementation (backed by `nodedb`'s Calvin sequencer inbox diff --git a/nodedb-cluster/src/raft_loop/handle_rpc/dispatch.rs b/nodedb-cluster/src/raft_loop/handle_rpc/dispatch.rs index c66d43cc0..775dbe7a1 100644 --- a/nodedb-cluster/src/raft_loop/handle_rpc/dispatch.rs +++ b/nodedb-cluster/src/raft_loop/handle_rpc/dispatch.rs @@ -41,6 +41,11 @@ impl RaftRpcHandler for RaftLoop { RaftRpc::MetadataProposeRequest(req) => self.handle_metadata_propose_rpc(req), // Data-group proposal forwarding. RaftRpc::DataProposeRequest(req) => self.handle_data_propose_rpc(req), + // Read index for a node that does not lead the group. + RaftRpc::ReadIndexRequest(req) => self.handle_read_index_rpc(req).await, + // Authorization lease renewal and barrier, answered by the host hook. + RaftRpc::AuthLeaseRenewRequest(req) => self.handle_auth_lease_renew_rpc(req).await, + RaftRpc::AuthBarrierRequest(req) => self.handle_auth_barrier_rpc(req).await, // VShardEnvelope — dispatch to registered handler (Event Plane, etc.). RaftRpc::VShardEnvelope(bytes) => self.handle_vshard_envelope_rpc(bytes).await, other => Err(ClusterError::Transport { diff --git a/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs b/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs index 83129fbe5..9d6058d43 100644 --- a/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs +++ b/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs @@ -3,10 +3,13 @@ //! Physical-plan execution (C-β), metadata/data propose forwarding, and //! VShardEnvelope routing RPC bodies. +use crate::calvin::SEQUENCER_GROUP_ID; use crate::error::{ClusterError, Result}; use crate::forward::{ChunkSink, PlanExecutor}; +use crate::multi_raft::MultiRaft; use crate::rpc_codec::{ - DataProposeRequest, ExecuteRequest, MetadataProposeRequest, RaftRpc, TypedClusterError, + DataProposeRequest, DataProposeResponse, ExecuteRequest, MetadataProposeRequest, ProposeTarget, + RaftRpc, TypedClusterError, VShardRefusal, }; use super::super::loop_core::{CommitApplier, RaftLoop}; @@ -37,35 +40,29 @@ impl RaftLoop { Ok(RaftRpc::MetadataProposeResponse(resp)) } - // Data-group proposal forwarding — apply locally if we are the - // data-group leader for the given vshard, otherwise return - // NotLeader with a hint so the forwarder can chase the redirect. + // Data-group and sequencer-group proposal forwarding — apply locally if + // we lead the target group, otherwise return NotLeader with a hint so the + // forwarder can chase the redirect. pub(super) fn handle_data_propose_rpc(&self, req: DataProposeRequest) -> Result { let resp = { let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); - match mr.propose(req.vshard_id, req.bytes) { - Ok((group_id, log_index)) => { - crate::rpc_codec::DataProposeResponse::ok(group_id, log_index) - } - Err(crate::error::ClusterError::Raft(nodedb_raft::RaftError::NotLeader { - leader_hint, - })) => crate::rpc_codec::DataProposeResponse::err("not leader", leader_hint), - Err(e) => crate::rpc_codec::DataProposeResponse::err(e.to_string(), None), - } + propose_forwarded(&mut mr, req) }; Ok(RaftRpc::DataProposeResponse(resp)) } // VShardEnvelope — dispatch to registered handler (Event Plane, etc.). + // Every handler error answers as a typed `VShardRefusal` frame, so the + // caller rebuilds the same `ClusterError` instead of seeing a closed + // stream. pub(super) async fn handle_vshard_envelope_rpc(&self, bytes: Vec) -> Result { - if let Some(ref handler) = self.vshard_handler { - let response_bytes = handler(bytes).await?; - Ok(RaftRpc::VShardEnvelope(response_bytes)) - } else { - Err(ClusterError::Transport { + let result = match self.vshard_handler { + Some(ref handler) => handler(bytes).await, + None => Err(ClusterError::Transport { detail: "VShardEnvelope handler not configured".into(), - }) - } + }), + }; + Ok(vshard_answer(result)) } // Streaming physical-plan execution (L4) — delegate to the PlanExecutor's @@ -79,3 +76,121 @@ impl RaftLoop { self.plan_executor.execute_plan_streaming(req, sink).await } } + +/// The frame that answers a VShardEnvelope request: the handler's response +/// envelope, or its typed error as a refusal. +fn vshard_answer(result: Result>) -> RaftRpc { + match result { + Ok(response_bytes) => RaftRpc::VShardEnvelope(response_bytes), + Err(error) => RaftRpc::VShardRefusal(VShardRefusal::from(error)), + } +} + +/// Propose a forwarded entry to its target group on this node. +/// +/// Answers `not leader` with the known leader as a hint when this node does +/// not lead the target group. +fn propose_forwarded(mr: &mut MultiRaft, req: DataProposeRequest) -> DataProposeResponse { + let proposed = match req.target { + ProposeTarget::VShard(vshard_id) => mr.propose(vshard_id, req.bytes), + ProposeTarget::Sequencer => mr + .propose_to_group(SEQUENCER_GROUP_ID, req.bytes) + .map(|log_index| (SEQUENCER_GROUP_ID, log_index)), + }; + match proposed { + Ok((group_id, log_index)) => DataProposeResponse::ok(group_id, log_index), + Err(error) => DataProposeResponse::refused(&error), + } +} + +#[cfg(test)] +mod tests { + use std::time::{Duration, Instant}; + + use super::*; + use crate::routing::RoutingTable; + + fn multi_raft_with_sequencer(dir: &std::path::Path) -> MultiRaft { + let mut mr = MultiRaft::new(1, RoutingTable::uniform(1, &[1], 1), dir.to_path_buf()); + mr.add_group(SEQUENCER_GROUP_ID, vec![]) + .expect("add sequencer group"); + mr + } + + fn elect_sequencer_leader(mr: &mut MultiRaft) { + if let Some(node) = mr.groups_mut().get_mut(&SEQUENCER_GROUP_ID) { + node.election_deadline_override(Instant::now() - Duration::from_millis(1)); + } + for _ in 0..20 { + mr.tick().expect("tick"); + if mr.is_group_leader(SEQUENCER_GROUP_ID) { + return; + } + } + panic!("sequencer group did not elect this single node"); + } + + fn sequencer_request() -> DataProposeRequest { + DataProposeRequest { + target: ProposeTarget::Sequencer, + bytes: vec![7, 7, 7], + } + } + + /// A handler error such as `WrongOwner` answers as a typed refusal the + /// caller rebuilds, not as a closed stream. + #[test] + fn a_handler_error_answers_as_a_typed_refusal() { + let answer = vshard_answer(Err(ClusterError::WrongOwner { + vshard_id: 7, + expected_owner_node: None, + })); + match answer { + RaftRpc::VShardRefusal(refusal) => assert!(matches!( + ClusterError::from(refusal.error), + ClusterError::WrongOwner { + vshard_id: 7, + expected_owner_node: None + } + )), + other => panic!("expected a refusal frame, got {other:?}"), + } + } + + #[test] + fn a_handler_response_answers_as_an_envelope() { + match vshard_answer(Ok(vec![1, 2, 3])) { + RaftRpc::VShardEnvelope(bytes) => assert_eq!(bytes, vec![1, 2, 3]), + other => panic!("expected a response envelope, got {other:?}"), + } + } + + #[test] + fn sequencer_target_is_proposed_to_the_sequencer_group_on_its_leader() { + let dir = tempfile::tempdir().expect("tempdir"); + let mut mr = multi_raft_with_sequencer(dir.path()); + elect_sequencer_leader(&mut mr); + let before = mr.last_log_index(SEQUENCER_GROUP_ID).unwrap_or(0); + + let resp = propose_forwarded(&mut mr, sequencer_request()); + + assert!(resp.success, "{}", resp.error_message); + assert_eq!(resp.group_id, SEQUENCER_GROUP_ID); + assert!(resp.log_index > before); + assert_eq!(mr.last_log_index(SEQUENCER_GROUP_ID), Some(resp.log_index)); + } + + #[test] + fn sequencer_target_on_a_non_leader_answers_not_leader() { + let dir = tempfile::tempdir().expect("tempdir"); + let mut mr = multi_raft_with_sequencer(dir.path()); + + let resp = propose_forwarded(&mut mr, sequencer_request()); + + assert!(!resp.success); + assert_eq!( + resp.refusal, + Some(crate::rpc_codec::ForwardedProposeRefusal::NotLeader) + ); + } +} diff --git a/nodedb-cluster/src/raft_loop/loop_core.rs b/nodedb-cluster/src/raft_loop/loop_core.rs index a6ff5b419..6c1bf03b0 100644 --- a/nodedb-cluster/src/raft_loop/loop_core.rs +++ b/nodedb-cluster/src/raft_loop/loop_core.rs @@ -231,6 +231,11 @@ pub struct RaftLoop { /// configured" error. pub(super) release_reservation: Option>, + /// Optional authorization lease service. When set (by the `nodedb` + /// binary via `with_auth_lease`), lease renewals and authorization + /// barriers are answered through it. + pub(super) auth_lease: Option>, + /// Optional per-group snapshot builder for the SEND path. /// /// When set (by the `nodedb` binary via `with_snapshot_builder`), the @@ -334,6 +339,7 @@ impl RaftLoop { calvin_submit_inbox: None, reserve_read: None, release_reservation: None, + auth_lease: None, snapshot_builder: None, snapshot_applier: None, partial_snapshots: Arc::new(std::sync::Mutex::new(std::collections::HashMap::new())), diff --git a/nodedb-cluster/src/raft_loop/mod.rs b/nodedb-cluster/src/raft_loop/mod.rs index b93c669f9..c4c6119b8 100644 --- a/nodedb-cluster/src/raft_loop/mod.rs +++ b/nodedb-cluster/src/raft_loop/mod.rs @@ -15,6 +15,8 @@ //! peer, propose `AddLearner` on every group, wait for commit, //! broadcast topology, persist catalog, build the wire response. +mod auth_lease; +pub mod auth_lease_hook; mod builder; pub mod handle_rpc; pub mod hooks; @@ -26,8 +28,10 @@ pub mod loop_core; mod membership_convergence; mod placement_reconcile; pub mod proposals; +mod read_index; pub mod tick; +pub use auth_lease_hook::AuthLeaseService; pub use hooks::{ AssignRemoteSurrogate, CalvinSubmit, CalvinSubmitInbox, ReleaseReservation, ReserveRead, ShuffleAggregator, ShuffleConsumer, ShuffleProducer, ShuffleReceiver, SnapshotApplier, diff --git a/nodedb-cluster/src/raft_loop/proposals.rs b/nodedb-cluster/src/raft_loop/proposals.rs index 8417a779e..50df25f26 100644 --- a/nodedb-cluster/src/raft_loop/proposals.rs +++ b/nodedb-cluster/src/raft_loop/proposals.rs @@ -244,7 +244,7 @@ impl RaftLoop { let req = crate::rpc_codec::RaftRpc::DataProposeRequest(crate::rpc_codec::DataProposeRequest { - vshard_id, + target: crate::rpc_codec::ProposeTarget::VShard(vshard_id), bytes: data, }); let resp = self.transport.send_rpc(leader_id, req).await?; @@ -252,16 +252,8 @@ impl RaftLoop { crate::rpc_codec::RaftRpc::DataProposeResponse(r) => { if r.success { Ok((r.group_id, r.log_index)) - } else if let Some(hint) = r.leader_hint { - Err(crate::error::ClusterError::Raft( - nodedb_raft::RaftError::NotLeader { - leader_hint: Some(hint), - }, - )) } else { - Err(crate::error::ClusterError::Transport { - detail: format!("data propose forward failed: {}", r.error_message), - }) + Err(r.refusal_error()) } } other => Err(crate::error::ClusterError::Transport { diff --git a/nodedb-cluster/src/raft_loop/read_index.rs b/nodedb-cluster/src/raft_loop/read_index.rs new file mode 100644 index 000000000..e6d6ca56f --- /dev/null +++ b/nodedb-cluster/src/raft_loop/read_index.rs @@ -0,0 +1,109 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Read index for any node: confirmed locally on the leader, asked of the +//! leader over the transport on every other node. +//! +//! A read index is the leader's commit index at a moment a quorum confirmed +//! its leadership. A node whose state machine has applied a group through +//! that index observes every entry committed before the read index was taken. + +use std::time::Duration; + +use crate::error::{ClusterError, Result}; +use crate::forward::PlanExecutor; +use crate::read_index_wait::confirm_read_index; +use crate::rpc_codec::{RaftRpc, ReadIndexOutcome, ReadIndexRequest, ReadIndexResponse}; + +use super::loop_core::{CommitApplier, RaftLoop}; + +/// Upper bound on the quorum wait a remote node may ask for. +const MAX_REMOTE_TIMEOUT: Duration = Duration::from_secs(10); + +impl RaftLoop { + /// Obtain a read index for `group_id`. + /// + /// On the group leader, confirms leadership against a quorum. On any other + /// node, asks the known leader. Fails with + /// [`ClusterError::ReadIndexNotLeader`] when no leader is known or the + /// asked node no longer leads, and with [`ClusterError::ReadIndexTimeout`] + /// when no quorum answered within `timeout`. + pub async fn read_index_via_leader(&self, group_id: u64, timeout: Duration) -> Result { + let leader = { + let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + if mr.is_group_leader(group_id) { + None + } else { + Some(mr.group_leader(group_id)) + } + }; + let leader_id = match leader { + None => return confirm_read_index(&self.multi_raft, group_id, timeout).await, + Some(id) if id == 0 || id == self.node_id => { + return Err(ClusterError::ReadIndexNotLeader { group_id }); + } + Some(id) => id, + }; + self.register_peer_addr(leader_id)?; + let request = RaftRpc::ReadIndexRequest(ReadIndexRequest { + group_id, + timeout_ms: u64::try_from(timeout.as_millis()).unwrap_or(u64::MAX), + }); + match self.transport.send_rpc(leader_id, request).await? { + RaftRpc::ReadIndexResponse(ReadIndexResponse { outcome }) => match outcome { + ReadIndexOutcome::Confirmed { read_index } => Ok(read_index), + ReadIndexOutcome::NotLeader { .. } => { + Err(ClusterError::ReadIndexNotLeader { group_id }) + } + ReadIndexOutcome::Timeout { waited_ms } => Err(ClusterError::ReadIndexTimeout { + group_id, + waited_ms, + }), + }, + other => Err(ClusterError::Transport { + detail: format!("read index: unexpected response variant {other:?}"), + }), + } + } + + /// Answer a remote node's [`ReadIndexRequest`] for a group this node may + /// lead. + pub(super) async fn handle_read_index_rpc(&self, req: ReadIndexRequest) -> Result { + let timeout = Duration::from_millis(req.timeout_ms).min(MAX_REMOTE_TIMEOUT); + let outcome = match confirm_read_index(&self.multi_raft, req.group_id, timeout).await { + Ok(read_index) => ReadIndexOutcome::Confirmed { read_index }, + Err(ClusterError::ReadIndexTimeout { waited_ms, .. }) => { + ReadIndexOutcome::Timeout { waited_ms } + } + Err(_) => { + let hint = self + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .group_leader(req.group_id); + ReadIndexOutcome::NotLeader { + leader_hint: (hint != 0).then_some(hint), + } + } + }; + Ok(RaftRpc::ReadIndexResponse(ReadIndexResponse { outcome })) + } + + /// Register `node_id`'s listen address with the transport, from the local + /// topology. + fn register_peer_addr(&self, node_id: u64) -> Result<()> { + let topo = self.topology.read().unwrap_or_else(|p| p.into_inner()); + let node = topo + .get_node(node_id) + .ok_or_else(|| ClusterError::Transport { + detail: format!("read index: leader {node_id} not in local topology"), + })?; + let addr = node.socket_addr().ok_or_else(|| ClusterError::Transport { + detail: format!( + "read index: leader {node_id} has unparseable addr {:?}", + node.addr + ), + })?; + self.transport.register_peer(node_id, addr); + Ok(()) + } +} diff --git a/nodedb-cluster/src/raft_loop/tick/apply_committed.rs b/nodedb-cluster/src/raft_loop/tick/apply_committed.rs index 61028aeb9..123d0fd55 100644 --- a/nodedb-cluster/src/raft_loop/tick/apply_committed.rs +++ b/nodedb-cluster/src/raft_loop/tick/apply_committed.rs @@ -72,13 +72,21 @@ impl RaftLoop { let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); if let Err(e) = mr.advance_applied(group_id, last_applied) { warn!(group_id, error = %e, "failed to advance applied index"); - } else if group_id == crate::metadata_group::METADATA_GROUP_ID { + } else if group_id == crate::metadata_group::METADATA_GROUP_ID + || group_id == crate::calvin::SEQUENCER_GROUP_ID + { // Metadata group: the metadata applier // applied entries synchronously to redb // before returning, so the apply // watermark is data-visible at this // point. Bump the watcher. // + // Sequencer group: the host applies each + // entry to the sequencer state machine + // inline, before returning. The watcher + // is the sequencer's applied index that + // authorization lease coverage reports. + // // Data groups are NOT bumped here — for // them `applier.apply_committed` only // enqueues entries onto the diff --git a/nodedb-cluster/src/routing.rs b/nodedb-cluster/src/routing.rs index 4a471f496..78b93e31f 100644 --- a/nodedb-cluster/src/routing.rs +++ b/nodedb-cluster/src/routing.rs @@ -2,7 +2,7 @@ use std::collections::HashMap; -use nodedb_types::id::{DatabaseId, VShardId}; +use nodedb_types::id::{CollectionKey, VShardId}; use crate::error::{ClusterError, Result}; @@ -298,17 +298,14 @@ impl RoutingTable { } } -/// Compute the primary vShard for a `(database, collection)` pair. +/// Compute the primary vShard for a collection. /// -/// Delegates to [`VShardId::from_collection_in_database`] so the cluster -/// routing layer and the types-layer hash function cannot drift. The database -/// id is folded into the hash so the same collection name in two different -/// databases routes to independent vShards — required for multi-database -/// isolation. Passing only the collection name (without `db`) would route -/// every database through the same vShard space and silently corrupt -/// cross-database deployments; the parameter is mandatory by design. -pub fn vshard_for_collection(database_id: DatabaseId, collection: &str) -> u32 { - VShardId::from_collection_in_database(database_id, collection).as_u32() +/// Delegates to [`VShardId::from_collection`], so the cluster routing layer +/// and the types-layer hash cannot drift. The [`CollectionKey`] carries the +/// database id and the bare catalog name, so a qualified name can never reach +/// the hash. +pub fn vshard_for_collection(key: CollectionKey<'_>) -> u32 { + VShardId::from_collection(key).as_u32() } /// FNV-1a 64-bit hash for deterministic key partitioning. @@ -467,11 +464,12 @@ mod tests { // gateway while the data plane still keys them by the correct // hash. This test pins the contract. for db_raw in [0u64, 1, 2, 1024, 999_999] { - let db = DatabaseId::new(db_raw); + let db = nodedb_types::id::DatabaseId::new(db_raw); for name in ["users", "orders", "events", "a", "this_is_a_long_name"] { + let key = CollectionKey::from_bare(db, name); assert_eq!( - vshard_for_collection(db, name), - VShardId::from_collection_in_database(db, name).as_u32(), + vshard_for_collection(key), + VShardId::from_collection(key).as_u32(), "drift detected: db={db_raw} collection={name}" ); } @@ -484,8 +482,11 @@ mod tests { // different vShards (probabilistic; "users" is a canonical // example whose hashes are known to differ across DEFAULT and // DatabaseId(1024)). - let v_default = vshard_for_collection(DatabaseId::DEFAULT, "users"); - let v_other = vshard_for_collection(DatabaseId::new(1024), "users"); + use nodedb_types::id::DatabaseId; + let v_default = + vshard_for_collection(CollectionKey::from_bare(DatabaseId::DEFAULT, "users")); + let v_other = + vshard_for_collection(CollectionKey::from_bare(DatabaseId::new(1024), "users")); assert_ne!( v_default, v_other, "same collection name across databases must route independently" diff --git a/nodedb-cluster/src/rpc_codec/auth_lease.rs b/nodedb-cluster/src/rpc_codec/auth_lease.rs new file mode 100644 index 000000000..219f8f9c0 --- /dev/null +++ b/nodedb-cluster/src/rpc_codec/auth_lease.rs @@ -0,0 +1,221 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Authorization lease wire types and codecs. +//! +//! Every node plans permission-checked statements from its local +//! authorization state only while it holds a lease from the metadata group +//! leader. A node renews its lease with a coverage report: for each Raft group +//! it names, the index through which its local authorization state holds +//! every change. +//! +//! A writer acknowledges an authorization change only after a barrier on the +//! metadata leader releases it. The barrier releases once every node holding +//! an unexpired lease reported coverage of the change, or its lease expired. + +use super::discriminants::*; +use super::header::write_frame; +use super::raft_rpc::RaftRpc; +use crate::error::{ClusterError, Result}; + +/// A holder's claim on one Raft group: its local authorization state holds +/// every change of `group_id` at or below `through`. +#[derive(Debug, Clone, Copy, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct GroupCoverage { + pub group_id: u64, + pub through: u64, +} + +/// Renew the sender's authorization lease. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct AuthLeaseRenewRequest { + pub node_id: u64, + /// Coverage of every group the sender knows. A group it does not + /// replicate is reported at `u64::MAX`: the sender never plans against it. + pub coverage: Vec, +} + +/// The leader's answer to an [`AuthLeaseRenewRequest`]. +#[derive(Debug, Clone, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum AuthLeaseRenewOutcome { + /// The lease runs `lease_ms` from the moment the sender sent the request. + Granted { lease_ms: u64 }, + /// The report does not cover every acknowledged change. The sender's + /// lease is not extended. + Withheld, + /// The receiver does not lead the metadata group. + NotLeader { leader_hint: Option }, +} + +/// Response to an [`AuthLeaseRenewRequest`]. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct AuthLeaseRenewResponse { + pub outcome: AuthLeaseRenewOutcome, +} + +/// Hold the sender's acknowledgement until every lease holder covers +/// `targets`, or its lease expired. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct AuthBarrierRequest { + pub targets: Vec, + /// How long the leader may hold the request. + pub timeout_ms: u64, +} + +/// The leader's answer to an [`AuthBarrierRequest`]. +#[derive(Debug, Clone, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum AuthBarrierOutcome { + /// No node can plan against state older than the targets. + Released, + /// The receiver does not lead the metadata group. + NotLeader { leader_hint: Option }, + /// The barrier did not release in time. + Timeout { waited_ms: u64 }, +} + +/// Response to an [`AuthBarrierRequest`]. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct AuthBarrierResponse { + pub outcome: AuthBarrierOutcome, +} + +macro_rules! to_bytes { + ($msg:expr) => { + rkyv::to_bytes::($msg) + .map(|b| b.to_vec()) + .map_err(|e| ClusterError::Codec { + detail: format!("rkyv serialize: {e}"), + }) + }; +} + +macro_rules! from_bytes { + ($payload:expr, $T:ty, $name:expr) => {{ + let mut aligned = rkyv::util::AlignedVec::<16>::with_capacity($payload.len()); + aligned.extend_from_slice($payload); + rkyv::from_bytes::<$T, rkyv::rancor::Error>(&aligned).map_err(|e| ClusterError::Codec { + detail: format!("rkyv deserialize {}: {e}", $name), + }) + }}; +} + +pub(super) fn encode_renew_req(msg: &AuthLeaseRenewRequest, out: &mut Vec) -> Result<()> { + write_frame(RPC_AUTH_LEASE_RENEW_REQ, &to_bytes!(msg)?, out) +} +pub(super) fn encode_renew_resp(msg: &AuthLeaseRenewResponse, out: &mut Vec) -> Result<()> { + write_frame(RPC_AUTH_LEASE_RENEW_RESP, &to_bytes!(msg)?, out) +} +pub(super) fn encode_barrier_req(msg: &AuthBarrierRequest, out: &mut Vec) -> Result<()> { + write_frame(RPC_AUTH_BARRIER_REQ, &to_bytes!(msg)?, out) +} +pub(super) fn encode_barrier_resp(msg: &AuthBarrierResponse, out: &mut Vec) -> Result<()> { + write_frame(RPC_AUTH_BARRIER_RESP, &to_bytes!(msg)?, out) +} + +pub(super) fn decode_renew_req(payload: &[u8]) -> Result { + Ok(RaftRpc::AuthLeaseRenewRequest(from_bytes!( + payload, + AuthLeaseRenewRequest, + "AuthLeaseRenewRequest" + )?)) +} +pub(super) fn decode_renew_resp(payload: &[u8]) -> Result { + Ok(RaftRpc::AuthLeaseRenewResponse(from_bytes!( + payload, + AuthLeaseRenewResponse, + "AuthLeaseRenewResponse" + )?)) +} +pub(super) fn decode_barrier_req(payload: &[u8]) -> Result { + Ok(RaftRpc::AuthBarrierRequest(from_bytes!( + payload, + AuthBarrierRequest, + "AuthBarrierRequest" + )?)) +} +pub(super) fn decode_barrier_resp(payload: &[u8]) -> Result { + Ok(RaftRpc::AuthBarrierResponse(from_bytes!( + payload, + AuthBarrierResponse, + "AuthBarrierResponse" + )?)) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::cluster_epoch::ClusterEpochState; + use crate::rpc_codec::{decode, encode}; + + fn roundtrip(rpc: RaftRpc) -> RaftRpc { + let epoch = ClusterEpochState::default(); + let encoded = encode(&rpc, &epoch).expect("encode"); + decode(&encoded, &epoch).expect("decode") + } + + fn coverage() -> Vec { + vec![ + GroupCoverage { + group_id: 0, + through: 41, + }, + GroupCoverage { + group_id: 3, + through: u64::MAX, + }, + ] + } + + #[test] + fn a_renewal_survives_the_wire() { + match roundtrip(RaftRpc::AuthLeaseRenewRequest(AuthLeaseRenewRequest { + node_id: 2, + coverage: coverage(), + })) { + RaftRpc::AuthLeaseRenewRequest(req) => { + assert_eq!(req.node_id, 2); + assert_eq!(req.coverage, coverage()); + } + other => panic!("decoded the wrong variant: {other:?}"), + } + for outcome in [ + AuthLeaseRenewOutcome::Granted { lease_ms: 150 }, + AuthLeaseRenewOutcome::Withheld, + AuthLeaseRenewOutcome::NotLeader { + leader_hint: Some(1), + }, + ] { + match roundtrip(RaftRpc::AuthLeaseRenewResponse(AuthLeaseRenewResponse { + outcome: outcome.clone(), + })) { + RaftRpc::AuthLeaseRenewResponse(resp) => assert_eq!(resp.outcome, outcome), + other => panic!("decoded the wrong variant: {other:?}"), + } + } + } + + #[test] + fn a_barrier_survives_the_wire() { + match roundtrip(RaftRpc::AuthBarrierRequest(AuthBarrierRequest { + targets: coverage(), + timeout_ms: 5000, + })) { + RaftRpc::AuthBarrierRequest(req) => { + assert_eq!(req.targets, coverage()); + assert_eq!(req.timeout_ms, 5000); + } + other => panic!("decoded the wrong variant: {other:?}"), + } + for outcome in [ + AuthBarrierOutcome::Released, + AuthBarrierOutcome::NotLeader { leader_hint: None }, + AuthBarrierOutcome::Timeout { waited_ms: 5000 }, + ] { + match roundtrip(RaftRpc::AuthBarrierResponse(AuthBarrierResponse { + outcome: outcome.clone(), + })) { + RaftRpc::AuthBarrierResponse(resp) => assert_eq!(resp.outcome, outcome), + other => panic!("decoded the wrong variant: {other:?}"), + } + } + } +} diff --git a/nodedb-cluster/src/rpc_codec/data_plane_error.rs b/nodedb-cluster/src/rpc_codec/data_plane_error.rs index 45927be57..506911ffc 100644 --- a/nodedb-cluster/src/rpc_codec/data_plane_error.rs +++ b/nodedb-cluster/src/rpc_codec/data_plane_error.rs @@ -72,8 +72,9 @@ pub enum DataPlaneErrorCode { collection: String, detail: String, }, - OverflowError { + CounterFault { collection: String, + fault: DataPlaneCounterFault, }, InsufficientBalance { collection: String, @@ -112,6 +113,15 @@ pub enum DataPlaneErrorCode { limit: u64, }, DivisionByZero, + /// Expression evaluation called a function no evaluator implements. + UndefinedFunction { + name: String, + }, + /// A function received an argument it cannot compute on (SQLSTATE + /// `22000`). `detail` names the function and the value. + DataException { + detail: String, + }, /// A period-lock reference row exists but does not carry the /// configured `status_column` — a misconfigured column name, not a /// locked period. @@ -121,4 +131,66 @@ pub enum DataPlaneErrorCode { status_column: String, row_identity: String, }, + /// The bridge dispatcher refused the request at a capacity limit; nothing + /// was enqueued. `reason` names the limit and its counts. + DispatchCapacity { + reason: String, + }, + /// The request's deadline passed before the core started it; nothing + /// ran. + ExpiredBeforeExecution, + /// A sync frame the validator refused for good. Nothing applied. The + /// four provenance fields name the stream position the high-water mark + /// advanced to. + SyncRejected { + violation: nodedb_types::sync::violation::ViolationType, + applied_seq: u64, + producer_id: u64, + epoch: u64, + stream_id: u64, + seq: u64, + }, + /// A sync frame the idempotency gate held back. Nothing applied, and + /// the stream's mark did not move. + SyncNotApplied { + hold: DataPlaneSyncHold, + applied_seq: u64, + }, + /// The request itself is malformed (SQLSTATE `42601`). + BadRequest { + detail: String, + }, + /// The whole transaction aborted before any read-set was validated + /// (SQLSTATE `40000`). + TransactionRollback { + detail: String, + }, + /// The statement cannot run in the current transaction state (SQLSTATE + /// `25001`). + ActiveSqlTransaction { + detail: String, + }, + /// A DROP refused because other objects depend on `object` (SQLSTATE + /// `2BP01`). + DependentObjectsExist { + object: String, + detail: String, + }, +} + +/// Wire mirror of `nodedb::bridge::envelope::SyncHold`. +#[derive(Debug, Clone, Copy, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum DataPlaneSyncHold { + Duplicate, + Fenced, + Gap { expected: u64 }, +} + +/// Wire mirror of `nodedb_physical::kv_atomic::CounterFault`. +#[derive(Debug, Clone, Copy, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum DataPlaneCounterFault { + NotAnInteger, + NotAFloat, + IntegerOverflow, + NonFinite, } diff --git a/nodedb-cluster/src/rpc_codec/data_propose.rs b/nodedb-cluster/src/rpc_codec/data_propose.rs index 25b61c18d..f1fcb4831 100644 --- a/nodedb-cluster/src/rpc_codec/data_propose.rs +++ b/nodedb-cluster/src/rpc_codec/data_propose.rs @@ -2,26 +2,47 @@ //! DataProposeRequest / DataProposeResponse wire types and codecs. //! -//! Used to forward data-group (non-metadata) Raft proposals from a follower -//! node to the group leader. The leader applies the proposal locally and -//! returns `(group_id, log_index)` so the forwarder can register a -//! `ProposeTracker` waiter and await commit. +//! Used to forward a non-metadata Raft proposal from a node that does not +//! lead the target group to the group leader. The target is a vShard's data +//! group or the Calvin sequencer group. The leader applies the proposal +//! locally and returns `(group_id, log_index)`. use super::discriminants::*; use super::header::write_frame; use super::raft_rpc::RaftRpc; use crate::error::{ClusterError, Result}; -/// Forward an opaque data-group proposal payload to the data-group leader. -/// -/// `vshard_id` identifies the vShard (and thus the Raft group) the entry -/// belongs to. `bytes` is the serialized `ReplicatedEntry`. +/// The Raft group a forwarded proposal is for. +#[derive(Debug, Clone, Copy, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum ProposeTarget { + /// The data group that owns this vShard. The bytes are a serialized + /// `ReplicatedEntry`. + VShard(u32), + /// The Calvin sequencer group. The bytes are a msgpack-encoded + /// `SequencerEntry`. + Sequencer, +} + +/// Forward an opaque proposal payload to the leader of its target group. #[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] pub struct DataProposeRequest { - pub vshard_id: u32, + pub target: ProposeTarget, pub bytes: Vec, } +/// Why a leader refused a forwarded proposal, typed so the forwarding node +/// can tell a transient refusal from a final one. +#[derive(Debug, Clone, Copy, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum ForwardedProposeRefusal { + /// The node does not lead the target group. `leader_hint` names the + /// leader it knows, if any. + NotLeader, + /// A leadership transfer of the target group is in flight. + LeadershipTransferInProgress, + /// Any other failure; `error_message` describes it. + Failed, +} + /// Response to a forwarded data-group proposal. #[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] pub struct DataProposeResponse { @@ -29,6 +50,8 @@ pub struct DataProposeResponse { pub group_id: u64, pub log_index: u64, pub leader_hint: Option, + /// The typed reason of a refusal. `None` on success. + pub refusal: Option, pub error_message: String, } @@ -39,17 +62,98 @@ impl DataProposeResponse { group_id, log_index, leader_hint: None, + refusal: None, error_message: String::new(), } } - pub fn err(message: impl Into, leader_hint: Option) -> Self { + /// The response for a proposal the leader refused with `error`. + pub fn refused(error: &ClusterError) -> Self { + let (refusal, leader_hint) = match error { + ClusterError::Raft(nodedb_raft::RaftError::NotLeader { leader_hint }) => { + (ForwardedProposeRefusal::NotLeader, *leader_hint) + } + ClusterError::Raft(nodedb_raft::RaftError::LeadershipTransferInProgress) => { + (ForwardedProposeRefusal::LeadershipTransferInProgress, None) + } + // A refusal with no retry contract. The forwarding node reads it + // as a transport error carrying the leader's message. + ClusterError::Raft( + nodedb_raft::RaftError::LogCompacted { .. } + | nodedb_raft::RaftError::CompactionAheadOfApplied { .. } + | nodedb_raft::RaftError::ProposalRejected { .. } + | nodedb_raft::RaftError::InvalidTransferTarget { .. } + | nodedb_raft::RaftError::GroupNotFound { .. } + | nodedb_raft::RaftError::Transport { .. } + | nodedb_raft::RaftError::Storage { .. } + | nodedb_raft::RaftError::Serialization { .. } + | nodedb_raft::RaftError::SnapshotFormat { .. } + | nodedb_raft::RaftError::Shutdown, + ) + | ClusterError::VShardNotMapped { .. } + | ClusterError::GroupNotFound { .. } + | ClusterError::LearnerNotCaughtUp { .. } + | ClusterError::MigrationInProgress { .. } + | ClusterError::MigrationPauseBudgetExceeded { .. } + | ClusterError::NodeUnreachable { .. } + | ClusterError::GhostNotFound { .. } + | ClusterError::Transport { .. } + | ClusterError::ShardTimeout { .. } + | ClusterError::StreamTerminal { .. } + | ClusterError::Storage { .. } + | ClusterError::DataPlane { .. } + | ClusterError::Codec { .. } + | ClusterError::UnsupportedWireVersion { .. } + | ClusterError::CircuitOpen { .. } + | ClusterError::JoinGroupDisappeared { .. } + | ClusterError::JoinCommitTimeout { .. } + | ClusterError::ReadIndexNotLeader { .. } + | ClusterError::ReadIndexTimeout { .. } + | ClusterError::Config { .. } + | ClusterError::MigrationCheckpoint(_) + | ClusterError::MigrationRecovery(_) + | ClusterError::WrongOwner { .. } + | ClusterError::Calvin(_) + | ClusterError::SnapshotCrcMismatch { .. } + | ClusterError::SnapshotOffsetRegression { .. } + | ClusterError::PartialSnapshotCorrupt { .. } + | ClusterError::PartialSnapshotCleanupFailed { .. } + | ClusterError::SnapshotApplyFailed { .. } + | ClusterError::Mirror(_) + | ClusterError::BspBarrier(_) + | ClusterError::VectorGather(_) + | ClusterError::SpatialGather(_) + | ClusterError::Bm25Gather(_) + | ClusterError::TsGather(_) + | ClusterError::RemoteUntyped { .. } + | ClusterError::ShardExecution { .. } => (ForwardedProposeRefusal::Failed, None), + }; Self { success: false, group_id: 0, log_index: 0, leader_hint, - error_message: message.into(), + refusal: Some(refusal), + error_message: error.to_string(), + } + } + + /// The typed error a refused response stands for on the forwarding node. + /// A refusal the leader typed keeps its Raft error; any other stays a + /// transport error carrying the leader's message. + pub fn refusal_error(&self) -> ClusterError { + match self.refusal { + Some(ForwardedProposeRefusal::NotLeader) => { + ClusterError::Raft(nodedb_raft::RaftError::NotLeader { + leader_hint: self.leader_hint, + }) + } + Some(ForwardedProposeRefusal::LeadershipTransferInProgress) => { + ClusterError::Raft(nodedb_raft::RaftError::LeadershipTransferInProgress) + } + Some(ForwardedProposeRefusal::Failed) | None => ClusterError::Transport { + detail: format!("data propose forward failed: {}", self.error_message), + }, } } } @@ -95,3 +199,80 @@ pub(super) fn decode_data_propose_resp(payload: &[u8]) -> Result { "DataProposeResponse" )?)) } + +#[cfg(test)] +mod tests { + use super::*; + use crate::cluster_epoch::ClusterEpochState; + use crate::rpc_codec::{decode, encode}; + + fn roundtrip(target: ProposeTarget) -> DataProposeRequest { + let rpc = RaftRpc::DataProposeRequest(DataProposeRequest { + target, + bytes: vec![1, 2, 3], + }); + let epoch = ClusterEpochState::default(); + let encoded = encode(&rpc, &epoch).expect("encode"); + match decode(&encoded, &epoch).expect("decode") { + RaftRpc::DataProposeRequest(req) => req, + other => panic!("decoded the wrong variant: {other:?}"), + } + } + + #[test] + fn sequencer_target_survives_the_wire() { + let req = roundtrip(ProposeTarget::Sequencer); + assert_eq!(req.target, ProposeTarget::Sequencer); + assert_eq!(req.bytes, vec![1, 2, 3]); + } + + #[test] + fn vshard_target_survives_the_wire() { + let req = roundtrip(ProposeTarget::VShard(42)); + assert_eq!(req.target, ProposeTarget::VShard(42)); + } + + fn refusal_across_the_wire(error: ClusterError) -> ClusterError { + let rpc = RaftRpc::DataProposeResponse(DataProposeResponse::refused(&error)); + let epoch = ClusterEpochState::default(); + let encoded = encode(&rpc, &epoch).expect("encode"); + match decode(&encoded, &epoch).expect("decode") { + RaftRpc::DataProposeResponse(resp) => { + assert!(!resp.success); + resp.refusal_error() + } + other => panic!("decoded the wrong variant: {other:?}"), + } + } + + #[test] + fn a_transfer_in_progress_keeps_its_raft_error_across_the_wire() { + let error = refusal_across_the_wire(ClusterError::Raft( + nodedb_raft::RaftError::LeadershipTransferInProgress, + )); + assert!(matches!( + error, + ClusterError::Raft(nodedb_raft::RaftError::LeadershipTransferInProgress) + )); + } + + #[test] + fn a_not_leader_keeps_its_hint_across_the_wire() { + let error = + refusal_across_the_wire(ClusterError::Raft(nodedb_raft::RaftError::NotLeader { + leader_hint: Some(3), + })); + assert!(matches!( + error, + ClusterError::Raft(nodedb_raft::RaftError::NotLeader { + leader_hint: Some(3) + }) + )); + } + + #[test] + fn any_other_refusal_stays_a_transport_error() { + let error = refusal_across_the_wire(ClusterError::VShardNotMapped { vshard_id: 7 }); + assert!(matches!(error, ClusterError::Transport { .. })); + } +} diff --git a/nodedb-cluster/src/rpc_codec/discriminants.rs b/nodedb-cluster/src/rpc_codec/discriminants.rs index 5b706aa20..1682fcff9 100644 --- a/nodedb-cluster/src/rpc_codec/discriminants.rs +++ b/nodedb-cluster/src/rpc_codec/discriminants.rs @@ -134,6 +134,27 @@ pub const RPC_RELEASE_RESERVATION_RESP: u8 = 44; pub const RPC_PRE_VOTE_REQ: u8 = 45; pub const RPC_PRE_VOTE_RESP: u8 = 46; +/// Routed read index. A node that does not lead a Raft group sends a +/// `RPC_READ_INDEX_REQ` to the group leader; the leader confirms its +/// leadership against a quorum and replies with exactly one +/// `RPC_READ_INDEX_RESP` carrying its read index, a leader hint, or a +/// timeout. One-shot request/response — no streaming. +pub const RPC_READ_INDEX_REQ: u8 = 47; +pub const RPC_READ_INDEX_RESP: u8 = 48; +/// Authorization lease renewal: a node reports its authorization coverage to +/// the metadata group leader in `RPC_AUTH_LEASE_RENEW_REQ` and receives one +/// `RPC_AUTH_LEASE_RENEW_RESP` granting or withholding its lease. +pub const RPC_AUTH_LEASE_RENEW_REQ: u8 = 49; +pub const RPC_AUTH_LEASE_RENEW_RESP: u8 = 50; +/// Authorization barrier: a writer holds its acknowledgement until the +/// metadata group leader answers `RPC_AUTH_BARRIER_REQ` with one +/// `RPC_AUTH_BARRIER_RESP`. +pub const RPC_AUTH_BARRIER_REQ: u8 = 51; +pub const RPC_AUTH_BARRIER_RESP: u8 = 52; +/// Answer to an `RPC_VSHARD_ENVELOPE` request whose handler failed. It +/// carries the handler's typed error in place of a response envelope. +pub const RPC_VSHARD_REFUSAL: u8 = 53; + // VShardMessageType discriminants for distributed array ops (u16, range 80-89). // These mirror `crate::wire::VShardMessageType` repr values and are declared // here so external code can reference them without importing the full enum. diff --git a/nodedb-cluster/src/rpc_codec/mod.rs b/nodedb-cluster/src/rpc_codec/mod.rs index a962dae83..b38419f8c 100644 --- a/nodedb-cluster/src/rpc_codec/mod.rs +++ b/nodedb-cluster/src/rpc_codec/mod.rs @@ -9,6 +9,7 @@ //! - All wire types re-exported from their sub-modules. pub mod auth_envelope; +pub mod auth_lease; pub mod calvin_submit; pub mod cluster_mgmt; pub mod data_plane_error; @@ -21,7 +22,9 @@ pub mod metadata; pub mod peer_seq; pub mod raft_msgs; pub mod raft_rpc; +pub mod read_index; pub mod reservation; +pub mod shard_error; pub mod shuffle; pub mod surrogate; pub mod vshard; @@ -29,6 +32,10 @@ pub mod vshard; pub use auth_envelope::{ ENVELOPE_OVERHEAD, ENVELOPE_VERSION, EnvelopeFields, parse_envelope, write_envelope, }; +pub use auth_lease::{ + AuthBarrierOutcome, AuthBarrierRequest, AuthBarrierResponse, AuthLeaseRenewOutcome, + AuthLeaseRenewRequest, AuthLeaseRenewResponse, GroupCoverage, +}; pub use calvin_submit::{ SubmitCalvinInboxRequest, SubmitCalvinInboxResponse, SubmitCalvinTxnRequest, SubmitCalvinTxnResponse, @@ -37,8 +44,10 @@ pub use cluster_mgmt::{ JoinGroupInfo, JoinNodeInfo, JoinRequest, JoinResponse, LEADER_REDIRECT_PREFIX, PingRequest, PongResponse, TopologyAck, TopologyUpdate, }; -pub use data_plane_error::DataPlaneErrorCode; -pub use data_propose::{DataProposeRequest, DataProposeResponse}; +pub use data_plane_error::{DataPlaneCounterFault, DataPlaneErrorCode, DataPlaneSyncHold}; +pub use data_propose::{ + DataProposeRequest, DataProposeResponse, ForwardedProposeRefusal, ProposeTarget, +}; pub use execute::{ DescriptorVersionEntry, ExecuteRequest, ExecuteResponse, ExecuteStreamChunk, ExecuteStreamEnd, PLAN_DECODE_FAILED, TypedClusterError, @@ -48,12 +57,15 @@ pub use mac::{MAC_LEN, MacKey}; pub use metadata::{MetadataProposeRequest, MetadataProposeResponse}; pub use peer_seq::{PeerSeqSender, PeerSeqWindow, REPLAY_WINDOW}; pub use raft_rpc::{RaftRpc, decode, encode, frame_size}; +pub use read_index::{ReadIndexOutcome, ReadIndexRequest, ReadIndexResponse}; pub use reservation::{ ReleaseReservationRequest, ReleaseReservationResponse, ReserveReadRequest, ReserveReadResponse, }; +pub use shard_error::{RaftErrorWire, ShardErrorWire}; pub use shuffle::{ JoinKeyPair, PartNodeEntry, ShuffleAggregateConsumeRequest, ShuffleAggregateConsumeResponse, ShuffleConsumeRequest, ShuffleConsumeResponse, ShuffleProduceRequest, ShuffleProduceResponse, ShufflePushChunk, ShufflePushEnd, ShufflePushRequest, SortKey, }; pub use surrogate::{AssignSurrogateRequest, AssignSurrogateResponse}; +pub use vshard::VShardRefusal; diff --git a/nodedb-cluster/src/rpc_codec/raft_rpc.rs b/nodedb-cluster/src/rpc_codec/raft_rpc.rs index c0b9c7f2a..b01149021 100644 --- a/nodedb-cluster/src/rpc_codec/raft_rpc.rs +++ b/nodedb-cluster/src/rpc_codec/raft_rpc.rs @@ -7,6 +7,9 @@ use nodedb_raft::message::{ PreVoteRequest, PreVoteResponse, RequestVoteRequest, RequestVoteResponse, TimeoutNowRequest, }; +use super::auth_lease::{ + AuthBarrierRequest, AuthBarrierResponse, AuthLeaseRenewRequest, AuthLeaseRenewResponse, +}; use super::calvin_submit::{ SubmitCalvinInboxRequest, SubmitCalvinInboxResponse, SubmitCalvinTxnRequest, SubmitCalvinTxnResponse, @@ -19,6 +22,7 @@ use super::discriminants::*; use super::execute::{ExecuteRequest, ExecuteResponse, ExecuteStreamChunk, ExecuteStreamEnd}; use super::header::HEADER_SIZE; use super::metadata::{MetadataProposeRequest, MetadataProposeResponse}; +use super::read_index::{ReadIndexRequest, ReadIndexResponse}; use super::reservation::{ ReleaseReservationRequest, ReleaseReservationResponse, ReserveReadRequest, ReserveReadResponse, }; @@ -28,9 +32,10 @@ use super::shuffle::{ ShufflePushEnd, ShufflePushRequest, }; use super::surrogate::{AssignSurrogateRequest, AssignSurrogateResponse}; +use super::vshard::VShardRefusal; use super::{ - calvin_submit, cluster_mgmt, data_propose, execute, metadata, raft_msgs, reservation, shuffle, - surrogate, vshard, + auth_lease, calvin_submit, cluster_mgmt, data_propose, execute, metadata, raft_msgs, + read_index, reservation, shuffle, surrogate, vshard, }; use crate::error::{ClusterError, Result}; use crate::wire_version::{unwrap_bytes_versioned, wrap_bytes_versioned}; @@ -143,6 +148,19 @@ pub enum RaftRpc { // Data-group proposal forwarding (groups 1+) DataProposeRequest(DataProposeRequest), DataProposeResponse(DataProposeResponse), + // Routed read index. A node that does not lead a group asks the leader + // for a read index confirmed against a quorum. + ReadIndexRequest(ReadIndexRequest), + ReadIndexResponse(ReadIndexResponse), + // Authorization lease renewal and the writer-side barrier, both answered + // by the metadata group leader. + AuthLeaseRenewRequest(AuthLeaseRenewRequest), + AuthLeaseRenewResponse(AuthLeaseRenewResponse), + AuthBarrierRequest(AuthBarrierRequest), + AuthBarrierResponse(AuthBarrierResponse), + // Answer to a `VShardEnvelope` request whose handler failed. It carries + // the handler's typed error. + VShardRefusal(VShardRefusal), } /// Encode a [`RaftRpc`] into a framed binary message stamped with `epoch`. @@ -212,6 +230,13 @@ pub fn encode(rpc: &RaftRpc, epoch: &crate::cluster_epoch::ClusterEpochState) -> } RaftRpc::DataProposeRequest(m) => data_propose::encode_data_propose_req(m, &mut out), RaftRpc::DataProposeResponse(m) => data_propose::encode_data_propose_resp(m, &mut out), + RaftRpc::ReadIndexRequest(m) => read_index::encode_read_index_req(m, &mut out), + RaftRpc::ReadIndexResponse(m) => read_index::encode_read_index_resp(m, &mut out), + RaftRpc::AuthLeaseRenewRequest(m) => auth_lease::encode_renew_req(m, &mut out), + RaftRpc::AuthLeaseRenewResponse(m) => auth_lease::encode_renew_resp(m, &mut out), + RaftRpc::AuthBarrierRequest(m) => auth_lease::encode_barrier_req(m, &mut out), + RaftRpc::AuthBarrierResponse(m) => auth_lease::encode_barrier_resp(m, &mut out), + RaftRpc::VShardRefusal(m) => vshard::encode_vshard_refusal(m, &mut out), }?; super::header::stamp_epoch(&mut out, epoch)?; Ok(out) @@ -307,6 +332,13 @@ pub fn decode(data: &[u8], epoch: &crate::cluster_epoch::ClusterEpochState) -> R RPC_RELEASE_RESERVATION_RESP => reservation::decode_release_reservation_resp(payload), RPC_DATA_PROPOSE_REQ => data_propose::decode_data_propose_req(payload), RPC_DATA_PROPOSE_RESP => data_propose::decode_data_propose_resp(payload), + RPC_READ_INDEX_REQ => read_index::decode_read_index_req(payload), + RPC_READ_INDEX_RESP => read_index::decode_read_index_resp(payload), + RPC_AUTH_LEASE_RENEW_REQ => auth_lease::decode_renew_req(payload), + RPC_AUTH_LEASE_RENEW_RESP => auth_lease::decode_renew_resp(payload), + RPC_AUTH_BARRIER_REQ => auth_lease::decode_barrier_req(payload), + RPC_AUTH_BARRIER_RESP => auth_lease::decode_barrier_resp(payload), + RPC_VSHARD_REFUSAL => vshard::decode_vshard_refusal(payload), _ => Err(ClusterError::Codec { detail: format!("unknown rpc_type: {rpc_type}"), }), diff --git a/nodedb-cluster/src/rpc_codec/read_index.rs b/nodedb-cluster/src/rpc_codec/read_index.rs new file mode 100644 index 000000000..420fbff29 --- /dev/null +++ b/nodedb-cluster/src/rpc_codec/read_index.rs @@ -0,0 +1,130 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! ReadIndexRequest / ReadIndexResponse wire types and codecs. +//! +//! A node that does not lead a Raft group asks the group leader for a read +//! index. The leader confirms its leadership against a quorum and answers +//! with its commit index at the time of the request. Once the asking node has +//! applied the group through that index, its state includes every entry +//! committed before the request. + +use super::discriminants::*; +use super::header::write_frame; +use super::raft_rpc::RaftRpc; +use crate::error::{ClusterError, Result}; + +/// Ask the leader of `group_id` for a confirmed read index. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct ReadIndexRequest { + pub group_id: u64, + /// How long the leader may wait for a quorum to confirm it. + pub timeout_ms: u64, +} + +/// The leader's answer to a [`ReadIndexRequest`]. +#[derive(Debug, Clone, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum ReadIndexOutcome { + /// A quorum confirmed the leader. Reads may be served at `read_index`. + Confirmed { read_index: u64 }, + /// The receiver does not lead the group. `leader_hint` names the leader + /// it knows of. + NotLeader { leader_hint: Option }, + /// The receiver leads the group, but no quorum answered in time. + Timeout { waited_ms: u64 }, +} + +/// Response to a [`ReadIndexRequest`]. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct ReadIndexResponse { + pub outcome: ReadIndexOutcome, +} + +macro_rules! to_bytes { + ($msg:expr) => { + rkyv::to_bytes::($msg) + .map(|b| b.to_vec()) + .map_err(|e| ClusterError::Codec { + detail: format!("rkyv serialize: {e}"), + }) + }; +} + +macro_rules! from_bytes { + ($payload:expr, $T:ty, $name:expr) => {{ + let mut aligned = rkyv::util::AlignedVec::<16>::with_capacity($payload.len()); + aligned.extend_from_slice($payload); + rkyv::from_bytes::<$T, rkyv::rancor::Error>(&aligned).map_err(|e| ClusterError::Codec { + detail: format!("rkyv deserialize {}: {e}", $name), + }) + }}; +} + +pub(super) fn encode_read_index_req(msg: &ReadIndexRequest, out: &mut Vec) -> Result<()> { + write_frame(RPC_READ_INDEX_REQ, &to_bytes!(msg)?, out) +} +pub(super) fn encode_read_index_resp(msg: &ReadIndexResponse, out: &mut Vec) -> Result<()> { + write_frame(RPC_READ_INDEX_RESP, &to_bytes!(msg)?, out) +} + +pub(super) fn decode_read_index_req(payload: &[u8]) -> Result { + Ok(RaftRpc::ReadIndexRequest(from_bytes!( + payload, + ReadIndexRequest, + "ReadIndexRequest" + )?)) +} +pub(super) fn decode_read_index_resp(payload: &[u8]) -> Result { + Ok(RaftRpc::ReadIndexResponse(from_bytes!( + payload, + ReadIndexResponse, + "ReadIndexResponse" + )?)) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::cluster_epoch::ClusterEpochState; + use crate::rpc_codec::{decode, encode}; + + fn roundtrip(rpc: RaftRpc) -> RaftRpc { + let epoch = ClusterEpochState::default(); + let encoded = encode(&rpc, &epoch).expect("encode"); + decode(&encoded, &epoch).expect("decode") + } + + #[test] + fn a_request_survives_the_wire() { + let rpc = roundtrip(RaftRpc::ReadIndexRequest(ReadIndexRequest { + group_id: 7, + timeout_ms: 750, + })); + match rpc { + RaftRpc::ReadIndexRequest(req) => { + assert_eq!(req.group_id, 7); + assert_eq!(req.timeout_ms, 750); + } + other => panic!("decoded the wrong variant: {other:?}"), + } + } + + #[test] + fn every_outcome_survives_the_wire() { + for outcome in [ + ReadIndexOutcome::Confirmed { read_index: 42 }, + ReadIndexOutcome::NotLeader { + leader_hint: Some(3), + }, + ReadIndexOutcome::NotLeader { leader_hint: None }, + ReadIndexOutcome::Timeout { waited_ms: 750 }, + ] { + let rpc = roundtrip(RaftRpc::ReadIndexResponse(ReadIndexResponse { + outcome: outcome.clone(), + })); + match rpc { + RaftRpc::ReadIndexResponse(resp) => assert_eq!(resp.outcome, outcome), + other => panic!("decoded the wrong variant: {other:?}"), + } + } + } +} diff --git a/nodedb-cluster/src/rpc_codec/shard_error/convert.rs b/nodedb-cluster/src/rpc_codec/shard_error/convert.rs new file mode 100644 index 000000000..eda394f3b --- /dev/null +++ b/nodedb-cluster/src/rpc_codec/shard_error/convert.rs @@ -0,0 +1,265 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Conversion between `ClusterError` and its wire mirror. +//! +//! Both matches are exhaustive with no catch-all, so a new `ClusterError` +//! variant fails to compile here until it has a wire form. + +use super::wire::ShardErrorWire; +use crate::error::ClusterError; + +impl From for ShardErrorWire { + fn from(error: ClusterError) -> Self { + match error { + ClusterError::Raft(error) => Self::Raft { + error: error.into(), + }, + ClusterError::VShardNotMapped { vshard_id } => Self::VShardNotMapped { vshard_id }, + ClusterError::GroupNotFound { group_id } => Self::GroupNotFound { group_id }, + ClusterError::LearnerNotCaughtUp { + group_id, + node_id, + match_index, + commit_index, + } => Self::LearnerNotCaughtUp { + group_id, + node_id, + match_index, + commit_index, + }, + ClusterError::MigrationInProgress { vshard_id } => { + Self::MigrationInProgress { vshard_id } + } + ClusterError::MigrationPauseBudgetExceeded { + estimated_us, + budget_us, + } => Self::MigrationPauseBudgetExceeded { + estimated_us, + budget_us, + }, + ClusterError::NodeUnreachable { node_id } => Self::NodeUnreachable { node_id }, + ClusterError::GhostNotFound { node_id, shard_id } => { + Self::GhostNotFound { node_id, shard_id } + } + ClusterError::Transport { detail } => Self::Transport { detail }, + ClusterError::ShardTimeout { + vshard_id, + elapsed_ms, + } => Self::ShardTimeout { + vshard_id, + elapsed_ms, + }, + ClusterError::StreamTerminal { error, detail } => Self::StreamTerminal { + error: *error, + detail, + }, + ClusterError::Storage { detail } => Self::Storage { detail }, + ClusterError::DataPlane { code } => Self::DataPlane { code }, + ClusterError::Codec { detail } => Self::Codec { detail }, + ClusterError::UnsupportedWireVersion { + got, + supported_min, + supported_max, + } => Self::UnsupportedWireVersion { + got, + supported_min, + supported_max, + }, + ClusterError::CircuitOpen { node_id, failures } => { + Self::CircuitOpen { node_id, failures } + } + ClusterError::JoinGroupDisappeared { group_id } => { + Self::JoinGroupDisappeared { group_id } + } + ClusterError::JoinCommitTimeout { + group_id, + log_index, + } => Self::JoinCommitTimeout { + group_id, + log_index, + }, + ClusterError::ReadIndexNotLeader { group_id } => Self::ReadIndexNotLeader { group_id }, + ClusterError::ReadIndexTimeout { + group_id, + waited_ms, + } => Self::ReadIndexTimeout { + group_id, + waited_ms, + }, + ClusterError::Config { detail } => Self::Config { detail }, + ClusterError::WrongOwner { + vshard_id, + expected_owner_node, + } => Self::WrongOwner { + vshard_id, + expected_owner_node, + }, + ClusterError::SnapshotCrcMismatch { + group_id, + stored, + computed, + } => Self::SnapshotCrcMismatch { + group_id, + stored, + computed, + }, + ClusterError::SnapshotOffsetRegression { + group_id, + expected, + actual, + } => Self::SnapshotOffsetRegression { + group_id, + expected, + actual, + }, + ClusterError::PartialSnapshotCorrupt { group_id, detail } => { + Self::PartialSnapshotCorrupt { group_id, detail } + } + ClusterError::PartialSnapshotCleanupFailed { group_id, detail } => { + Self::PartialSnapshotCleanupFailed { group_id, detail } + } + ClusterError::SnapshotApplyFailed { group_id, detail } => { + Self::SnapshotApplyFailed { group_id, detail } + } + ClusterError::RemoteUntyped { detail } => Self::Untyped { detail }, + ClusterError::ShardExecution { error, detail } => Self::ShardExecution { + error: *error, + detail, + }, + // Coordinator-side error families. Their message crosses. + other @ (ClusterError::MigrationCheckpoint(_) + | ClusterError::MigrationRecovery(_) + | ClusterError::Calvin(_) + | ClusterError::Mirror(_) + | ClusterError::BspBarrier(_) + | ClusterError::VectorGather(_) + | ClusterError::SpatialGather(_) + | ClusterError::Bm25Gather(_) + | ClusterError::TsGather(_)) => Self::Untyped { + detail: other.to_string(), + }, + } + } +} + +impl From for ClusterError { + fn from(wire: ShardErrorWire) -> Self { + match wire { + ShardErrorWire::Raft { error } => Self::Raft(error.into()), + ShardErrorWire::VShardNotMapped { vshard_id } => Self::VShardNotMapped { vshard_id }, + ShardErrorWire::GroupNotFound { group_id } => Self::GroupNotFound { group_id }, + ShardErrorWire::LearnerNotCaughtUp { + group_id, + node_id, + match_index, + commit_index, + } => Self::LearnerNotCaughtUp { + group_id, + node_id, + match_index, + commit_index, + }, + ShardErrorWire::MigrationInProgress { vshard_id } => { + Self::MigrationInProgress { vshard_id } + } + ShardErrorWire::MigrationPauseBudgetExceeded { + estimated_us, + budget_us, + } => Self::MigrationPauseBudgetExceeded { + estimated_us, + budget_us, + }, + ShardErrorWire::NodeUnreachable { node_id } => Self::NodeUnreachable { node_id }, + ShardErrorWire::GhostNotFound { node_id, shard_id } => { + Self::GhostNotFound { node_id, shard_id } + } + ShardErrorWire::Transport { detail } => Self::Transport { detail }, + ShardErrorWire::ShardTimeout { + vshard_id, + elapsed_ms, + } => Self::ShardTimeout { + vshard_id, + elapsed_ms, + }, + ShardErrorWire::StreamTerminal { error, detail } => Self::StreamTerminal { + error: Box::new(error), + detail, + }, + ShardErrorWire::Storage { detail } => Self::Storage { detail }, + ShardErrorWire::DataPlane { code } => Self::DataPlane { code }, + ShardErrorWire::Codec { detail } => Self::Codec { detail }, + ShardErrorWire::UnsupportedWireVersion { + got, + supported_min, + supported_max, + } => Self::UnsupportedWireVersion { + got, + supported_min, + supported_max, + }, + ShardErrorWire::CircuitOpen { node_id, failures } => { + Self::CircuitOpen { node_id, failures } + } + ShardErrorWire::JoinGroupDisappeared { group_id } => { + Self::JoinGroupDisappeared { group_id } + } + ShardErrorWire::JoinCommitTimeout { + group_id, + log_index, + } => Self::JoinCommitTimeout { + group_id, + log_index, + }, + ShardErrorWire::ReadIndexNotLeader { group_id } => { + Self::ReadIndexNotLeader { group_id } + } + ShardErrorWire::ReadIndexTimeout { + group_id, + waited_ms, + } => Self::ReadIndexTimeout { + group_id, + waited_ms, + }, + ShardErrorWire::Config { detail } => Self::Config { detail }, + ShardErrorWire::WrongOwner { + vshard_id, + expected_owner_node, + } => Self::WrongOwner { + vshard_id, + expected_owner_node, + }, + ShardErrorWire::SnapshotCrcMismatch { + group_id, + stored, + computed, + } => Self::SnapshotCrcMismatch { + group_id, + stored, + computed, + }, + ShardErrorWire::SnapshotOffsetRegression { + group_id, + expected, + actual, + } => Self::SnapshotOffsetRegression { + group_id, + expected, + actual, + }, + ShardErrorWire::PartialSnapshotCorrupt { group_id, detail } => { + Self::PartialSnapshotCorrupt { group_id, detail } + } + ShardErrorWire::PartialSnapshotCleanupFailed { group_id, detail } => { + Self::PartialSnapshotCleanupFailed { group_id, detail } + } + ShardErrorWire::SnapshotApplyFailed { group_id, detail } => { + Self::SnapshotApplyFailed { group_id, detail } + } + ShardErrorWire::Untyped { detail } => Self::RemoteUntyped { detail }, + ShardErrorWire::ShardExecution { error, detail } => Self::ShardExecution { + error: Box::new(error), + detail, + }, + } + } +} diff --git a/nodedb-cluster/src/rpc_codec/shard_error/mod.rs b/nodedb-cluster/src/rpc_codec/shard_error/mod.rs new file mode 100644 index 000000000..31a3d7e1f --- /dev/null +++ b/nodedb-cluster/src/rpc_codec/shard_error/mod.rs @@ -0,0 +1,15 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Wire mirror of the typed error a shard-side handler answers with. +//! +//! A `VShardEnvelope` handler that fails answers with a +//! [`VShardRefusal`](super::VShardRefusal) frame carrying this mirror. The +//! caller rebuilds the same `ClusterError`, so its retry and reroute logic +//! sees the shard's own error instead of a closed stream. + +pub mod convert; +pub mod raft; +pub mod wire; + +pub use raft::RaftErrorWire; +pub use wire::ShardErrorWire; diff --git a/nodedb-cluster/src/rpc_codec/shard_error/raft.rs b/nodedb-cluster/src/rpc_codec/shard_error/raft.rs new file mode 100644 index 000000000..f4d4512d6 --- /dev/null +++ b/nodedb-cluster/src/rpc_codec/shard_error/raft.rs @@ -0,0 +1,110 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Wire mirror of `nodedb_raft::RaftError`. Variant order is the wire ABI: +//! append only. + +use nodedb_raft::RaftError; + +/// A `RaftError` carried across a node hop. `NotLeader` keeps its leader +/// hint, so the caller can chase the redirect. +#[derive(Debug, Clone, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum RaftErrorWire { + NotLeader { + leader_hint: Option, + }, + LogCompacted { + requested: u64, + first_available: u64, + }, + CompactionAheadOfApplied { + requested: u64, + last_applied: u64, + }, + ProposalRejected { + reason: String, + }, + InvalidTransferTarget { + target: u64, + }, + LeadershipTransferInProgress, + GroupNotFound { + group_id: u64, + }, + Transport { + detail: String, + }, + Storage { + detail: String, + }, + Serialization { + detail: String, + }, + SnapshotFormat { + detail: String, + }, + Shutdown, +} + +impl From for RaftErrorWire { + fn from(error: RaftError) -> Self { + match error { + RaftError::NotLeader { leader_hint } => Self::NotLeader { leader_hint }, + RaftError::LogCompacted { + requested, + first_available, + } => Self::LogCompacted { + requested, + first_available, + }, + RaftError::CompactionAheadOfApplied { + requested, + last_applied, + } => Self::CompactionAheadOfApplied { + requested, + last_applied, + }, + RaftError::ProposalRejected { reason } => Self::ProposalRejected { reason }, + RaftError::InvalidTransferTarget { target } => Self::InvalidTransferTarget { target }, + RaftError::LeadershipTransferInProgress => Self::LeadershipTransferInProgress, + RaftError::GroupNotFound { group_id } => Self::GroupNotFound { group_id }, + RaftError::Transport { detail } => Self::Transport { detail }, + RaftError::Storage { detail } => Self::Storage { detail }, + RaftError::Serialization { detail } => Self::Serialization { detail }, + RaftError::SnapshotFormat { detail } => Self::SnapshotFormat { detail }, + RaftError::Shutdown => Self::Shutdown, + } + } +} + +impl From for RaftError { + fn from(wire: RaftErrorWire) -> Self { + match wire { + RaftErrorWire::NotLeader { leader_hint } => Self::NotLeader { leader_hint }, + RaftErrorWire::LogCompacted { + requested, + first_available, + } => Self::LogCompacted { + requested, + first_available, + }, + RaftErrorWire::CompactionAheadOfApplied { + requested, + last_applied, + } => Self::CompactionAheadOfApplied { + requested, + last_applied, + }, + RaftErrorWire::ProposalRejected { reason } => Self::ProposalRejected { reason }, + RaftErrorWire::InvalidTransferTarget { target } => { + Self::InvalidTransferTarget { target } + } + RaftErrorWire::LeadershipTransferInProgress => Self::LeadershipTransferInProgress, + RaftErrorWire::GroupNotFound { group_id } => Self::GroupNotFound { group_id }, + RaftErrorWire::Transport { detail } => Self::Transport { detail }, + RaftErrorWire::Storage { detail } => Self::Storage { detail }, + RaftErrorWire::Serialization { detail } => Self::Serialization { detail }, + RaftErrorWire::SnapshotFormat { detail } => Self::SnapshotFormat { detail }, + RaftErrorWire::Shutdown => Self::Shutdown, + } + } +} diff --git a/nodedb-cluster/src/rpc_codec/shard_error/wire.rs b/nodedb-cluster/src/rpc_codec/shard_error/wire.rs new file mode 100644 index 000000000..6c585d80a --- /dev/null +++ b/nodedb-cluster/src/rpc_codec/shard_error/wire.rs @@ -0,0 +1,126 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The shard error wire enum. Variant order is the wire ABI: append only. + +use super::raft::RaftErrorWire; +use crate::rpc_codec::data_plane_error::DataPlaneErrorCode; +use crate::rpc_codec::execute::TypedClusterError; + +/// A `ClusterError` carried across a node hop. +/// +/// One variant per `ClusterError` variant whose fields cross the wire. The +/// coordinator-side error families (gather, barrier, mirror, Calvin, +/// migration) cross as [`Self::Untyped`] with their message. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum ShardErrorWire { + Raft { + error: RaftErrorWire, + }, + VShardNotMapped { + vshard_id: u32, + }, + GroupNotFound { + group_id: u64, + }, + LearnerNotCaughtUp { + group_id: u64, + node_id: u64, + match_index: u64, + commit_index: u64, + }, + MigrationInProgress { + vshard_id: u32, + }, + MigrationPauseBudgetExceeded { + estimated_us: u64, + budget_us: u64, + }, + NodeUnreachable { + node_id: u64, + }, + GhostNotFound { + node_id: String, + shard_id: u32, + }, + Transport { + detail: String, + }, + ShardTimeout { + vshard_id: u32, + elapsed_ms: u64, + }, + StreamTerminal { + error: TypedClusterError, + detail: String, + }, + Storage { + detail: String, + }, + DataPlane { + code: DataPlaneErrorCode, + }, + Codec { + detail: String, + }, + UnsupportedWireVersion { + got: u8, + supported_min: u8, + supported_max: u8, + }, + CircuitOpen { + node_id: u64, + failures: u32, + }, + JoinGroupDisappeared { + group_id: u64, + }, + JoinCommitTimeout { + group_id: u64, + log_index: u64, + }, + ReadIndexNotLeader { + group_id: u64, + }, + ReadIndexTimeout { + group_id: u64, + waited_ms: u64, + }, + Config { + detail: String, + }, + WrongOwner { + vshard_id: u32, + expected_owner_node: Option, + }, + SnapshotCrcMismatch { + group_id: u64, + stored: u32, + computed: u32, + }, + SnapshotOffsetRegression { + group_id: u64, + expected: u64, + actual: u64, + }, + PartialSnapshotCorrupt { + group_id: u64, + detail: String, + }, + PartialSnapshotCleanupFailed { + group_id: u64, + detail: String, + }, + SnapshotApplyFailed { + group_id: u64, + detail: String, + }, + /// An error whose type has no wire mirror. `detail` is its message. + Untyped { + detail: String, + }, + /// A shard's classified local-execution error, in its typed wire form. + ShardExecution { + error: TypedClusterError, + detail: String, + }, +} diff --git a/nodedb-cluster/src/rpc_codec/surrogate.rs b/nodedb-cluster/src/rpc_codec/surrogate.rs index c8c00607e..9087f7451 100644 --- a/nodedb-cluster/src/rpc_codec/surrogate.rs +++ b/nodedb-cluster/src/rpc_codec/surrogate.rs @@ -41,7 +41,8 @@ pub struct AssignSurrogateRequest { pub vshard_id: u32, pub database_id: u64, pub tenant_id: u64, - /// Collection the surrogate is scoped to. + /// Bare catalog name of the collection the surrogate is scoped to. With + /// `database_id` it forms the canonical collection key. pub collection: String, /// Primary-key bytes of the endpoint whose surrogate is being resolved. pub pk: Vec, diff --git a/nodedb-cluster/src/rpc_codec/vshard.rs b/nodedb-cluster/src/rpc_codec/vshard.rs index 1c441b622..45c4bd98b 100644 --- a/nodedb-cluster/src/rpc_codec/vshard.rs +++ b/nodedb-cluster/src/rpc_codec/vshard.rs @@ -6,11 +6,30 @@ //! retention, and archival messages. The inner VShardMessageType determines //! the handler. The envelope bytes are passed through raw (already serialized //! in their own binary format). +//! +//! A handler that fails answers with a [`VShardRefusal`] frame instead of a +//! response envelope. The frame carries the handler's typed error. -use super::discriminants::RPC_VSHARD_ENVELOPE; +use super::discriminants::{RPC_VSHARD_ENVELOPE, RPC_VSHARD_REFUSAL}; use super::header::write_frame; use super::raft_rpc::RaftRpc; -use crate::error::Result; +use super::shard_error::ShardErrorWire; +use crate::error::{ClusterError, Result}; + +/// The typed error that answers a VShardEnvelope request. The caller +/// rebuilds it with `ClusterError::from`. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct VShardRefusal { + pub error: ShardErrorWire, +} + +impl From for VShardRefusal { + fn from(error: ClusterError) -> Self { + Self { + error: error.into(), + } + } +} pub(super) fn encode_vshard_envelope(bytes: &[u8], out: &mut Vec) -> Result<()> { write_frame(RPC_VSHARD_ENVELOPE, bytes, out) @@ -20,3 +39,132 @@ pub(super) fn decode_vshard_envelope(payload: &[u8]) -> Result { // VShardEnvelope is already in its own binary format — pass through raw. Ok(RaftRpc::VShardEnvelope(payload.to_vec())) } + +pub(super) fn encode_vshard_refusal(msg: &VShardRefusal, out: &mut Vec) -> Result<()> { + let bytes = rkyv::to_bytes::(msg).map_err(|e| ClusterError::Codec { + detail: format!("rkyv serialize VShardRefusal: {e}"), + })?; + write_frame(RPC_VSHARD_REFUSAL, &bytes, out) +} + +pub(super) fn decode_vshard_refusal(payload: &[u8]) -> Result { + let mut aligned = rkyv::util::AlignedVec::<16>::with_capacity(payload.len()); + aligned.extend_from_slice(payload); + let refusal = + rkyv::from_bytes::(&aligned).map_err(|e| { + ClusterError::Codec { + detail: format!("rkyv deserialize VShardRefusal: {e}"), + } + })?; + Ok(RaftRpc::VShardRefusal(refusal)) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::cluster_epoch::ClusterEpochState; + use crate::rpc_codec::DataPlaneErrorCode; + use crate::rpc_codec::{decode, encode}; + + /// Encode a shard error as a refusal frame, decode it, and rebuild it. + fn round_trip(error: ClusterError) -> ClusterError { + let epoch = ClusterEpochState::default(); + let rpc = RaftRpc::VShardRefusal(VShardRefusal::from(error)); + let encoded = encode(&rpc, &epoch).expect("encode"); + match decode(&encoded, &epoch).expect("decode") { + RaftRpc::VShardRefusal(refusal) => ClusterError::from(refusal.error), + other => panic!("decoded the wrong variant: {other:?}"), + } + } + + #[test] + fn a_data_plane_refusal_survives_the_wire() { + let code = DataPlaneErrorCode::Unsupported { + detail: "not on this engine".into(), + }; + match round_trip(ClusterError::DataPlane { code: code.clone() }) { + ClusterError::DataPlane { code: rebuilt } => assert_eq!(rebuilt, code), + other => panic!("expected the typed refusal, got {other:?}"), + } + } + + /// `WrongOwner` crosses typed, so the coordinator's reroute retry sees it. + #[test] + fn wrong_owner_survives_the_wire() { + let error = ClusterError::WrongOwner { + vshard_id: 7, + expected_owner_node: Some(3), + }; + assert!(matches!( + round_trip(error), + ClusterError::WrongOwner { + vshard_id: 7, + expected_owner_node: Some(3) + } + )); + } + + #[test] + fn a_raft_redirect_keeps_its_leader_hint() { + let error = ClusterError::Raft(nodedb_raft::RaftError::NotLeader { + leader_hint: Some(5), + }); + assert!(matches!( + round_trip(error), + ClusterError::Raft(nodedb_raft::RaftError::NotLeader { + leader_hint: Some(5) + }) + )); + } + + #[test] + fn a_codec_error_survives_the_wire() { + let error = ClusterError::Codec { + detail: "bad request body".into(), + }; + match round_trip(error) { + ClusterError::Codec { detail } => assert_eq!(detail, "bad request body"), + other => panic!("expected the codec error, got {other:?}"), + } + } + + /// An error with no wire mirror keeps its message. + #[test] + fn an_untyped_error_keeps_its_message() { + let error = + ClusterError::BspBarrier(crate::distributed_graph::BspBarrierError::Incomplete { + algorithm: "pagerank".into(), + iteration: 3, + acked: 1, + expected: 2, + }); + let message = error.to_string(); + match round_trip(error) { + ClusterError::RemoteUntyped { detail } => assert_eq!(detail, message), + other => panic!("expected the untyped error, got {other:?}"), + } + } + + /// A shard's typed execution error crosses the wire with its typed form. + #[test] + fn a_shard_execution_error_survives_the_wire() { + let typed = crate::rpc_codec::TypedClusterError::Internal { + code: 2000, + message: "permission denied on orders".into(), + }; + let error = ClusterError::ShardExecution { + error: Box::new(typed), + detail: "array put: permission denied on orders".into(), + }; + match round_trip(error) { + ClusterError::ShardExecution { error, detail } => { + assert!(matches!( + *error, + crate::rpc_codec::TypedClusterError::Internal { code: 2000, .. } + )); + assert_eq!(detail, "array put: permission denied on orders"); + } + other => panic!("expected the shard execution error, got {other:?}"), + } + } +} diff --git a/nodedb-cluster/src/shard_split.rs b/nodedb-cluster/src/shard_split.rs index 7500f4508..b20c391d1 100644 --- a/nodedb-cluster/src/shard_split.rs +++ b/nodedb-cluster/src/shard_split.rs @@ -213,16 +213,16 @@ pub fn plan_graph_split( pub fn speculative_prefetch_shards( query_vshards: &[u32], _routing: &RoutingTable, - tenant_collections: &[(nodedb_types::id::DatabaseId, u32, String)], + tenant_collections: &[(u32, nodedb_types::id::CollectionKey<'_>)], ) -> Vec { let mut prefetch: HashSet = HashSet::new(); let queried: HashSet = query_vshards.iter().copied().collect(); - // For each (database, tenant, collection), find all vShards that might hold - // data for the same collection (co-located shards). - for (database_id, _tenant_id, collection) in tenant_collections { - // Hash the (database, collection) pair to find its primary vShard. - let primary = crate::routing::vshard_for_collection(*database_id, collection); + // For each (tenant, collection), find all vShards that might hold data + // for the same collection (co-located shards). + for (_tenant_id, key) in tenant_collections { + // The collection key names its primary vShard. + let primary = crate::routing::vshard_for_collection(*key); // Adjacent vShards (±1, ±2) are likely to hold related data // due to hash distribution locality. @@ -294,7 +294,13 @@ mod tests { let prefetch = speculative_prefetch_shards( &[0, 1], &routing, - &[(nodedb_types::id::DatabaseId::DEFAULT, 1, "users".into())], + &[( + 1, + nodedb_types::id::CollectionKey::from_bare( + nodedb_types::id::DatabaseId::DEFAULT, + "users", + ), + )], ); assert!(prefetch.len() <= 8); // Max prefetch limit. } diff --git a/nodedb-cluster/src/transport/client/close.rs b/nodedb-cluster/src/transport/client/close.rs new file mode 100644 index 000000000..3b40b1db9 --- /dev/null +++ b/nodedb-cluster/src/transport/client/close.rs @@ -0,0 +1,40 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Close this node's QUIC endpoint. +//! +//! Dropping every handle of an endpoint does not release its UDP socket. The +//! endpoint's driver task owns the socket, and it runs until every connection +//! is gone. A peer that keeps its connection alive keeps the socket bound +//! until the idle timeout, so a node that stops and starts again on the same +//! address fails to bind. Closing the endpoint ends every connection at once: +//! peers see the close immediately, and the driver releases the socket once +//! the last handle drops. + +use std::time::Duration; + +use super::transport::NexarTransport; + +/// The QUIC application close code a node sends when it shuts down. +const SHUTDOWN_CLOSE_CODE: u32 = 0; + +impl NexarTransport { + /// Close every connection, refuse new ones, and wait until the peers + /// acknowledged the close or `timeout` passed. Returns `false` when + /// `timeout` passed first; the connections are closed either way. + pub async fn close(&self, timeout: Duration) -> bool { + let endpoint = self.listener.endpoint(); + endpoint.close( + quinn::VarInt::from_u32(SHUTDOWN_CLOSE_CODE), + b"node shutdown", + ); + // The cached connections are closed: a send after this fails at once + // instead of reusing one. + self.peers + .write() + .unwrap_or_else(|p| p.into_inner()) + .clear(); + tokio::time::timeout(timeout, endpoint.wait_idle()) + .await + .is_ok() + } +} diff --git a/nodedb-cluster/src/transport/client/mod.rs b/nodedb-cluster/src/transport/client/mod.rs index ea7827884..2b2a55b5c 100644 --- a/nodedb-cluster/src/transport/client/mod.rs +++ b/nodedb-cluster/src/transport/client/mod.rs @@ -9,11 +9,14 @@ //! //! [`RaftTransport`]: nodedb_raft::transport::RaftTransport +pub mod close; pub mod pool; pub mod raft_impl; pub mod send; pub mod serve; +pub mod sever; +pub mod shuffle_push; pub mod transport; -pub use send::ShufflePushStream; +pub use shuffle_push::ShufflePushStream; pub use transport::{NexarTransport, TransportPeerSnapshot}; diff --git a/nodedb-cluster/src/transport/client/send.rs b/nodedb-cluster/src/transport/client/send.rs index 4790e34de..f662269a8 100644 --- a/nodedb-cluster/src/transport/client/send.rs +++ b/nodedb-cluster/src/transport/client/send.rs @@ -16,15 +16,9 @@ use futures::Stream; use rustls::pki_types::CertificateDer; use tracing::debug; -use std::sync::Arc; - use crate::circuit_breaker::RetryPolicy; use crate::error::{ClusterError, Result}; -use crate::rpc_codec::{ - self, RaftRpc, ShufflePushChunk, ShufflePushEnd, ShufflePushRequest, TypedClusterError, - auth_envelope, -}; -use crate::transport::auth_context::AuthContext; +use crate::rpc_codec::{self, RaftRpc, TypedClusterError, auth_envelope}; use crate::transport::config::SNI_HOSTNAME; use crate::transport::peer_identity_verifier::{ VerifyOutcome, spki_pin_from_cert_der, verify_peer_identity, @@ -165,6 +159,7 @@ impl NexarTransport { rpc: RaftRpc, read_timeout: Duration, ) -> Result { + self.check_not_severed(target)?; // Encode the inner RPC once (codec errors are not retryable). // Each retry wraps it in a fresh envelope so the seq advances // per attempt — a retry is a new frame, not a replayed frame. @@ -320,49 +315,6 @@ impl NexarTransport { }) } - /// Open a cross-node streaming-shuffle push to `target` (E1). - /// - /// Producer → receiver direction (the mirror of [`send_rpc_stream`], which - /// streams a response back): opens a bidi stream on the pooled connection, - /// writes the [`ShufflePushRequest`] envelope, then one [`ShufflePushChunk`] - /// envelope per pre-batched payload, then exactly one [`ShufflePushEnd`] - /// (clean EOF), and `finish()`es the send half. It does **not** read a - /// reply — the server deposits the chunks and never writes back, so the - /// helper fire-and-finishes. - /// - /// Each `batches` element is a standalone msgpack array of rows (the same - /// convention as `RowBatch.payload`). E4 — the planner-side caller — is - /// responsible for computing `partition_hash(row, keys) % num_parts` and - /// grouping rows into per-partition batches before calling this; E1's - /// helper takes already-partitioned payloads. - pub async fn send_shuffle_push( - &self, - target: u64, - req: ShufflePushRequest, - batches: Vec>, - ) -> Result<()> { - // Reimplemented on top of the incremental [`ShufflePushStream`] handle so - // the one-shot path and the E4a fanout sink share ONE wire encoder: open - // → push each pre-batched payload → finish with a clean EOF. - let mut stream = ShufflePushStream::open(self, target, req).await?; - for payload in batches { - stream.push_chunk(payload).await?; - } - stream.finish(None).await - } - - /// Open an incremental shuffle-push stream to `target`. - /// - /// Thin convenience wrapper around [`ShufflePushStream::open`] for callers - /// that hold `&self` (the E4a fanout sink opens one per part). - pub async fn open_shuffle_push_stream( - &self, - target: u64, - req: ShufflePushRequest, - ) -> Result { - ShufflePushStream::open(self, target, req).await - } - /// Single-attempt RPC send (no retry, no circuit breaker). async fn try_send_once( &self, @@ -433,7 +385,11 @@ impl NexarTransport { /// guard the second accept trips on its own first — the envelope /// was never replayed, the same window simply saw traffic from both /// directions for `peer_id == local_node_id`. - fn verify_connection_target(&self, conn: &quinn::Connection, target: u64) -> Result<()> { + pub(super) fn verify_connection_target( + &self, + conn: &quinn::Connection, + target: u64, + ) -> Result<()> { if !self.identity_store.enforces_peer_identity() || self.identity_store.bootstrap_window_open() { @@ -487,97 +443,6 @@ impl NexarTransport { } } -/// An incremental producer → receiver shuffle-push stream (E4a). -/// -/// The one-shot [`NexarTransport::send_shuffle_push`] writes every chunk up -/// front; the E4a fanout sink instead opens one of these per target part and -/// pushes chunks as the local scan produces rows, finishing the stream (clean -/// EOF or terminal error) only once the scan ends. It owns its `quinn::SendStream` -/// and an `Arc` clone of the transport's [`AuthContext`] so each frame is wrapped -/// with a fresh outbound `seq` exactly like the one-shot path — no borrow of the -/// transport is held for the stream's lifetime. -/// -/// Each `write_all` is awaited inline, so QUIC flow control back-pressures the -/// producer when the receiver falls behind — bounded memory, never a buffered -/// whole side. -pub struct ShufflePushStream { - send: quinn::SendStream, - auth: Arc, - target: u64, -} - -impl ShufflePushStream { - /// Open a bidi stream to `target` and write the opening - /// [`ShufflePushRequest`] envelope. The send half stays open for subsequent - /// [`push_chunk`](Self::push_chunk) calls; the receiver deposits frames and - /// never writes back. - pub async fn open( - transport: &NexarTransport, - target: u64, - req: ShufflePushRequest, - ) -> Result { - let conn = transport.get_or_connect(target).await?; - transport.verify_connection_target(&conn, target)?; - let (mut send, _recv) = conn.open_bi().await.map_err(|e| ClusterError::Transport { - detail: format!("open_bi (shuffle push) to node {target}: {e}"), - })?; - - let auth = Arc::clone(transport.auth()); - let req_envelope = wrap_with_auth(&auth, &RaftRpc::ShufflePushRequest(req))?; - send.write_all(&req_envelope) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("write shuffle push request to node {target}: {e}"), - })?; - - Ok(Self { send, auth, target }) - } - - /// Write one [`ShufflePushChunk`] envelope (a standalone msgpack row array). - pub async fn push_chunk(&mut self, payload: Vec) -> Result<()> { - let chunk_envelope = wrap_with_auth( - &self.auth, - &RaftRpc::ShufflePushChunk(ShufflePushChunk { payload }), - )?; - self.send - .write_all(&chunk_envelope) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("write shuffle push chunk to node {}: {e}", self.target), - }) - } - - /// Write the terminal [`ShufflePushEnd`] envelope (`error: None` for a clean - /// EOF, `Some(e)` to fail the receiver fast) and finish the send half. - pub async fn finish(mut self, error: Option) -> Result<()> { - let end_envelope = wrap_with_auth( - &self.auth, - &RaftRpc::ShufflePushEnd(ShufflePushEnd { error }), - )?; - self.send - .write_all(&end_envelope) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("write shuffle push end to node {}: {e}", self.target), - })?; - self.send.finish().map_err(|e| ClusterError::Transport { - detail: format!("finish shuffle push to node {}: {e}", self.target), - })?; - Ok(()) - } -} - -/// Encode `rpc` and wrap it in an authenticated envelope with a fresh outbound -/// `seq` — the standalone form of [`NexarTransport::wrap_outbound`] for the -/// owned-[`AuthContext`] [`ShufflePushStream`]. -fn wrap_with_auth(auth: &AuthContext, rpc: &RaftRpc) -> Result> { - let inner = rpc_codec::encode(rpc, &auth.epoch)?; - let seq = auth.peer_seq_out.next(); - let mut out = Vec::with_capacity(auth_envelope::ENVELOPE_OVERHEAD + inner.len()); - auth_envelope::write_envelope(auth.local_node_id, seq, &inner, &auth.mac_key, &mut out)?; - Ok(out) -} - /// Map a terminal [`TypedClusterError`] carried by `ExecuteStreamEnd` into a /// [`ClusterError`] for the stream's `Err` item. /// @@ -586,5 +451,8 @@ fn wrap_with_auth(auth: &AuthContext, rpc: &RaftRpc) -> Result> { /// surfaced as-is. The original typed shape is preserved in the detail string. fn stream_terminal_error(err: TypedClusterError) -> ClusterError { let detail = format!("{err:?}"); - ClusterError::StreamTerminal { error: err, detail } + ClusterError::StreamTerminal { + error: Box::new(err), + detail, + } } diff --git a/nodedb-cluster/src/transport/client/sever.rs b/nodedb-cluster/src/transport/client/sever.rs new file mode 100644 index 000000000..b2bed0598 --- /dev/null +++ b/nodedb-cluster/src/transport/client/sever.rs @@ -0,0 +1,46 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Sever this node from chosen peers. +//! +//! A severed peer receives no RPC from this node: every send to it fails at +//! once with a transport error, as if the network dropped it. Severing each +//! side from the other models a network partition between them, which is +//! how tests show that a partitioned node loses its leases and catches up +//! once the partition heals. + +use crate::error::{ClusterError, Result}; + +use super::transport::NexarTransport; + +impl NexarTransport { + /// Stop sending to `peer` until [`Self::heal`] is called. + pub fn sever(&self, peer: u64) { + self.severed + .write() + .unwrap_or_else(|p| p.into_inner()) + .insert(peer); + } + + /// Resume sending to `peer`. + pub fn heal(&self, peer: u64) { + self.severed + .write() + .unwrap_or_else(|p| p.into_inner()) + .remove(&peer); + } + + /// Refuse a send to a severed peer. + pub(super) fn check_not_severed(&self, target: u64) -> Result<()> { + if self + .severed + .read() + .unwrap_or_else(|p| p.into_inner()) + .contains(&target) + { + return Err(ClusterError::Transport { + detail: format!("node {} is severed from node {target}", self.node_id), + }); + } + Ok(()) + } +} diff --git a/nodedb-cluster/src/transport/client/shuffle_push.rs b/nodedb-cluster/src/transport/client/shuffle_push.rs new file mode 100644 index 000000000..413f57baa --- /dev/null +++ b/nodedb-cluster/src/transport/client/shuffle_push.rs @@ -0,0 +1,152 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Outbound streaming-shuffle push: a producer streams row batches to a +//! receiver over one QUIC bidi stream, each frame in an authenticated +//! envelope. + +use std::sync::Arc; + +use crate::error::{ClusterError, Result}; +use crate::rpc_codec::{ + self, RaftRpc, ShufflePushChunk, ShufflePushEnd, ShufflePushRequest, TypedClusterError, + auth_envelope, +}; +use crate::transport::auth_context::AuthContext; + +use super::transport::NexarTransport; + +impl NexarTransport { + /// Open a cross-node streaming-shuffle push to `target`. + /// + /// Producer → receiver direction (the mirror of [`NexarTransport::send_rpc_stream`], which + /// streams a response back): opens a bidi stream on the pooled connection, + /// writes the [`ShufflePushRequest`] envelope, then one [`ShufflePushChunk`] + /// envelope per pre-batched payload, then exactly one [`ShufflePushEnd`] + /// (clean EOF), and `finish()`es the send half. It does **not** read a + /// reply — the server deposits the chunks and never writes back, so the + /// helper fire-and-finishes. + /// + /// Each `batches` element is a standalone msgpack array of rows (the same + /// convention as `RowBatch.payload`). The planner-side caller is + /// responsible for computing `partition_hash(row, keys) % num_parts` and + /// grouping rows into per-partition batches before calling this. This + /// helper takes already-partitioned payloads. + pub async fn send_shuffle_push( + &self, + target: u64, + req: ShufflePushRequest, + batches: Vec>, + ) -> Result<()> { + // Reimplemented on top of the incremental [`ShufflePushStream`] handle so + // the one-shot path and the fanout sink share one wire encoder: open + // → push each pre-batched payload → finish with a clean EOF. + let mut stream = ShufflePushStream::open(self, target, req).await?; + for payload in batches { + stream.push_chunk(payload).await?; + } + stream.finish(None).await + } + + /// Open an incremental shuffle-push stream to `target`. + /// + /// Thin convenience wrapper around [`ShufflePushStream::open`] for callers + /// that hold `&self` (the fanout sink opens one per part). + pub async fn open_shuffle_push_stream( + &self, + target: u64, + req: ShufflePushRequest, + ) -> Result { + ShufflePushStream::open(self, target, req).await + } +} + +/// An incremental producer → receiver shuffle-push stream. +/// +/// The one-shot [`NexarTransport::send_shuffle_push`] writes every chunk up +/// front. The fanout sink instead opens one of these per target part and +/// pushes chunks as the local scan produces rows, finishing the stream (clean +/// EOF or terminal error) only once the scan ends. It owns its `quinn::SendStream` +/// and an `Arc` clone of the transport's [`AuthContext`] so each frame is wrapped +/// with a fresh outbound `seq` exactly like the one-shot path — no borrow of the +/// transport is held for the stream's lifetime. +/// +/// Each `write_all` is awaited inline, so QUIC flow control back-pressures the +/// producer when the receiver falls behind — bounded memory, never a buffered +/// whole side. +pub struct ShufflePushStream { + send: quinn::SendStream, + auth: Arc, + target: u64, +} + +impl ShufflePushStream { + /// Open a bidi stream to `target` and write the opening + /// [`ShufflePushRequest`] envelope. The send half stays open for subsequent + /// [`push_chunk`](Self::push_chunk) calls; the receiver deposits frames and + /// never writes back. + pub async fn open( + transport: &NexarTransport, + target: u64, + req: ShufflePushRequest, + ) -> Result { + let conn = transport.get_or_connect(target).await?; + transport.verify_connection_target(&conn, target)?; + let (mut send, _recv) = conn.open_bi().await.map_err(|e| ClusterError::Transport { + detail: format!("open_bi (shuffle push) to node {target}: {e}"), + })?; + + let auth = Arc::clone(transport.auth()); + let req_envelope = wrap_with_auth(&auth, &RaftRpc::ShufflePushRequest(req))?; + send.write_all(&req_envelope) + .await + .map_err(|e| ClusterError::Transport { + detail: format!("write shuffle push request to node {target}: {e}"), + })?; + + Ok(Self { send, auth, target }) + } + + /// Write one [`ShufflePushChunk`] envelope (a standalone msgpack row array). + pub async fn push_chunk(&mut self, payload: Vec) -> Result<()> { + let chunk_envelope = wrap_with_auth( + &self.auth, + &RaftRpc::ShufflePushChunk(ShufflePushChunk { payload }), + )?; + self.send + .write_all(&chunk_envelope) + .await + .map_err(|e| ClusterError::Transport { + detail: format!("write shuffle push chunk to node {}: {e}", self.target), + }) + } + + /// Write the terminal [`ShufflePushEnd`] envelope (`error: None` for a clean + /// EOF, `Some(e)` to fail the receiver fast) and finish the send half. + pub async fn finish(mut self, error: Option) -> Result<()> { + let end_envelope = wrap_with_auth( + &self.auth, + &RaftRpc::ShufflePushEnd(ShufflePushEnd { error }), + )?; + self.send + .write_all(&end_envelope) + .await + .map_err(|e| ClusterError::Transport { + detail: format!("write shuffle push end to node {}: {e}", self.target), + })?; + self.send.finish().map_err(|e| ClusterError::Transport { + detail: format!("finish shuffle push to node {}: {e}", self.target), + })?; + Ok(()) + } +} + +/// Encode `rpc` and wrap it in an authenticated envelope with a fresh outbound +/// `seq` — the standalone form of [`NexarTransport::wrap_outbound`] for the +/// owned-[`AuthContext`] [`ShufflePushStream`]. +fn wrap_with_auth(auth: &AuthContext, rpc: &RaftRpc) -> Result> { + let inner = rpc_codec::encode(rpc, &auth.epoch)?; + let seq = auth.peer_seq_out.next(); + let mut out = Vec::with_capacity(auth_envelope::ENVELOPE_OVERHEAD + inner.len()); + auth_envelope::write_envelope(auth.local_node_id, seq, &inner, &auth.mac_key, &mut out)?; + Ok(out) +} diff --git a/nodedb-cluster/src/transport/client/transport.rs b/nodedb-cluster/src/transport/client/transport.rs index b2a11e876..3b5829b86 100644 --- a/nodedb-cluster/src/transport/client/transport.rs +++ b/nodedb-cluster/src/transport/client/transport.rs @@ -60,6 +60,8 @@ pub struct NexarTransport { /// inside the cluster transport and never crosses the SPSC bridge /// into the Data Plane. pub(super) agreed_versions: RwLock>, + /// Peers this node refuses to send to. See [`super::sever`]. + pub(super) severed: RwLock>, } fn default_identity_store(creds: &TransportCredentials) -> Arc { @@ -226,6 +228,7 @@ impl NexarTransport { local_spki_pin, bootstrap_peer_spki, agreed_versions: RwLock::new(HashMap::new()), + severed: RwLock::new(std::collections::HashSet::new()), }) } diff --git a/nodedb-columnar/src/mutation/engine.rs b/nodedb-columnar/src/mutation/engine.rs index 27987cccb..29534c539 100644 --- a/nodedb-columnar/src/mutation/engine.rs +++ b/nodedb-columnar/src/mutation/engine.rs @@ -442,6 +442,7 @@ mod tests { Value::String("Alice Updated".into()), Value::Float(0.75), ], + None, ) .expect("update"); @@ -616,6 +617,63 @@ mod tests { assert_eq!(engine.memtable_surrogates(), &[None]); } + #[test] + fn an_update_keeps_the_old_rows_surrogate() { + let mut engine = MutationEngine::new("col".into(), test_schema()); + engine + .insert_with_surrogate( + &[Value::Integer(1), Value::String("x".into()), Value::Null], + Surrogate(42), + ) + .expect("insert"); + engine + .update( + &Value::Integer(1), + &[Value::Integer(2), Value::String("y".into()), Value::Null], + Some(Surrogate(7)), + ) + .expect("update"); + let live: Vec> = engine + .scan_memtable_rows_with_surrogates() + .map(|(surrogate, _)| surrogate) + .collect(); + assert_eq!( + live, + vec![Some(Surrogate(42))], + "the replacement row carries the identity the memtable row had" + ); + } + + #[test] + fn an_update_of_a_flushed_row_carries_the_segment_surrogate() { + let mut engine = MutationEngine::new("col".into(), test_schema()); + engine + .insert_with_surrogate( + &[Value::Integer(1), Value::String("x".into()), Value::Null], + Surrogate(42), + ) + .expect("insert"); + let segment_id = engine.next_segment_id(); + let _drained = engine.memtable_mut().drain_optimized(); + engine.on_memtable_flushed(segment_id).expect("flush"); + engine + .update( + &Value::Integer(1), + &[Value::Integer(1), Value::String("y".into()), Value::Null], + Some(Surrogate(42)), + ) + .expect("update"); + let live: Vec> = engine + .scan_memtable_rows_with_surrogates() + .map(|(surrogate, _)| surrogate) + .collect(); + assert_eq!( + live, + vec![Some(Surrogate(42))], + "the replacement row carries the identity the flushed row had" + ); + } + #[test] fn flush_clears_surrogate_table() { let mut engine = MutationEngine::new("col".into(), test_schema()); diff --git a/nodedb-columnar/src/mutation/write.rs b/nodedb-columnar/src/mutation/write.rs index 827342c3f..b77341bf4 100644 --- a/nodedb-columnar/src/mutation/write.rs +++ b/nodedb-columnar/src/mutation/write.rs @@ -203,16 +203,36 @@ impl MutationEngine { /// /// NOTE: The caller must provide the full old row values for the re-insert. /// This method takes the complete new row (already merged with old values). + /// + /// The new row keeps the cross-engine surrogate the old row carried. A + /// memtable row's surrogate is in this engine's side table. A flushed + /// row's surrogate is in its segment's sidecar, outside this engine, so + /// the caller passes it as `flushed_surrogate`. It is read only when the + /// old row is flushed. pub fn update( &mut self, old_pk: &Value, new_values: &[Value], + flushed_surrogate: Option, ) -> Result { + let surrogate = match self.pk_index.get(&encode_pk(old_pk)) { + Some(loc) if loc.segment_id == self.memtable_segment_id => self + .memtable_surrogates + .get(loc.row_index as usize) + .copied() + .flatten(), + Some(_) => flushed_surrogate, + None => None, + }; + // Delete the old row. let delete_result = self.delete(old_pk)?; - // Insert the new row. - let insert_result = self.insert(new_values)?; + // Insert the new row under the old row's surrogate. + let insert_result = match surrogate { + Some(surrogate) => self.insert_with_surrogate(new_values, surrogate)?, + None => self.insert(new_values)?, + }; // Combine WAL records. let mut wal_records = delete_result.wal_records; diff --git a/nodedb-crdt/src/dead_letter.rs b/nodedb-crdt/src/dead_letter.rs index 7d29f55ab..ce188d613 100644 --- a/nodedb-crdt/src/dead_letter.rs +++ b/nodedb-crdt/src/dead_letter.rs @@ -25,7 +25,16 @@ use crate::constraint::Constraint; use crate::error::{CrdtError, Result}; /// Suggested action the application should take to resolve a constraint violation. -#[derive(Debug, Clone, Serialize, Deserialize)] +#[derive( + Debug, + Clone, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] pub enum CompensationHint { /// Retry with a different value for the conflicting field. /// Example: UNIQUE violation — suggest appending a suffix. @@ -58,7 +67,16 @@ pub enum CompensationHint { } /// A rejected delta with metadata for debugging and recovery. -#[derive(Debug, Clone, Serialize, Deserialize)] +#[derive( + Debug, + Clone, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] pub struct DeadLetter { /// Unique ID for this dead letter entry. pub id: u64, @@ -103,6 +121,12 @@ pub struct DeadLetter { /// Number of times this delta has been retried. pub retry_count: u32, + + /// The log position of the record whose apply rejected this delta, once + /// the applier binds it. One record yields at most one entry, so a + /// re-applied record never adds a second. + #[serde(default)] + pub source_lsn: Option, } /// Parameters for [`DeadLetterQueue::enqueue`]. @@ -175,11 +199,60 @@ impl DeadLetterQueue { hint, rejected_at: now, retry_count: 0, + source_lsn: None, }); Ok(id) } + /// Bind entry `id` to the record at `source_lsn` that produced it. + /// + /// Returns the bound entry. Returns `None` when another entry already + /// names that record: entry `id` is a repeat of it and is removed. Also + /// `None` when no entry `id` exists. + pub fn bind_source(&mut self, id: u64, source_lsn: u64) -> Option<&DeadLetter> { + let repeat = self + .entries + .iter() + .any(|dl| dl.id != id && dl.source_lsn == Some(source_lsn)); + if repeat { + self.remove(id); + return None; + } + let entry = self.entries.iter_mut().find(|dl| dl.id == id)?; + entry.source_lsn = Some(source_lsn); + Some(&*entry) + } + + /// Put back an entry read from durable storage. + /// + /// An entry whose source record an existing entry already names is + /// skipped. Fresh ids stay above every restored id. + pub fn restore(&mut self, entry: DeadLetter) -> Result<()> { + let known = entry.source_lsn.is_some() + && self + .entries + .iter() + .any(|dl| dl.source_lsn == entry.source_lsn); + if known { + return Ok(()); + } + if self.entries.len() >= self.capacity { + return Err(CrdtError::DlqFull { + capacity: self.capacity, + pending: self.entries.len(), + }); + } + self.next_id = self.next_id.max(entry.id.saturating_add(1)); + self.entries.push_back(entry); + Ok(()) + } + + /// Every pending entry, oldest first. + pub fn iter(&self) -> impl Iterator { + self.entries.iter() + } + /// Peek at the oldest dead letter without removing it. pub fn peek(&self) -> Option<&DeadLetter> { self.entries.front() @@ -490,4 +563,55 @@ mod tests { assert_eq!(removed.reason, "a"); assert_eq!(dlq.len(), 1); } + + fn enqueue_one(dlq: &mut DeadLetterQueue, peer_id: u64) -> u64 { + dlq.enqueue(EnqueueDeadLetterArgs { + peer_id, + user_id: 0, + tenant_id: 0, + delta: b"delta".to_vec(), + constraint: &test_constraint(), + reason: "duplicate".into(), + hint: CompensationHint::ManualIntervention { + reason: "duplicate".into(), + }, + }) + .expect("enqueue") + } + + #[test] + fn a_record_bound_twice_keeps_one_entry() { + let mut dlq = DeadLetterQueue::new(10); + let first = enqueue_one(&mut dlq, 1); + assert!(dlq.bind_source(first, 7).is_some()); + let repeat = enqueue_one(&mut dlq, 1); + assert!(dlq.bind_source(repeat, 7).is_none()); + assert_eq!(dlq.len(), 1); + assert_eq!(dlq.peek().map(|dl| dl.source_lsn), Some(Some(7))); + } + + #[test] + fn a_restored_entry_is_kept_once_and_fresh_ids_stay_above_it() { + let mut source = DeadLetterQueue::new(10); + let id = enqueue_one(&mut source, 1); + let stored = source.bind_source(id, 9).cloned().expect("bound"); + + let mut restored = DeadLetterQueue::new(10); + restored.restore(stored.clone()).expect("restore"); + restored.restore(stored.clone()).expect("restore again"); + assert_eq!(restored.len(), 1); + assert_eq!(restored.peek(), Some(&stored)); + let fresh = enqueue_one(&mut restored, 2); + assert!(fresh > stored.id); + } + + #[test] + fn a_dead_letter_round_trips_through_messagepack() { + let mut dlq = DeadLetterQueue::new(10); + let id = enqueue_one(&mut dlq, 3); + let entry = dlq.bind_source(id, 11).cloned().expect("bound"); + let bytes = zerompk::to_msgpack_vec(&entry).expect("encode"); + let decoded: DeadLetter = zerompk::from_msgpack(&bytes).expect("decode"); + assert_eq!(decoded, entry); + } } diff --git a/nodedb-crdt/src/lib.rs b/nodedb-crdt/src/lib.rs index 643fd107f..d38961cb7 100644 --- a/nodedb-crdt/src/lib.rs +++ b/nodedb-crdt/src/lib.rs @@ -38,7 +38,7 @@ pub mod validator; pub use auth::CrdtAuthContext; pub use constraint::{Constraint, ConstraintKind, ConstraintSet}; -pub use dead_letter::{CompensationHint, DeadLetterQueue, EnqueueDeadLetterArgs}; +pub use dead_letter::{CompensationHint, DeadLetter, DeadLetterQueue, EnqueueDeadLetterArgs}; pub use deferred::DeferredQueue; pub use error::{CrdtError, Result}; pub use policy::{ diff --git a/nodedb-crdt/src/validator/types.rs b/nodedb-crdt/src/validator/types.rs index ae06f1870..ec0ef0e45 100644 --- a/nodedb-crdt/src/validator/types.rs +++ b/nodedb-crdt/src/validator/types.rs @@ -23,7 +23,7 @@ pub enum ValidationOutcome { EvalError { /// The CHECK constraint whose predicate failed to evaluate. constraint_name: String, - /// The underlying evaluation error (currently only `DivisionByZero`). + /// The underlying evaluation error. error: nodedb_query::EvalError, }, } diff --git a/nodedb-graph/src/csr/index/interning.rs b/nodedb-graph/src/csr/index/interning.rs index df9f702fb..48f76b829 100644 --- a/nodedb-graph/src/csr/index/interning.rs +++ b/nodedb-graph/src/csr/index/interning.rs @@ -5,6 +5,7 @@ use std::collections::hash_map::Entry; use super::types::CsrIndex; +use crate::csr::rebuild::journal::{CsrWriteOp, OpOutcome}; impl CsrIndex { /// Get or create a dense ID for a node. @@ -114,6 +115,18 @@ impl CsrIndex { /// is a no-op — the zero sentinel is the initial state and has no meaning. pub fn set_node_surrogate(&mut self, node: &str, surrogate: nodedb_types::Surrogate) { let raw = surrogate.as_u32(); + self.apply_set_node_surrogate(node, raw); + self.journal_record( + || CsrWriteOp::SetNodeSurrogate { + node: node.to_string(), + surrogate: raw, + }, + OpOutcome::Applied, + ); + } + + /// The surrogate bind itself, unjournaled. `raw == 0` is a no-op. + pub(crate) fn apply_set_node_surrogate(&mut self, node: &str, raw: u32) { if raw == 0 { return; } @@ -215,6 +228,23 @@ impl CsrIndex { /// label is silently ignored). Returns `Err(GraphError::NodeOverflow)` if /// the node is new and the partition's node-id space is exhausted. pub fn add_node_label(&mut self, node: &str, label: &str) -> Result { + let result = self.apply_add_node_label(node, label); + self.journal_record( + || CsrWriteOp::AddNodeLabel { + node: node.to_string(), + label: label.to_string(), + }, + OpOutcome::of_label(&result), + ); + result + } + + /// The label add itself, unjournaled. + pub(crate) fn apply_add_node_label( + &mut self, + node: &str, + label: &str, + ) -> Result { let node_id = self.ensure_node(node)?; let Some(label_id) = self.ensure_node_label(label) else { return Ok(false); @@ -225,6 +255,18 @@ impl CsrIndex { /// Remove a label from a node. pub fn remove_node_label(&mut self, node: &str, label: &str) { + self.apply_remove_node_label(node, label); + self.journal_record( + || CsrWriteOp::RemoveNodeLabel { + node: node.to_string(), + label: label.to_string(), + }, + OpOutcome::Applied, + ); + } + + /// The label removal itself, unjournaled. + pub(crate) fn apply_remove_node_label(&mut self, node: &str, label: &str) { let Some(&node_id) = self.node_to_id.get(node) else { return; }; diff --git a/nodedb-graph/src/csr/index/lookup.rs b/nodedb-graph/src/csr/index/lookup.rs index f39af8171..80cf8de13 100644 --- a/nodedb-graph/src/csr/index/lookup.rs +++ b/nodedb-graph/src/csr/index/lookup.rs @@ -14,6 +14,7 @@ use nodedb_mem::ScopedMemory; use super::types::{CsrIndex, Direction}; use crate::GraphError; use crate::csr::LocalNodeId; +use crate::csr::rebuild::journal::{CsrWriteOp, OpOutcome}; /// Contiguous CSR adjacency arrays produced by [`CsrIndex::build_dense`]. pub(crate) struct DenseAdjacency { @@ -127,8 +128,14 @@ impl CsrIndex { /// Returns `Err(GraphError::NodeOverflow)` when the partition's node-id /// space is exhausted (more than `MAX_NODES_PER_CSR` distinct nodes). pub fn add_node(&mut self, name: &str) -> Result { - let raw = self.ensure_node(name)?; - Ok(LocalNodeId::new(raw, self.partition_tag)) + let result = self.ensure_node(name); + self.journal_record( + || CsrWriteOp::AddNode { + name: name.to_string(), + }, + OpOutcome::of(&result), + ); + Ok(LocalNodeId::new(result?, self.partition_tag)) } pub fn node_count(&self) -> usize { diff --git a/nodedb-graph/src/csr/index/mod.rs b/nodedb-graph/src/csr/index/mod.rs index bbd2ac74c..5afdfebfb 100644 --- a/nodedb-graph/src/csr/index/mod.rs +++ b/nodedb-graph/src/csr/index/mod.rs @@ -8,10 +8,12 @@ //! - `mutation` — `add_edge`, `remove_edge`, `remove_node_edges` //! - `lookup` — neighbor queries, accessors, degree, iterators //! - `scoped` — collection-scoped read paths (MATCH / RAG) +//! - `restore` — exact edge writes and the reversals a rollback uses pub mod interning; pub mod lookup; pub mod mutation; +pub mod restore; pub mod scoped; pub mod types; diff --git a/nodedb-graph/src/csr/index/mutation.rs b/nodedb-graph/src/csr/index/mutation.rs index d2e39d913..6a1599036 100644 --- a/nodedb-graph/src/csr/index/mutation.rs +++ b/nodedb-graph/src/csr/index/mutation.rs @@ -3,6 +3,7 @@ //! Edge insert / remove paths and node-edge cleanup. use super::types::CsrIndex; +use crate::csr::rebuild::journal::{CsrWriteOp, OpOutcome}; impl CsrIndex { /// Incrementally add an unweighted edge (goes into mutable buffer). @@ -69,6 +70,31 @@ impl CsrIndex { collection: &str, weight: f64, force_weights: bool, + ) -> Result<(), crate::GraphError> { + let result = self.apply_add_edge(src, label, dst, collection, weight, force_weights); + self.journal_record( + || CsrWriteOp::AddEdge { + src: src.to_string(), + label: label.to_string(), + dst: dst.to_string(), + collection: collection.to_string(), + weight, + force_weights, + }, + OpOutcome::of(&result), + ); + result + } + + /// The edge insert itself, unjournaled. + pub(crate) fn apply_add_edge( + &mut self, + src: &str, + label: &str, + dst: &str, + collection: &str, + weight: f64, + force_weights: bool, ) -> Result<(), crate::GraphError> { let src_id = self.ensure_node(src)?; let dst_id = self.ensure_node(dst)?; @@ -87,8 +113,10 @@ impl CsrIndex { { return Ok(()); } - // Check for duplicates in dense CSR (collection-aware). + // A dense copy is the edge itself: a deleted one comes back. if self.dense_has_edge(src_id, label_id, dst_id, collection_id) { + self.deleted_edges + .remove(&(src_id, label_id, dst_id, collection_id)); return Ok(()); } @@ -107,8 +135,8 @@ impl CsrIndex { self.buffer_in_weights[dst_id as usize].push(weight); } - // If this exact `(src, label, dst, collection)` copy was previously - // deleted, un-delete it. + // A node-edge removal marks buffered edges deleted too. With no dense + // copy that mark names this edge only, so it goes. self.deleted_edges .remove(&(src_id, label_id, dst_id, collection_id)); Ok(()) @@ -134,6 +162,26 @@ impl CsrIndex { label: &str, dst: &str, collection: &str, + ) { + self.apply_remove_edge(src, label, dst, collection); + self.journal_record( + || CsrWriteOp::RemoveEdge { + src: src.to_string(), + label: label.to_string(), + dst: dst.to_string(), + collection: collection.to_string(), + }, + OpOutcome::Applied, + ); + } + + /// The edge removal itself, unjournaled. + pub(crate) fn apply_remove_edge( + &mut self, + src: &str, + label: &str, + dst: &str, + collection: &str, ) { let (Some(&src_id), Some(&dst_id)) = (self.node_to_id.get(src), self.node_to_id.get(dst)) else { @@ -183,6 +231,18 @@ impl CsrIndex { /// Remove ALL edges touching a node. Returns the number of edges removed. pub fn remove_node_edges(&mut self, node: &str) -> usize { + let removed = self.apply_remove_node_edges(node); + self.journal_record( + || CsrWriteOp::RemoveNodeEdges { + node: node.to_string(), + }, + OpOutcome::Applied, + ); + removed + } + + /// The node-edge removal itself, unjournaled. + pub(crate) fn apply_remove_node_edges(&mut self, node: &str) -> usize { let Some(&node_id) = self.node_to_id.get(node) else { return 0; }; diff --git a/nodedb-graph/src/csr/index/restore.rs b/nodedb-graph/src/csr/index/restore.rs new file mode 100644 index 000000000..7b5b0d4b6 --- /dev/null +++ b/nodedb-graph/src/csr/index/restore.rs @@ -0,0 +1,443 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Exact edge writes, and the reversal primitives a rollback of a graph +//! write uses. +//! +//! A rollback puts the index back to what it held before the write: the +//! edge's presence and weight, each node surrogate the write rebound, and +//! each node or node label the write interned. Interning appends, so a +//! rollback withdraws the newest entry first and refuses any other. + +use super::types::CsrIndex; +use crate::GraphError; +use crate::csr::rebuild::journal::{CsrWriteOp, OpOutcome}; + +impl CsrIndex { + /// Weight of the live `(src, label, dst)` edge in `collection`, `None` + /// when that edge is not live. + pub fn edge_weight_in_collection( + &self, + src: &str, + label: &str, + dst: &str, + collection: &str, + ) -> Option { + let src_id = *self.node_to_id.get(src)?; + let dst_id = *self.node_to_id.get(dst)?; + let label_id = *self.label_to_id.get(label)?; + let coll_id = *self.collection_to_id.get(collection)?; + let idx = src_id as usize; + + if let (Some(edges), Some(colls)) = ( + self.buffer_out.get(idx), + self.buffer_out_collections.get(idx), + ) { + let position = edges + .iter() + .zip(colls.iter()) + .position(|(&(l, d), &c)| l == label_id && d == dst_id && c == coll_id); + if let Some(k) = position { + let weight = if self.has_weights { + self.buffer_out_weights + .get(idx) + .and_then(|weights| weights.get(k)) + .copied() + .unwrap_or(1.0) + } else { + 1.0 + }; + return Some(weight); + } + } + + if self + .deleted_edges + .contains(&(src_id, label_id, dst_id, coll_id)) + || idx + 1 >= self.out_offsets.len() + { + return None; + } + let start = self.out_offsets[idx] as usize; + let end = self.out_offsets[idx + 1] as usize; + (start..end) + .find(|&i| { + self.out_labels[i] == label_id + && self.out_targets[i] == dst_id + && self.out_collections.get(i).copied().unwrap_or(0) == coll_id + }) + .map(|i| self.out_edge_weight(i)) + } + + /// Make the `(src, label, dst)` edge in `collection` live with `weight`. + /// + /// A live edge with another weight takes the new one. Returns the weight + /// the edge had while live before, `None` when it was not live, so a + /// rollback can put it back. + pub fn put_edge_in_collection( + &mut self, + src: &str, + label: &str, + dst: &str, + collection: &str, + weight: f64, + ) -> Result, GraphError> { + let result = self.apply_put_edge(src, label, dst, collection, weight); + self.journal_record( + || CsrWriteOp::PutEdge { + src: src.to_string(), + label: label.to_string(), + dst: dst.to_string(), + collection: collection.to_string(), + weight, + }, + OpOutcome::of(&result), + ); + result + } + + /// The edge put itself, unjournaled. + pub(crate) fn apply_put_edge( + &mut self, + src: &str, + label: &str, + dst: &str, + collection: &str, + weight: f64, + ) -> Result, GraphError> { + let prior = self.edge_weight_in_collection(src, label, dst, collection); + match prior { + Some(current) if current == weight => return Ok(prior), + Some(_) => self.apply_remove_edge(src, label, dst, collection), + None => {} + } + let src_id = self.ensure_node(src)?; + let dst_id = self.ensure_node(dst)?; + let label_id = self.ensure_label(label)?; + let coll_id = self.ensure_collection(collection); + if weight != 1.0 && !self.has_weights { + self.enable_weights(); + } + // A deleted dense copy stays deleted: the live copy is the buffer one. + // With no dense copy, a deletion mark names this edge only, so it goes. + if !self.dense_has_edge(src_id, label_id, dst_id, coll_id) { + self.deleted_edges + .remove(&(src_id, label_id, dst_id, coll_id)); + } + self.buffer_out[src_id as usize].push((label_id, dst_id)); + self.buffer_in[dst_id as usize].push((label_id, src_id)); + self.buffer_out_collections[src_id as usize].push(coll_id); + self.buffer_in_collections[dst_id as usize].push(coll_id); + if self.has_weights { + self.buffer_out_weights[src_id as usize].push(weight); + self.buffer_in_weights[dst_id as usize].push(weight); + } + Ok(prior) + } + + /// Put the edge back to `prior`: live with that weight, or absent when + /// `prior` is `None`. + pub fn restore_edge_in_collection( + &mut self, + src: &str, + label: &str, + dst: &str, + collection: &str, + prior: Option, + ) -> Result<(), GraphError> { + match prior { + Some(weight) => self + .put_edge_in_collection(src, label, dst, collection, weight) + .map(drop), + None => { + self.remove_edge_in_collection(src, label, dst, collection); + Ok(()) + } + } + } + + /// Put `node`'s surrogate back to `prior`, `0` for none. + pub fn restore_node_surrogate(&mut self, node: &str, prior: u32) { + self.apply_restore_node_surrogate(node, prior); + self.journal_record( + || CsrWriteOp::RestoreNodeSurrogate { + node: node.to_string(), + prior, + }, + OpOutcome::Applied, + ); + } + + /// The surrogate restore itself, unjournaled. + pub(crate) fn apply_restore_node_surrogate(&mut self, node: &str, prior: u32) { + let Some(&id) = self.node_to_id.get(node) else { + return; + }; + let Some(slot) = self.node_surrogates.get_mut(id as usize) else { + return; + }; + let current = *slot; + if current == prior { + return; + } + *slot = prior; + if current != 0 && self.surrogate_to_local.get(¤t) == Some(&id) { + self.surrogate_to_local.remove(¤t); + } + if prior != 0 { + self.surrogate_to_local.insert(prior, id); + } + } + + /// Whether the node label `label` is interned. + pub fn has_node_label_name(&self, label: &str) -> bool { + self.node_label_to_id.contains_key(label) + } + + /// Withdraw `node`, which a rolled-back write interned. + /// + /// An absent node is a no-op. The node must be the newest one, with no + /// edge and no label: withdrawing any other would renumber or orphan live + /// state, so that is refused. + pub fn withdraw_newest_node(&mut self, node: &str) -> Result<(), GraphError> { + let result = self.apply_withdraw_newest_node(node); + self.journal_record( + || CsrWriteOp::WithdrawNewestNode { + node: node.to_string(), + }, + OpOutcome::of(&result), + ); + result + } + + /// The node withdraw itself, unjournaled. + pub(crate) fn apply_withdraw_newest_node(&mut self, node: &str) -> Result<(), GraphError> { + let Some(&id) = self.node_to_id.get(node) else { + return Ok(()); + }; + let idx = id as usize; + let newest = idx + 1 == self.id_to_node.len(); + let no_buffered_edge = self.buffer_out.get(idx).is_none_or(Vec::is_empty) + && self.buffer_in.get(idx).is_none_or(Vec::is_empty); + let empty_range = |offsets: &[u32]| match (offsets.get(idx), offsets.get(idx + 1)) { + (Some(start), Some(end)) => start == end, + _ => true, + }; + let no_dense_edge = + empty_range(self.out_offsets.as_slice()) && empty_range(self.in_offsets.as_slice()); + let unlabeled = self.node_label_bits.get(idx).copied().unwrap_or(0) == 0; + if !(newest && no_buffered_edge && no_dense_edge && unlabeled) { + return Err(GraphError::WithdrawRefused { + kind: "node", + name: node.to_string(), + }); + } + + self.node_to_id.remove(node); + self.id_to_node.pop(); + // Offsets hold one entry more than there are nodes. + if self.out_offsets.len() > idx + 1 { + self.out_offsets.pop(); + } + if self.in_offsets.len() > idx + 1 { + self.in_offsets.pop(); + } + self.buffer_out.truncate(idx); + self.buffer_in.truncate(idx); + self.buffer_out_weights.truncate(idx); + self.buffer_in_weights.truncate(idx); + self.buffer_out_collections.truncate(idx); + self.buffer_in_collections.truncate(idx); + self.node_label_bits.truncate(idx); + self.node_surrogates.truncate(idx); + self.surrogate_to_local.retain(|_, local| *local != id); + // The id goes back to the pool: no deletion mark may name it. + self.deleted_edges + .retain(|&(src, _, dst, _)| src != id && dst != id); + self.access_counts.truncate(idx); + Ok(()) + } + + /// Withdraw the node label `label`, which a rolled-back write interned. + /// + /// An absent label is a no-op. The label must be the newest one and no + /// node may carry it, or the withdraw is refused. + pub fn withdraw_newest_node_label(&mut self, label: &str) -> Result<(), GraphError> { + let result = self.apply_withdraw_newest_node_label(label); + self.journal_record( + || CsrWriteOp::WithdrawNewestNodeLabel { + label: label.to_string(), + }, + OpOutcome::of(&result), + ); + result + } + + /// The node-label withdraw itself, unjournaled. + pub(crate) fn apply_withdraw_newest_node_label( + &mut self, + label: &str, + ) -> Result<(), GraphError> { + let Some(&id) = self.node_label_to_id.get(label) else { + return Ok(()); + }; + let newest = usize::from(id) + 1 == self.node_label_names.len(); + let bit = 1u64 << id; + let carried = self.node_label_bits.iter().any(|bits| bits & bit != 0); + if !newest || carried { + return Err(GraphError::WithdrawRefused { + kind: "node label", + name: label.to_string(), + }); + } + self.node_label_to_id.remove(label); + self.node_label_names.pop(); + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::csr::index::types::Direction; + use crate::test_support::test_memory; + + #[test] + fn put_edge_replaces_the_weight_of_a_live_edge_and_reports_the_old_one() { + let mut csr = CsrIndex::new(test_memory()); + assert_eq!( + csr.put_edge_in_collection("a", "L", "b", "c", 2.5) + .expect("first put"), + None + ); + assert_eq!( + csr.put_edge_in_collection("a", "L", "b", "c", 9.0) + .expect("second put"), + Some(2.5) + ); + assert_eq!(csr.edge_weight_in_collection("a", "L", "b", "c"), Some(9.0)); + assert_eq!(csr.neighbors("a", None, Direction::Out).len(), 1); + } + + #[test] + fn put_edge_replaces_the_weight_of_a_compacted_edge() { + let mut csr = CsrIndex::new(test_memory()); + csr.put_edge_in_collection("a", "L", "b", "c", 2.5) + .expect("put"); + csr.compact().expect("compact"); + assert_eq!( + csr.put_edge_in_collection("a", "L", "b", "c", 9.0) + .expect("reweigh"), + Some(2.5) + ); + assert_eq!(csr.edge_weight_in_collection("a", "L", "b", "c"), Some(9.0)); + assert_eq!(csr.neighbors("a", None, Direction::Out).len(), 1); + csr.compact().expect("compact again"); + assert_eq!(csr.edge_weight_in_collection("a", "L", "b", "c"), Some(9.0)); + assert_eq!(csr.neighbors("a", None, Direction::Out).len(), 1); + } + + #[test] + fn restoring_an_edge_brings_back_a_deleted_compacted_edge() { + let mut csr = CsrIndex::new(test_memory()); + csr.put_edge_in_collection("a", "L", "b", "c", 1.0) + .expect("put"); + csr.compact().expect("compact"); + csr.remove_edge_in_collection("a", "L", "b", "c"); + assert_eq!(csr.edge_weight_in_collection("a", "L", "b", "c"), None); + + csr.restore_edge_in_collection("a", "L", "b", "c", Some(1.0)) + .expect("restore"); + assert_eq!(csr.edge_weight_in_collection("a", "L", "b", "c"), Some(1.0)); + assert_eq!(csr.neighbors("a", None, Direction::Out).len(), 1); + } + + #[test] + fn re_adding_a_deleted_compacted_edge_brings_it_back() { + let mut csr = CsrIndex::new(test_memory()); + csr.add_edge_in_collection("a", "L", "b", "c") + .expect("edge"); + csr.compact().expect("compact"); + csr.remove_edge_in_collection("a", "L", "b", "c"); + csr.add_edge_in_collection("a", "L", "b", "c") + .expect("re-add"); + assert_eq!(csr.neighbors("a", None, Direction::Out).len(), 1); + } + + #[test] + fn withdrawing_the_newest_isolated_node_removes_it() { + let mut csr = CsrIndex::new(test_memory()); + csr.add_edge("a", "L", "b").expect("edge"); + csr.put_edge_in_collection("b", "L", "z", "c", 1.0) + .expect("edge to z"); + csr.remove_edge_in_collection("b", "L", "z", "c"); + csr.set_node_surrogate("z", nodedb_types::Surrogate::new(42)); + + csr.withdraw_newest_node("z").expect("withdraw"); + + assert!(!csr.contains_node("z")); + assert_eq!(csr.node_count(), 2); + assert_eq!( + csr.node_id_for_surrogate(nodedb_types::Surrogate::new(42)), + None + ); + csr.add_edge("a", "L", "y") + .expect("a later node takes the freed id"); + csr.compact().expect("compact"); + assert_eq!(csr.neighbors("a", None, Direction::Out).len(), 2); + } + + #[test] + fn withdrawing_a_node_that_is_not_the_newest_is_refused() { + let mut csr = CsrIndex::new(test_memory()); + csr.add_node_label("x", "Person").expect("label x"); + csr.remove_node_label("x", "Person"); + csr.add_edge("a", "L", "b").expect("edge"); + assert!(matches!( + csr.withdraw_newest_node("x"), + Err(GraphError::WithdrawRefused { .. }) + )); + assert!(csr.contains_node("x")); + } + + #[test] + fn withdrawing_a_node_with_an_edge_is_refused() { + let mut csr = CsrIndex::new(test_memory()); + csr.add_edge("a", "L", "b").expect("edge"); + assert!(matches!( + csr.withdraw_newest_node("b"), + Err(GraphError::WithdrawRefused { .. }) + )); + } + + #[test] + fn withdrawing_the_newest_unused_node_label_frees_its_slot() { + let mut csr = CsrIndex::new(test_memory()); + csr.add_node_label("x", "Person").expect("label"); + csr.remove_node_label("x", "Person"); + csr.withdraw_newest_node_label("Person").expect("withdraw"); + assert!(!csr.has_node_label_name("Person")); + } + + #[test] + fn withdrawing_a_carried_node_label_is_refused() { + let mut csr = CsrIndex::new(test_memory()); + csr.add_node_label("x", "Person").expect("label"); + assert!(matches!( + csr.withdraw_newest_node_label("Person"), + Err(GraphError::WithdrawRefused { .. }) + )); + } + + #[test] + fn restoring_a_surrogate_unbinds_the_one_a_write_set() { + let mut csr = CsrIndex::new(test_memory()); + csr.add_edge("a", "L", "b").expect("edge"); + csr.set_node_surrogate("a", nodedb_types::Surrogate::new(5)); + csr.restore_node_surrogate("a", 0); + assert_eq!(csr.node_surrogate("a"), None); + assert_eq!( + csr.node_id_for_surrogate(nodedb_types::Surrogate::new(5)), + None + ); + } +} diff --git a/nodedb-graph/src/csr/index/types.rs b/nodedb-graph/src/csr/index/types.rs index 4d99e3fa2..fb6303f11 100644 --- a/nodedb-graph/src/csr/index/types.rs +++ b/nodedb-graph/src/csr/index/types.rs @@ -147,6 +147,10 @@ pub struct CsrIndex { /// reserve bytes against the bound database, tenant, and `EngineId::Graph` /// before allocating and release them on drop via `ReservationToken`. pub(crate) memory: ScopedMemory, + + /// Mutations recorded while a rebuild of this index runs. `None` when + /// no rebuild runs. See [`crate::csr::rebuild`]. + pub(crate) rebuild_journal: Option, } impl CsrIndex { @@ -191,6 +195,7 @@ impl CsrIndex { query_epoch: 0, partition_tag: crate::csr::local_node_id::next_partition_tag(), memory, + rebuild_journal: None, } } diff --git a/nodedb-graph/src/csr/mod.rs b/nodedb-graph/src/csr/mod.rs index ceccab3ac..bf9aa654b 100644 --- a/nodedb-graph/src/csr/mod.rs +++ b/nodedb-graph/src/csr/mod.rs @@ -6,6 +6,7 @@ pub mod index; pub mod local_node_id; pub mod memory; pub mod persist; +pub mod rebuild; pub mod slice_accessors; pub mod statistics; pub mod weights; diff --git a/nodedb-graph/src/csr/persist.rs b/nodedb-graph/src/csr/persist.rs index 896bbd3ec..7c39b345a 100644 --- a/nodedb-graph/src/csr/persist.rs +++ b/nodedb-graph/src/csr/persist.rs @@ -291,6 +291,7 @@ impl CsrIndex { query_epoch: 0, partition_tag: crate::csr::local_node_id::next_partition_tag(), memory, + rebuild_journal: None, }) } @@ -394,6 +395,7 @@ impl CsrIndex { query_epoch: 0, partition_tag: crate::csr::local_node_id::next_partition_tag(), memory, + rebuild_journal: None, } } } diff --git a/nodedb-graph/src/csr/rebuild/install.rs b/nodedb-graph/src/csr/rebuild/install.rs new file mode 100644 index 000000000..740034bc0 --- /dev/null +++ b/nodedb-graph/src/csr/rebuild/install.rs @@ -0,0 +1,185 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Finishing a rebuild on the owning thread: restore the compacted copy +//! and replay the journal onto it. + +use nodedb_mem::ScopedMemory; + +use super::seed::{CsrRebuilt, NodeLabelState, restore_checkpoint}; +use crate::GraphError; +use crate::csr::index::CsrIndex; + +/// Most node labels one index interns. The label bitset is a `u64`. +const MAX_NODE_LABELS: usize = 64; + +impl CsrIndex { + /// Close the journal of `rebuilt` and return the copy that replaces + /// this index. + /// + /// The copy holds the snapshot, compacted, plus every mutation this + /// index took since `begin_rebuild`, in order. It keeps this index's + /// partition tag, so node ids handed out before the swap stay valid. + /// The caller installs it in one step. On error the journal is closed, + /// this index stays as it is, and the copy is dropped: + /// + /// - [`GraphError::RebuildSuperseded`]: no journal of this rebuild is open. + /// - [`GraphError::RebuildJournalOverflow`]: the journal hit its bound. + /// - [`GraphError::RebuildReplayDiverged`]: a replayed mutation returned + /// another outcome than it did here. + /// - [`GraphError::RebuildSnapshotInvalid`]: the copy does not decode. + pub fn finish_rebuild( + &mut self, + rebuilt: CsrRebuilt, + memory: ScopedMemory, + ) -> Result { + let journal = match self.rebuild_journal.take() { + Some(journal) if journal.token == rebuilt.token => journal, + other => { + self.rebuild_journal = other; + return Err(GraphError::RebuildSuperseded); + } + }; + let ops = journal.into_ops()?; + let mut copy = restore_checkpoint(&rebuilt.checkpoint, memory)?; + copy.install_node_labels(rebuilt.node_labels)?; + for (op, live_outcome) in &ops { + if copy.replay_op(op) != *live_outcome { + return Err(GraphError::RebuildReplayDiverged { op: op.kind() }); + } + } + copy.partition_tag = self.partition_tag; + Ok(copy) + } + + /// Put the node labels a rebuild carried onto a freshly restored copy. + fn install_node_labels(&mut self, labels: NodeLabelState) -> Result<(), GraphError> { + if labels.bits.len() != self.id_to_node.len() { + return Err(GraphError::RebuildSnapshotInvalid { + detail: format!( + "node label bitsets cover {} nodes, the snapshot holds {}", + labels.bits.len(), + self.id_to_node.len() + ), + }); + } + if labels.names.len() > MAX_NODE_LABELS { + return Err(GraphError::RebuildSnapshotInvalid { + detail: format!( + "{} node labels exceed the {MAX_NODE_LABELS}-label bitset", + labels.names.len() + ), + }); + } + self.node_label_to_id = labels + .names + .iter() + .enumerate() + .map(|(id, name)| (name.clone(), id as u8)) + .collect(); + self.node_label_names = labels.names; + self.node_label_bits = labels.bits; + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use nodedb_types::Surrogate; + + use crate::GraphError; + use crate::csr::index::{CsrIndex, Direction}; + use crate::test_support::test_memory; + + const JOURNAL_BYTES: usize = 1 << 20; + + fn seeded() -> CsrIndex { + let mut csr = CsrIndex::new(test_memory()); + csr.add_edge_in_collection("a", "L", "b", "c").unwrap(); + csr.add_edge_in_collection("b", "L", "c", "c").unwrap(); + csr.add_node_label("a", "Person").unwrap(); + csr.set_node_surrogate("a", Surrogate::new(7)); + csr + } + + fn out_of(csr: &CsrIndex, node: &str) -> Vec { + let mut dsts: Vec = csr + .neighbors(node, None, Direction::Out) + .into_iter() + .map(|(_, d)| d) + .collect(); + dsts.sort(); + dsts + } + + #[test] + fn writes_during_the_build_reach_the_installed_copy() { + let mut live = seeded(); + let seed = live.begin_rebuild(JOURNAL_BYTES).unwrap(); + + // Writes that land after the snapshot. + live.add_edge_in_collection("a", "L", "d", "c").unwrap(); + live.remove_edge_in_collection("b", "L", "c", "c"); + live.put_edge_in_collection("x", "L", "y", "c", 2.5) + .unwrap(); + live.add_node_label("x", "Person").unwrap(); + live.set_node_surrogate("x", Surrogate::new(9)); + + let rebuilt = seed.build(test_memory()).unwrap(); + let tag_before = live.partition_tag; + let copy = live.finish_rebuild(rebuilt, test_memory()).unwrap(); + + assert_eq!(out_of(©, "a"), vec!["b".to_string(), "d".to_string()]); + assert!( + out_of(©, "b").is_empty(), + "a delete during the build carries over" + ); + assert_eq!( + copy.edge_weight_in_collection("x", "L", "y", "c"), + Some(2.5) + ); + let x = copy.node_id_raw("x").unwrap(); + assert!(copy.node_has_label(x, "Person")); + let a = copy.node_id_raw("a").unwrap(); + assert!(copy.node_has_label(a, "Person"), "snapshot labels survive"); + assert_eq!(copy.node_id_for_surrogate(Surrogate::new(9)), Some("x")); + assert_eq!(copy.partition_tag, tag_before); + assert!(!live.rebuild_in_progress(), "finishing closes the journal"); + } + + #[test] + fn a_result_from_another_rebuild_is_refused() { + let mut live = seeded(); + let seed = live.begin_rebuild(JOURNAL_BYTES).unwrap(); + live.abort_rebuild(seed.token()); + let _second = live.begin_rebuild(JOURNAL_BYTES).unwrap(); + let rebuilt = seed.build(test_memory()).unwrap(); + assert!(matches!( + live.finish_rebuild(rebuilt, test_memory()), + Err(GraphError::RebuildSuperseded) + )); + assert!(live.rebuild_in_progress(), "the newer journal stays open"); + } + + #[test] + fn a_second_rebuild_is_refused_while_one_runs() { + let mut live = seeded(); + let _seed = live.begin_rebuild(JOURNAL_BYTES).unwrap(); + assert!(matches!( + live.begin_rebuild(JOURNAL_BYTES), + Err(GraphError::RebuildInProgress) + )); + } + + #[test] + fn journal_overflow_refuses_the_copy_and_keeps_the_live_writes() { + let mut live = seeded(); + let seed = live.begin_rebuild(1).unwrap(); + live.add_edge_in_collection("a", "L", "z", "c").unwrap(); + let rebuilt = seed.build(test_memory()).unwrap(); + assert!(matches!( + live.finish_rebuild(rebuilt, test_memory()), + Err(GraphError::RebuildJournalOverflow { cap_bytes: 1 }) + )); + assert!(out_of(&live, "a").contains(&"z".to_string())); + } +} diff --git a/nodedb-graph/src/csr/rebuild/journal.rs b/nodedb-graph/src/csr/rebuild/journal.rs new file mode 100644 index 000000000..86edf9052 --- /dev/null +++ b/nodedb-graph/src/csr/rebuild/journal.rs @@ -0,0 +1,301 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The write journal a `CsrIndex` keeps while a rebuild of it runs. +//! +//! Every public mutation records itself here with the outcome it had on +//! the live index. At cutover the journal replays onto the rebuilt copy, +//! so the copy holds every write the live index took after the snapshot. +//! +//! The journal is bounded in bytes. Past the bound it drops its entries, +//! frees their memory and marks itself overflowed. The cutover then +//! refuses the rebuilt copy with [`GraphError::RebuildJournalOverflow`]. +//! The live index keeps every write either way. + +use std::mem::size_of; + +use crate::GraphError; +use crate::csr::index::CsrIndex; + +/// One mutation of the live index, recorded for replay. +#[derive(Debug, Clone, PartialEq)] +pub(crate) enum CsrWriteOp { + AddEdge { + src: String, + label: String, + dst: String, + collection: String, + weight: f64, + force_weights: bool, + }, + PutEdge { + src: String, + label: String, + dst: String, + collection: String, + weight: f64, + }, + RemoveEdge { + src: String, + label: String, + dst: String, + collection: String, + }, + RemoveNodeEdges { + node: String, + }, + AddNode { + name: String, + }, + SetNodeSurrogate { + node: String, + surrogate: u32, + }, + RestoreNodeSurrogate { + node: String, + prior: u32, + }, + AddNodeLabel { + node: String, + label: String, + }, + RemoveNodeLabel { + node: String, + label: String, + }, + WithdrawNewestNode { + node: String, + }, + WithdrawNewestNodeLabel { + label: String, + }, +} + +impl CsrWriteOp { + /// Name of the operation, for errors. + pub(crate) fn kind(&self) -> &'static str { + match self { + Self::AddEdge { .. } => "add_edge", + Self::PutEdge { .. } => "put_edge", + Self::RemoveEdge { .. } => "remove_edge", + Self::RemoveNodeEdges { .. } => "remove_node_edges", + Self::AddNode { .. } => "add_node", + Self::SetNodeSurrogate { .. } => "set_node_surrogate", + Self::RestoreNodeSurrogate { .. } => "restore_node_surrogate", + Self::AddNodeLabel { .. } => "add_node_label", + Self::RemoveNodeLabel { .. } => "remove_node_label", + Self::WithdrawNewestNode { .. } => "withdraw_newest_node", + Self::WithdrawNewestNodeLabel { .. } => "withdraw_newest_node_label", + } + } + + /// Bytes the entry holds: the enum itself plus its string contents. + fn byte_cost(&self) -> usize { + let strings = match self { + Self::AddEdge { + src, + label, + dst, + collection, + .. + } + | Self::PutEdge { + src, + label, + dst, + collection, + .. + } + | Self::RemoveEdge { + src, + label, + dst, + collection, + } => src.len() + label.len() + dst.len() + collection.len(), + Self::RemoveNodeEdges { node } + | Self::SetNodeSurrogate { node, .. } + | Self::RestoreNodeSurrogate { node, .. } + | Self::WithdrawNewestNode { node } => node.len(), + Self::AddNode { name } => name.len(), + Self::AddNodeLabel { node, label } | Self::RemoveNodeLabel { node, label } => { + node.len() + label.len() + } + Self::WithdrawNewestNodeLabel { label } => label.len(), + }; + size_of::<(Self, OpOutcome)>() + strings + } +} + +/// What a mutation returned on the index it ran against. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum OpOutcome { + /// The call returned `Ok`, or returns nothing. + Applied, + /// `add_node_label` returned `Ok(false)`: the label limit ignored it. + Ignored, + /// The call returned an error. + Failed, +} + +impl OpOutcome { + pub(crate) fn of(result: &Result) -> Self { + if result.is_ok() { + Self::Applied + } else { + Self::Failed + } + } + + pub(crate) fn of_label(result: &Result) -> Self { + match result { + Ok(true) => Self::Applied, + Ok(false) => Self::Ignored, + Err(_) => Self::Failed, + } + } +} + +/// Mutations recorded since a rebuild's snapshot, oldest first. +#[derive(Debug)] +pub struct CsrJournal { + pub(crate) token: u64, + cap_bytes: usize, + used_bytes: usize, + ops: Vec<(CsrWriteOp, OpOutcome)>, + overflowed: bool, +} + +impl CsrJournal { + pub(crate) fn new(token: u64, cap_bytes: usize) -> Self { + Self { + token, + cap_bytes, + used_bytes: 0, + ops: Vec::new(), + overflowed: false, + } + } + + fn record(&mut self, op: CsrWriteOp, outcome: OpOutcome) { + if self.overflowed { + return; + } + let cost = op.byte_cost(); + if self.used_bytes.saturating_add(cost) > self.cap_bytes { + self.overflowed = true; + self.ops = Vec::new(); + self.used_bytes = 0; + return; + } + self.used_bytes += cost; + self.ops.push((op, outcome)); + } + + /// The recorded entries, or the overflow error when the bound was hit. + pub(crate) fn into_ops(self) -> Result, GraphError> { + if self.overflowed { + return Err(GraphError::RebuildJournalOverflow { + cap_bytes: self.cap_bytes, + }); + } + Ok(self.ops) + } +} + +impl CsrIndex { + /// Record a mutation when a rebuild journal is open. `op` runs only + /// then, so an index with no rebuild allocates nothing here. + pub(crate) fn journal_record(&mut self, op: impl FnOnce() -> CsrWriteOp, outcome: OpOutcome) { + if let Some(journal) = self.rebuild_journal.as_mut() { + journal.record(op(), outcome); + } + } + + /// Run one recorded mutation against this index and return its outcome. + pub(crate) fn replay_op(&mut self, op: &CsrWriteOp) -> OpOutcome { + match op { + CsrWriteOp::AddEdge { + src, + label, + dst, + collection, + weight, + force_weights, + } => OpOutcome::of(&self.apply_add_edge( + src, + label, + dst, + collection, + *weight, + *force_weights, + )), + CsrWriteOp::PutEdge { + src, + label, + dst, + collection, + weight, + } => OpOutcome::of(&self.apply_put_edge(src, label, dst, collection, *weight)), + CsrWriteOp::RemoveEdge { + src, + label, + dst, + collection, + } => { + self.apply_remove_edge(src, label, dst, collection); + OpOutcome::Applied + } + CsrWriteOp::RemoveNodeEdges { node } => { + self.apply_remove_node_edges(node); + OpOutcome::Applied + } + CsrWriteOp::AddNode { name } => OpOutcome::of(&self.ensure_node(name)), + CsrWriteOp::SetNodeSurrogate { node, surrogate } => { + self.apply_set_node_surrogate(node, *surrogate); + OpOutcome::Applied + } + CsrWriteOp::RestoreNodeSurrogate { node, prior } => { + self.apply_restore_node_surrogate(node, *prior); + OpOutcome::Applied + } + CsrWriteOp::AddNodeLabel { node, label } => { + OpOutcome::of_label(&self.apply_add_node_label(node, label)) + } + CsrWriteOp::RemoveNodeLabel { node, label } => { + self.apply_remove_node_label(node, label); + OpOutcome::Applied + } + CsrWriteOp::WithdrawNewestNode { node } => { + OpOutcome::of(&self.apply_withdraw_newest_node(node)) + } + CsrWriteOp::WithdrawNewestNodeLabel { label } => { + OpOutcome::of(&self.apply_withdraw_newest_node_label(label)) + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn overflow_drops_entries_and_reports_the_bound() { + let op = CsrWriteOp::AddNode { + name: "n".to_string(), + }; + let cap = op.byte_cost() * 2; + let mut journal = CsrJournal::new(1, cap); + journal.record(op.clone(), OpOutcome::Applied); + journal.record(op.clone(), OpOutcome::Applied); + assert_eq!(journal.ops.len(), 2); + journal.record(op, OpOutcome::Applied); + assert!( + journal.ops.is_empty(), + "an overflowed journal frees its entries" + ); + assert!(matches!( + journal.into_ops(), + Err(GraphError::RebuildJournalOverflow { cap_bytes }) if cap_bytes == cap + )); + } +} diff --git a/nodedb-graph/src/csr/rebuild/mod.rs b/nodedb-graph/src/csr/rebuild/mod.rs new file mode 100644 index 000000000..c707e3ea8 --- /dev/null +++ b/nodedb-graph/src/csr/rebuild/mod.rs @@ -0,0 +1,15 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Rebuild of a live `CsrIndex` off its owning thread, with no lost writes. +//! +//! - `begin_rebuild` snapshots the index and opens a write journal. +//! - `CsrRebuildSeed::build` compacts the snapshot on any thread. +//! - `finish_rebuild` restores the compacted copy, replays the journal onto +//! it and returns it for the caller to install in one step. + +pub mod install; +pub mod journal; +pub mod seed; + +pub use journal::CsrJournal; +pub use seed::{CsrRebuildSeed, CsrRebuilt}; diff --git a/nodedb-graph/src/csr/rebuild/seed.rs b/nodedb-graph/src/csr/rebuild/seed.rs new file mode 100644 index 000000000..f8bc28129 --- /dev/null +++ b/nodedb-graph/src/csr/rebuild/seed.rs @@ -0,0 +1,126 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Starting a rebuild on the owning thread, and the build that runs off it. + +use std::sync::atomic::{AtomicU64, Ordering}; + +use nodedb_mem::ScopedMemory; + +use super::journal::CsrJournal; +use crate::GraphError; +use crate::csr::index::CsrIndex; + +static NEXT_REBUILD_TOKEN: AtomicU64 = AtomicU64::new(1); + +/// Node labels as the index holds them. The checkpoint format omits them, +/// so a rebuild carries them beside it. +#[derive(Debug, Clone)] +pub(crate) struct NodeLabelState { + /// Label names in id order. + pub(crate) names: Vec, + /// Label bitset per node id. + pub(crate) bits: Vec, +} + +/// A rebuild's input, taken on the owning thread. Every field is `Send`. +#[derive(Debug)] +pub struct CsrRebuildSeed { + token: u64, + checkpoint: Vec, + node_labels: NodeLabelState, +} + +/// A compacted copy on its way back to the owning thread. Every field is +/// `Send`. +#[derive(Debug)] +pub struct CsrRebuilt { + pub(crate) token: u64, + pub(crate) checkpoint: Vec, + pub(crate) node_labels: NodeLabelState, +} + +impl CsrRebuildSeed { + /// The token that ties this rebuild to the journal on the live index. + pub fn token(&self) -> u64 { + self.token + } + + /// Restore the snapshot, compact it and serialize the result. Runs on + /// any thread. `memory` bounds the copy's allocations. + pub fn build(self, memory: ScopedMemory) -> Result { + let mut copy = restore_checkpoint(&self.checkpoint, memory)?; + copy.compact()?; + let checkpoint = copy.checkpoint_to_bytes()?; + Ok(CsrRebuilt { + token: self.token, + checkpoint, + node_labels: self.node_labels, + }) + } +} + +impl CsrRebuilt { + /// The token of the rebuild that produced this copy. + pub fn token(&self) -> u64 { + self.token + } +} + +impl CsrIndex { + /// Snapshot this index and open its write journal. + /// + /// Every later mutation records itself until `finish_rebuild` or + /// `abort_rebuild` closes the journal. `max_journal_bytes` bounds the + /// journal. Returns [`GraphError::RebuildInProgress`] when a journal is + /// already open. + pub fn begin_rebuild( + &mut self, + max_journal_bytes: usize, + ) -> Result { + if self.rebuild_journal.is_some() { + return Err(GraphError::RebuildInProgress); + } + let checkpoint = self.checkpoint_to_bytes()?; + let token = NEXT_REBUILD_TOKEN.fetch_add(1, Ordering::Relaxed); + self.rebuild_journal = Some(CsrJournal::new(token, max_journal_bytes)); + Ok(CsrRebuildSeed { + token, + checkpoint, + node_labels: NodeLabelState { + names: self.node_label_names.clone(), + bits: self.node_label_bits.clone(), + }, + }) + } + + /// Whether a rebuild journal is open on this index. + pub fn rebuild_in_progress(&self) -> bool { + self.rebuild_journal.is_some() + } + + /// Close the journal of rebuild `token`. The index itself is unchanged. + /// A journal of another rebuild stays open. + pub fn abort_rebuild(&mut self, token: u64) { + if self + .rebuild_journal + .as_ref() + .is_some_and(|journal| journal.token == token) + { + self.rebuild_journal = None; + } + } +} + +/// Decode a checkpoint a rebuild produced. +pub(crate) fn restore_checkpoint( + bytes: &[u8], + memory: ScopedMemory, +) -> Result { + CsrIndex::from_checkpoint(bytes, memory) + .map_err(|e| GraphError::RebuildSnapshotInvalid { + detail: e.to_string(), + })? + .ok_or_else(|| GraphError::RebuildSnapshotInvalid { + detail: "checkpoint bytes carry no CSR header".to_string(), + }) +} diff --git a/nodedb-graph/src/error.rs b/nodedb-graph/src/error.rs index 7168c820e..962cea9d7 100644 --- a/nodedb-graph/src/error.rs +++ b/nodedb-graph/src/error.rs @@ -55,4 +55,39 @@ pub enum GraphError { /// Callers should apply backpressure and retry after memory is released. #[error("graph memory budget rejected: {0}")] MemoryBudget(#[from] MemError), + + /// A rollback asked to withdraw an interned node or node label that is + /// not the newest one, or that something still refers to. Withdrawing it + /// would renumber or orphan live state. + #[error("cannot withdraw {kind} '{name}': it is not the newest {kind} or it is still in use")] + WithdrawRefused { kind: &'static str, name: String }, + + /// A rebuild of this index is already running. The partition holds one + /// write journal at a time. + #[error("a rebuild of this CSR index is already running")] + RebuildInProgress, + + /// The index a rebuild started from is no longer the live one: it was + /// dropped, replaced, or its rebuild was aborted. The rebuilt copy is + /// discarded and the live index stays as it is. + #[error("the CSR index this rebuild started from is no longer live; rebuild discarded")] + RebuildSuperseded, + + /// The writes made during a rebuild exceeded the journal bound. The + /// rebuilt copy is discarded, the live index keeps every write, and a + /// new rebuild can run. + #[error( + "CSR writes during the rebuild exceeded the {cap_bytes}-byte journal bound; \ + rebuild discarded, live index unchanged; run REINDEX again" + )] + RebuildJournalOverflow { cap_bytes: usize }, + + /// A journaled write returned a different outcome on the rebuilt index + /// than on the live one. The rebuilt copy is discarded. + #[error("CSR rebuild replay of '{op}' diverged from the live index; rebuild discarded")] + RebuildReplayDiverged { op: &'static str }, + + /// The snapshot a rebuild carries does not decode into an index. + #[error("CSR rebuild snapshot is invalid: {detail}")] + RebuildSnapshotInvalid { detail: String }, } diff --git a/nodedb-physical/Cargo.toml b/nodedb-physical/Cargo.toml index 38af8ef5e..2dca5186c 100644 --- a/nodedb-physical/Cargo.toml +++ b/nodedb-physical/Cargo.toml @@ -16,6 +16,7 @@ nodedb-graph = { workspace = true } nodedb-query = { workspace = true } nodedb-sql = { workspace = true } nodedb-types = { workspace = true } +rust_decimal = { workspace = true } serde = { workspace = true } thiserror = { workspace = true } zerompk = { workspace = true } diff --git a/nodedb-physical/src/convert_context.rs b/nodedb-physical/src/convert_context.rs index d017c09dc..8f1afda29 100644 --- a/nodedb-physical/src/convert_context.rs +++ b/nodedb-physical/src/convert_context.rs @@ -21,10 +21,9 @@ use crate::SurrogateAssigner; /// implementations wrap it (Origin adds catalog/WAL handles, Lite passes /// it through unchanged). pub struct SharedConvertContext { - /// Database scope for vShard computation. All - /// `VShardId::from_collection_in_database` calls inside the converter - /// must use this value so collections in different databases route to - /// distinct shards. + /// Database scope for vShard computation. Every `CollectionKey` the + /// converter builds uses this value, so collections in different + /// databases route to distinct shards. pub database_id: DatabaseId, /// Per-tenant maximum vector dimension (0 = unlimited). Checked during diff --git a/nodedb-physical/src/kv_atomic/compute.rs b/nodedb-physical/src/kv_atomic/compute.rs new file mode 100644 index 000000000..02a7e2250 --- /dev/null +++ b/nodedb-physical/src/kv_atomic/compute.rs @@ -0,0 +1,543 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Pure value computation for `INCR`/`INCR_FLOAT`/`CAS`/`GETSET`. +//! +//! Every executor computes a stored value with these functions. On Origin +//! that is the autocommit `KvEngine` methods, transaction staging, the +//! resolve handlers, and WAL replay. On Lite it is the KV write path. All of +//! them store the same bytes for the same op. +//! +//! A body has one of two shapes ([`kv_body_shape`]). A typed row (a msgpack +//! map) keeps its typed column semantics. A raw body (the single-`value` SQL +//! form, RESP `SET`) is a byte string. `INCR` and `INCR_FLOAT` read it as +//! decimal text by the Redis rules and store the result as decimal text. + +use std::collections::HashMap; + +use nodedb_query::msgpack_scan::{KvBodyShape, kv_body_shape, row_to_kv_body}; +use nodedb_types::Value; + +use super::counter_fault::CounterFault; +use super::error::AtomicComputeError; +use super::float_text; +use crate::physical_plan::KvCounterShape; + +/// The field of a typed row an atomic never targets. +const KEY_FIELD: &str = "key"; + +/// Decode a map-shaped body into its typed columns. Returns `Ok(None)` for a +/// raw body, and `TypeMismatch` for a map-shaped body that does not decode. +fn typed_row(bytes: &[u8]) -> Result>, AtomicComputeError> { + if kv_body_shape(bytes) != KvBodyShape::Map { + return Ok(None); + } + match nodedb_types::value_from_msgpack(bytes) { + Ok(Value::Object(map)) => Ok(Some(map)), + Ok(other) => Err(AtomicComputeError::TypeMismatch { + detail: format!("stored row is {}, not an object", other.type_name()), + }), + Err(e) => Err(AtomicComputeError::TypeMismatch { + detail: format!("stored row does not decode: {e}"), + }), + } +} + +/// Encode typed columns back into a map-shaped body. The fields are written +/// in key order, so every replica and every WAL replay stores the same +/// bytes. +fn encode_map(map: HashMap) -> Result, AtomicComputeError> { + row_to_kv_body(&Value::Object(map), KvBodyShape::Map).map_err(|e| AtomicComputeError::Encode { + detail: format!("typed row re-encode: {e}"), + }) +} + +/// The column an atomic reads and writes in a typed row: the first column +/// in key order that `pick` accepts, never the `key` column. +/// +/// Key order is the order the row is stored in. A `HashMap` iterates in a +/// per-process random order, so choosing by iteration order lets two +/// replicas move two different columns. +fn target_field( + map: &HashMap, + pick: impl Fn(&Value) -> Option, +) -> Option<(String, T)> { + let mut chosen: Option<(&String, T)> = None; + for (name, value) in map { + if name == KEY_FIELD { + continue; + } + if chosen.as_ref().is_some_and(|(best, _)| *best <= name) { + continue; + } + if let Some(picked) = pick(value) { + chosen = Some((name, picked)); + } + } + chosen.map(|(name, picked)| (name.clone(), picked)) +} + +/// The i64 an `INCR` reads from a typed column. +fn column_i64(value: &Value) -> Option { + match value { + Value::Integer(i) => Some(*i), + Value::Float(f) => integral_f64_to_i64(*f), + _ => None, + } +} + +/// The f64 an `INCR_FLOAT` reads from a typed column. +fn column_f64(value: &Value) -> Option { + match value { + Value::Float(f) => Some(*f), + Value::Integer(i) => Some(*i as f64), + _ => None, + } +} + +/// The string a `CAS` or `GETSET` addresses in a typed column. +fn column_string(value: &Value) -> Option { + match value { + Value::String(s) => Some(s.clone()), + _ => None, + } +} + +/// A whole `f64` inside the i64 range, as an i64. +fn integral_f64_to_i64(v: f64) -> Option { + (v.fract() == 0.0 && v >= i64::MIN as f64 && v <= i64::MAX as f64).then_some(v as i64) +} + +fn not_an_integer_column() -> AtomicComputeError { + AtomicComputeError::TypeMismatch { + detail: "row has no integer column".into(), + } +} + +fn not_a_numeric_column() -> AtomicComputeError { + AtomicComputeError::TypeMismatch { + detail: "row has no numeric column".into(), + } +} + +/// Read a raw body as a decimal i64 by the Redis rule. +fn parse_raw_i64(bytes: &[u8]) -> Result { + std::str::from_utf8(bytes) + .ok() + .filter(|text| is_canonical_integer(text)) + .and_then(|text| text.parse::().ok()) + .ok_or(AtomicComputeError::Counter(CounterFault::NotAnInteger)) +} + +/// The Redis integer grammar: `0`, or an optional `-` then digits with no +/// leading zero. A `+` sign, whitespace, and an empty body are refused. +fn is_canonical_integer(text: &str) -> bool { + let digits = text.strip_prefix('-').unwrap_or(text); + text == "0" + || (digits + .bytes() + .next() + .is_some_and(|b| (b'1'..=b'9').contains(&b)) + && digits.bytes().all(|b| b.is_ascii_digit())) +} + +/// The raw body for an integer: its decimal text, the text +/// `scalar_to_raw_bytes` writes for the same value. +fn raw_decimal(v: i64) -> Vec { + v.to_string().into_bytes() +} + +/// The row an absent key becomes under a typed [`KvCounterShape`]: the +/// template with `column` set to `value`. +fn fresh_typed_row( + column: &Option, + template: &[u8], + value: Value, + missing_column: AtomicComputeError, +) -> Result, AtomicComputeError> { + let column = column.as_ref().ok_or(missing_column)?; + let mut map = typed_row(template)?.ok_or(AtomicComputeError::TypeMismatch { + detail: "fresh row template is not a typed row".into(), + })?; + map.insert(column.clone(), value); + encode_map(map) +} + +/// Compute the new value for `INCR`, given the current body (if any). +/// Returns `(new_i64, new_bytes)`. +/// +/// A typed row keeps its shape: the integer column [`target_field`] picks +/// moves, and every other column stays. A raw body is decimal text in and +/// decimal text out. An absent key starts at 0 and takes `shape`. +pub fn incr( + current: Option<&[u8]>, + delta: i64, + shape: &KvCounterShape, +) -> Result<(i64, Vec), AtomicComputeError> { + let overflow = AtomicComputeError::Counter(CounterFault::IntegerOverflow); + let Some(bytes) = current else { + let written = match shape { + KvCounterShape::Raw => raw_decimal(delta), + KvCounterShape::Typed { column, template } => fresh_typed_row( + column, + template, + Value::Integer(delta), + not_an_integer_column(), + )?, + }; + return Ok((delta, written)); + }; + if let Some(mut map) = typed_row(bytes)? { + let (field, old_i64) = target_field(&map, column_i64).ok_or(not_an_integer_column())?; + let new_i64 = old_i64.checked_add(delta).ok_or(overflow)?; + map.insert(field, Value::Integer(new_i64)); + return Ok((new_i64, encode_map(map)?)); + } + let new_i64 = parse_raw_i64(bytes)?.checked_add(delta).ok_or(overflow)?; + Ok((new_i64, raw_decimal(new_i64))) +} + +/// Compute the new value for `INCR_FLOAT`. `delta` is the client's decimal +/// text. Returns `(new_f64, new_bytes)`. +/// +/// A typed row keeps its shape, as in [`incr`], and its column adds in +/// `f64`. A raw body is decimal text in and decimal text out, added exactly +/// by the Redis rules (see `float_text`). An absent key starts at 0 and takes +/// `shape`. +pub fn incr_float( + current: Option<&[u8]>, + delta: &str, + shape: &KvCounterShape, +) -> Result<(f64, Vec), AtomicComputeError> { + let Some(bytes) = current else { + return match shape { + KvCounterShape::Raw => float_text::fresh(delta), + KvCounterShape::Typed { column, template } => { + let value = float_text::delta_to_f64(delta)?; + let written = fresh_typed_row( + column, + template, + Value::Float(value), + not_a_numeric_column(), + )?; + Ok((value, written)) + } + }; + }; + let Some(mut map) = typed_row(bytes)? else { + return float_text::add(bytes, delta); + }; + let delta = float_text::delta_to_f64(delta)?; + let (field, old_f64) = target_field(&map, column_f64).ok_or(not_a_numeric_column())?; + let new_f64 = old_f64 + delta; + if !new_f64.is_finite() { + return Err(AtomicComputeError::Counter(CounterFault::NonFinite)); + } + map.insert(field, Value::Float(new_f64)); + Ok((new_f64, encode_map(map)?)) +} + +/// Write `new_value` into the string column of the typed row `row` and +/// encode it. `column` is the column [`target_field`] picked. +fn swap_string_column( + mut row: HashMap, + column: String, + new_value: &[u8], +) -> Result, AtomicComputeError> { + row.insert( + column, + Value::String(String::from_utf8_lossy(new_value).into_owned()), + ); + encode_map(row) +} + +/// A typed row and its string column, when `current` is a typed row with +/// one. [`cas`] and [`getset`] address the same column. +fn string_column(current: Option<&[u8]>) -> Option<(HashMap, String, String)> { + let row = typed_row(current?).ok().flatten()?; + let (column, text) = target_field(&row, column_string)?; + Some((row, column, text)) +} + +/// Compute the CAS outcome: whether `expected` matches the current value, +/// and the bytes to write when it does. +/// +/// The current value matches when its bytes equal `expected`, or when it is +/// a typed row whose string column holds `expected`. A typed row with a +/// string column keeps its shape: only that column is swapped. +pub fn cas( + current: Option<&[u8]>, + expected: &[u8], + new_value: &[u8], +) -> Result<(bool, Vec), AtomicComputeError> { + let Some(cur) = current else { + return Ok(if expected.is_empty() { + (true, new_value.to_vec()) + } else { + (false, Vec::new()) + }); + }; + let typed = string_column(current); + let column_matches = typed + .as_ref() + .is_some_and(|(_, _, text)| *text == String::from_utf8_lossy(expected)); + if cur != expected && !column_matches { + return Ok((false, Vec::new())); + } + let write_bytes = match typed { + Some((row, column, _)) => swap_string_column(row, column, new_value)?, + None => new_value.to_vec(), + }; + Ok((true, write_bytes)) +} + +/// Compute the bytes to write for `GETSET`: the string column of a typed +/// row swapped in place, or a plain overwrite. +pub fn getset(current: Option<&[u8]>, new_value: &[u8]) -> Result, AtomicComputeError> { + match string_column(current) { + Some((row, column, _)) => swap_string_column(row, column, new_value), + None => Ok(new_value.to_vec()), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + static RAW: KvCounterShape = KvCounterShape::Raw; + + /// A typed shape moving `column`, with `rest` as the other stored columns. + fn typed_shape(column: Option<&str>, rest: &[(&str, Value)]) -> KvCounterShape { + KvCounterShape::Typed { + column: column.map(str::to_string), + template: row(rest), + } + } + + fn row(fields: &[(&str, Value)]) -> Vec { + let map: HashMap = fields + .iter() + .map(|(k, v)| ((*k).to_string(), v.clone())) + .collect(); + nodedb_types::value_to_msgpack(&Value::Object(map)).expect("encode row") + } + + fn columns(bytes: &[u8]) -> HashMap { + typed_row(bytes) + .expect("a typed row decodes") + .expect("a typed row stays a typed row") + } + + #[test] + fn incr_on_a_one_column_typed_row_keeps_the_row() { + let current = row(&[("n", Value::Integer(5))]); + let (new_i64, bytes) = incr(Some(¤t), 3, &RAW).expect("incr"); + assert_eq!(new_i64, 8); + assert_eq!(columns(&bytes).get("n"), Some(&Value::Integer(8))); + } + + #[test] + fn incr_moves_the_first_numeric_column_in_key_order() { + let current = row(&[ + ("b", Value::Integer(100)), + ("a", Value::Integer(1)), + ("label", Value::String("x".into())), + ]); + let (new_i64, bytes) = incr(Some(¤t), 1, &RAW).expect("incr"); + assert_eq!(new_i64, 2); + let cols = columns(&bytes); + assert_eq!(cols.get("a"), Some(&Value::Integer(2))); + assert_eq!(cols.get("b"), Some(&Value::Integer(100))); + assert_eq!(cols.get("label"), Some(&Value::String("x".into()))); + } + + #[test] + fn incr_on_a_typed_row_encodes_the_same_bytes_every_time() { + let current = row(&[ + ("a", Value::Integer(1)), + ("b", Value::Integer(2)), + ("c", Value::Integer(3)), + ]); + let (_, first) = incr(Some(¤t), 1, &RAW).expect("incr"); + for _ in 0..16 { + let (_, again) = incr(Some(¤t), 1, &RAW).expect("incr"); + assert_eq!(again, first); + } + } + + #[test] + fn incr_on_a_typed_row_without_a_numeric_column_is_a_type_mismatch() { + let current = row(&[("label", Value::String("x".into()))]); + assert!(matches!( + incr(Some(¤t), 1, &RAW), + Err(AtomicComputeError::TypeMismatch { .. }) + )); + } + + #[test] + fn incr_on_a_raw_body_reads_and_writes_decimal_text() { + let (new_i64, bytes) = incr(Some(b"5"), 1, &RAW).expect("incr"); + assert_eq!(new_i64, 6); + assert_eq!(bytes, b"6".to_vec()); + + let (new_i64, bytes) = incr(Some(b"-10"), 3, &RAW).expect("incr"); + assert_eq!(new_i64, -7); + assert_eq!(bytes, b"-7".to_vec()); + + let (fresh, bytes) = incr(None, 4, &RAW).expect("incr"); + assert_eq!(fresh, 4); + assert_eq!(bytes, b"4".to_vec()); + } + + #[test] + fn incr_on_non_integer_raw_text_is_not_an_integer() { + for body in [ + b"abc".as_slice(), + b"", + b"1.5", + b"+5", + b"05", + b"-0", + b" 5", + b"5 ", + b"99999999999999999999", + ] { + assert!( + matches!( + incr(Some(body), 1, &RAW), + Err(AtomicComputeError::Counter(CounterFault::NotAnInteger)) + ), + "{:?}", + String::from_utf8_lossy(body) + ); + } + } + + #[test] + fn incr_past_the_i64_range_is_an_overflow() { + let max = i64::MAX.to_string(); + assert!(matches!( + incr(Some(max.as_bytes()), 1, &RAW), + Err(AtomicComputeError::Counter(CounterFault::IntegerOverflow)) + )); + let min = i64::MIN.to_string(); + assert!(matches!( + incr(Some(min.as_bytes()), -1, &RAW), + Err(AtomicComputeError::Counter(CounterFault::IntegerOverflow)) + )); + let (value, bytes) = incr(Some(min.as_bytes()), 0, &RAW).expect("i64::MIN parses"); + assert_eq!(value, i64::MIN); + assert_eq!(bytes, min.into_bytes()); + } + + #[test] + fn incr_float_on_a_raw_body_reads_and_writes_decimal_text() { + let (new_f64, bytes) = incr_float(Some(b"1.5"), "1", &RAW).expect("incr_float"); + assert_eq!(new_f64, 2.5); + assert_eq!(bytes, b"2.5".to_vec()); + + let (new_f64, bytes) = incr_float(Some(b"10.5"), "0.5", &RAW).expect("incr_float"); + assert_eq!(new_f64, 11.0); + assert_eq!(bytes, b"11".to_vec()); + + let (_, bytes) = incr_float(Some(b"5"), "0.25", &RAW).expect("incr_float"); + assert_eq!(bytes, b"5.25".to_vec()); + + for (stored, delta, expected) in [ + ("0.1", "0.2", "0.3"), + ("10.5", "0.1", "10.6"), + ("5.0e3", "200", "5200"), + ("3.0", "0", "3"), + ("-1.5", "1.5", "0"), + ("1", "0.12345678901234567891", "1.12345678901234567891"), + ] { + let (_, bytes) = incr_float(Some(stored.as_bytes()), delta, &RAW).expect("incr_float"); + assert_eq!(bytes, expected.as_bytes().to_vec(), "{stored} + {delta}"); + } + } + + #[test] + fn incr_float_on_non_numeric_raw_text_is_not_a_float() { + for body in [b"abc".as_slice(), b"", b"NaN", b" 1.5"] { + assert!( + matches!( + incr_float(Some(body), "1", &RAW), + Err(AtomicComputeError::Counter(CounterFault::NotAFloat)) + ), + "{:?}", + String::from_utf8_lossy(body) + ); + } + } + + #[test] + fn incr_float_to_infinity_is_non_finite() { + let max = f64::MAX.to_string(); + assert!(matches!( + incr_float(Some(max.as_bytes()), &max, &RAW), + Err(AtomicComputeError::Counter(CounterFault::NonFinite)) + )); + } + + #[test] + fn incr_float_on_a_one_column_typed_row_keeps_the_row() { + let current = row(&[("score", Value::Float(1.5))]); + let (new_f64, bytes) = incr_float(Some(¤t), "1", &RAW).expect("incr_float"); + assert_eq!(new_f64, 2.5); + assert_eq!(columns(&bytes).get("score"), Some(&Value::Float(2.5))); + } + + #[test] + fn incr_on_an_absent_key_under_a_typed_shape_creates_the_typed_row() { + let shape = typed_shape(Some("n"), &[("status", Value::String("new".into()))]); + let (value, bytes) = incr(None, 7, &shape).expect("incr"); + assert_eq!(value, 7); + let cols = columns(&bytes); + assert_eq!(cols.get("n"), Some(&Value::Integer(7))); + assert_eq!(cols.get("status"), Some(&Value::String("new".into()))); + } + + #[test] + fn incr_float_on_an_absent_key_under_a_typed_shape_creates_the_typed_row() { + let shape = typed_shape(Some("score"), &[]); + let (value, bytes) = incr_float(None, "2.5", &shape).expect("incr_float"); + assert_eq!(value, 2.5); + assert_eq!(columns(&bytes).get("score"), Some(&Value::Float(2.5))); + } + + #[test] + fn an_absent_key_under_a_typed_shape_without_a_column_is_a_type_mismatch() { + let shape = typed_shape(None, &[]); + assert!(matches!( + incr(None, 1, &shape), + Err(AtomicComputeError::TypeMismatch { .. }) + )); + assert!(matches!( + incr_float(None, "1", &shape), + Err(AtomicComputeError::TypeMismatch { .. }) + )); + } + + #[test] + fn cas_on_a_one_column_typed_row_swaps_the_column() { + let current = row(&[("state", Value::String("idle".into()))]); + let (matched, bytes) = cas(Some(¤t), b"idle", b"busy").expect("cas"); + assert!(matched); + assert_eq!( + columns(&bytes).get("state"), + Some(&Value::String("busy".into())) + ); + let (matched, _) = cas(Some(¤t), b"busy", b"idle").expect("cas"); + assert!(!matched); + } + + #[test] + fn getset_on_a_one_column_typed_row_swaps_the_column() { + let current = row(&[("token", Value::String("old".into()))]); + let bytes = getset(Some(¤t), b"new").expect("getset"); + assert_eq!( + columns(&bytes).get("token"), + Some(&Value::String("new".into())) + ); + assert_eq!(getset(None, b"raw").expect("getset"), b"raw".to_vec()); + } +} diff --git a/nodedb-physical/src/kv_atomic/counter_fault.rs b/nodedb-physical/src/kv_atomic/counter_fault.rs new file mode 100644 index 000000000..4b950127c --- /dev/null +++ b/nodedb-physical/src/kv_atomic/counter_fault.rs @@ -0,0 +1,73 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Why a KV counter atomic (`INCR`, `INCRBY`, `DECR`, `INCRBYFLOAT`) +//! computed no value. + +/// Why a KV counter atomic computed no value. +/// +/// Each fault has one client message, the text Redis answers with for the +/// same condition. RESP sends it after `ERR`. The SQL surfaces send it with +/// the SQLSTATE [`CounterFault::sqlstate`] names. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum CounterFault { + /// The stored value is not a decimal integer in the `i64` range. + NotAnInteger, + /// The stored value is not a decimal float. + NotAFloat, + /// The integer result is outside the `i64` range. + IntegerOverflow, + /// The float result is NaN or infinite. + NonFinite, +} + +impl CounterFault { + /// The client message, without the RESP `ERR` prefix. + pub fn message(self) -> &'static str { + match self { + Self::NotAnInteger => "value is not an integer or out of range", + Self::NotAFloat => "value is not a valid float", + Self::IntegerOverflow => "increment or decrement would overflow", + Self::NonFinite => "increment would produce NaN or Infinity", + } + } + + /// Whether the fault is a result out of range. A fault that is not is a + /// stored value that does not parse. + pub fn is_out_of_range(self) -> bool { + matches!(self, Self::IntegerOverflow | Self::NonFinite) + } + + /// The SQLSTATE for the fault: `22P02` for a stored value that does not + /// parse, `22003` for a result out of range. + pub fn sqlstate(self) -> &'static str { + use nodedb_types::error::sqlstate; + if self.is_out_of_range() { + sqlstate::NUMERIC_VALUE_OUT_OF_RANGE + } else { + sqlstate::INVALID_TEXT_REPRESENTATION + } + } +} + +impl std::fmt::Display for CounterFault { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(self.message()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn parse_faults_answer_invalid_text_representation() { + assert_eq!(CounterFault::NotAnInteger.sqlstate(), "22P02"); + assert_eq!(CounterFault::NotAFloat.sqlstate(), "22P02"); + } + + #[test] + fn range_faults_answer_numeric_value_out_of_range() { + assert_eq!(CounterFault::IntegerOverflow.sqlstate(), "22003"); + assert_eq!(CounterFault::NonFinite.sqlstate(), "22003"); + } +} diff --git a/nodedb-physical/src/kv_atomic/error.rs b/nodedb-physical/src/kv_atomic/error.rs new file mode 100644 index 000000000..60d66abf2 --- /dev/null +++ b/nodedb-physical/src/kv_atomic/error.rs @@ -0,0 +1,23 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Why a KV atomic computed no stored value. + +use super::counter_fault::CounterFault; + +/// Why [`super::compute`] computed no stored value for a KV atomic. +/// +/// Each executor maps it into its own error. Origin maps it into the Data +/// Plane `ErrorCode`, and Lite maps it into `LiteError`. +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +pub enum AtomicComputeError { + /// A typed row has no column of the type the atomic reads. + #[error("{detail}")] + TypeMismatch { detail: String }, + /// A counter atomic read a stored value it cannot parse as a number, or + /// computed a result out of range. + #[error("{0}")] + Counter(CounterFault), + /// The computed new value failed to re-encode as MessagePack. + #[error("{detail}")] + Encode { detail: String }, +} diff --git a/nodedb-physical/src/kv_atomic/float_text.rs b/nodedb-physical/src/kv_atomic/float_text.rs new file mode 100644 index 000000000..4592122d5 --- /dev/null +++ b/nodedb-physical/src/kv_atomic/float_text.rs @@ -0,0 +1,224 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! `INCRBYFLOAT` on a raw KV body: decimal text in, decimal text out. +//! +//! Redis adds in `long double` and prints the sum with 17 fractional digits, +//! trailing zeros trimmed, so `"0.1"` plus `0.2` stores `"0.3"`. Rust has no +//! `long double`. Exact decimal addition gives the same text for every sum +//! that fits a [`Decimal`]: 28 significant digits, magnitude below 7.9e28. +//! An operand or a sum outside that range is added in `f64` instead. + +use std::str::FromStr; + +use rust_decimal::Decimal; +use rust_decimal::prelude::ToPrimitive; + +use super::counter_fault::CounterFault; +use super::error::AtomicComputeError; + +/// Add `delta` to the raw body `stored`. Returns the new value and the text +/// to store. +/// +/// `stored` and `delta` must both be decimal numbers (see +/// [`is_decimal_number`]). Anything else is `Counter(NotAFloat)`. A sum that +/// is not finite is `Counter(NonFinite)`. +pub(super) fn add(stored: &[u8], delta: &str) -> Result<(f64, Vec), AtomicComputeError> { + let text = std::str::from_utf8(stored) + .ok() + .filter(|text| is_decimal_number(text)) + .ok_or(AtomicComputeError::Counter(CounterFault::NotAFloat))?; + let delta_f64 = delta_to_f64(delta)?; + if let (Some(base), Some(step)) = (parse_decimal(text), parse_decimal(delta)) + && let Some(sum) = base.checked_add(step) + { + let sum = sum.normalize(); + let value = sum + .to_f64() + .ok_or(AtomicComputeError::Counter(CounterFault::NonFinite))?; + return Ok((value, sum.to_string().into_bytes())); + } + let base: f64 = text + .parse() + .map_err(|_| AtomicComputeError::Counter(CounterFault::NotAFloat))?; + let value = base + delta_f64; + if !value.is_finite() { + return Err(AtomicComputeError::Counter(CounterFault::NonFinite)); + } + Ok((value, float_text(value))) +} + +/// The text for a fresh float counter: `0` plus `delta`. +pub(super) fn fresh(delta: &str) -> Result<(f64, Vec), AtomicComputeError> { + add(b"0", delta) +} + +/// `delta` as a finite `f64`, for a typed column. A delta that is not a +/// decimal number is `Counter(NotAFloat)`. One outside the `f64` range is +/// `Counter(NonFinite)`. +pub fn delta_to_f64(delta: &str) -> Result { + if !is_decimal_number(delta) { + return Err(AtomicComputeError::Counter(CounterFault::NotAFloat)); + } + let value: f64 = delta + .parse() + .map_err(|_| AtomicComputeError::Counter(CounterFault::NotAFloat))?; + if value.is_finite() { + Ok(value) + } else { + Err(AtomicComputeError::Counter(CounterFault::NonFinite)) + } +} + +/// The exact value of `text`, or `None` when it does not fit a [`Decimal`]. +fn parse_decimal(text: &str) -> Option { + if text.contains(['e', 'E']) { + Decimal::from_scientific(text).ok() + } else { + Decimal::from_str(text).ok() + } +} + +/// The text for an `f64` sum outside the [`Decimal`] range: plain decimal +/// digits with no exponent, the form Redis prints. +fn float_text(value: f64) -> Vec { + let text = value.to_string(); + if text == "-0" { + b"0".to_vec() + } else { + text.into_bytes() + } +} + +/// The number grammar `INCRBYFLOAT` accepts, for a stored body and for the +/// client's increment: an optional sign, then digits with an optional point +/// (at least one digit), then an optional exponent `e` or `E` with an +/// optional sign and at least one digit. No whitespace, digit separators, +/// `inf`, or `nan`. +pub fn is_decimal_number(text: &str) -> bool { + let bytes = text.as_bytes(); + let mut i = 0; + if matches!(bytes.first(), Some(b'+' | b'-')) { + i += 1; + } + let int_digits = count_digits(&bytes[i..]); + i += int_digits; + let mut frac_digits = 0; + if bytes.get(i) == Some(&b'.') { + i += 1; + frac_digits = count_digits(&bytes[i..]); + i += frac_digits; + } + if int_digits + frac_digits == 0 { + return false; + } + if matches!(bytes.get(i), Some(b'e' | b'E')) { + i += 1; + if matches!(bytes.get(i), Some(b'+' | b'-')) { + i += 1; + } + let exp_digits = count_digits(&bytes[i..]); + if exp_digits == 0 { + return false; + } + i += exp_digits; + } + i == bytes.len() +} + +fn count_digits(bytes: &[u8]) -> usize { + bytes.iter().take_while(|b| b.is_ascii_digit()).count() +} + +#[cfg(test)] +mod tests { + use super::*; + + fn text_of(stored: &str, delta: &str) -> String { + let (_, bytes) = add(stored.as_bytes(), delta).expect("add"); + String::from_utf8(bytes).expect("UTF-8") + } + + #[test] + fn decimal_text_adds_exactly_like_redis() { + assert_eq!(text_of("0.1", "0.2"), "0.3"); + assert_eq!(text_of("10.5", "0.1"), "10.6"); + assert_eq!(text_of("5.0e3", "200"), "5200"); + assert_eq!(text_of("3.0", "0"), "3"); + assert_eq!(text_of("-1.5", "1.5"), "0"); + assert_eq!(text_of("1.5", "1"), "2.5"); + assert_eq!(text_of("+2", "-0.5"), "1.5"); + assert_eq!(text_of("1E-2", "0"), "0.01"); + assert_eq!(text_of("1", "1e1"), "11"); + } + + #[test] + fn a_twenty_digit_delta_adds_exactly() { + assert_eq!( + text_of("1", "0.12345678901234567891"), + "1.12345678901234567891" + ); + assert_eq!(text_of("10000000000000000000", "1"), "10000000000000000001"); + } + + #[test] + fn the_returned_value_matches_the_stored_text() { + let (value, bytes) = add(b"0.1", "0.2").expect("add"); + assert_eq!(value, 0.3); + assert_eq!(bytes, b"0.3".to_vec()); + } + + #[test] + fn a_fresh_counter_stores_the_delta_text() { + assert_eq!(fresh("2.5").expect("fresh").1, b"2.5".to_vec()); + assert_eq!(fresh("0").expect("fresh").1, b"0".to_vec()); + assert_eq!(fresh("-0.0").expect("fresh").1, b"0".to_vec()); + } + + #[test] + fn a_sum_outside_the_decimal_range_adds_in_f64() { + let (value, bytes) = add(b"1e300", "1").expect("add"); + assert_eq!(value, 1e300); + assert_eq!(bytes, 1e300f64.to_string().into_bytes()); + } + + #[test] + fn text_that_is_not_a_number_is_not_a_float() { + for stored in [ + "abc", "", "NaN", "inf", " 1.5", "1.5 ", "1_000", ".", "1e", "e5", "0x10", + ] { + assert!( + matches!( + add(stored.as_bytes(), "1"), + Err(AtomicComputeError::Counter(CounterFault::NotAFloat)) + ), + "{stored:?}" + ); + } + } + + #[test] + fn a_non_finite_sum_is_refused() { + let max = f64::MAX.to_string(); + assert!(matches!( + add(max.as_bytes(), &max), + Err(AtomicComputeError::Counter(CounterFault::NonFinite)) + )); + assert!(matches!( + add(b"1", "1e400"), + Err(AtomicComputeError::Counter(CounterFault::NonFinite)) + )); + } + + #[test] + fn a_delta_that_is_not_a_number_is_not_a_float() { + for delta in ["abc", "", "inf", "NaN", " 1"] { + assert!( + matches!( + add(b"1", delta), + Err(AtomicComputeError::Counter(CounterFault::NotAFloat)) + ), + "{delta:?}" + ); + } + } +} diff --git a/nodedb-physical/src/kv_atomic/mod.rs b/nodedb-physical/src/kv_atomic/mod.rs new file mode 100644 index 000000000..dc4968b07 --- /dev/null +++ b/nodedb-physical/src/kv_atomic/mod.rs @@ -0,0 +1,11 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Value semantics of the KV atomics, shared by every executor. + +pub mod compute; +pub mod counter_fault; +pub mod error; +pub mod float_text; + +pub use counter_fault::CounterFault; +pub use error::AtomicComputeError; diff --git a/nodedb-physical/src/lib.rs b/nodedb-physical/src/lib.rs index 9b7489305..301925bb9 100644 --- a/nodedb-physical/src/lib.rs +++ b/nodedb-physical/src/lib.rs @@ -11,6 +11,7 @@ pub mod convert_context; pub mod error; +pub mod kv_atomic; pub mod physical_plan; pub mod physical_task; pub mod surrogate; diff --git a/nodedb-physical/src/physical_plan/cluster_event.rs b/nodedb-physical/src/physical_plan/cluster_event.rs index 7052c29cd..2407d4849 100644 --- a/nodedb-physical/src/physical_plan/cluster_event.rs +++ b/nodedb-physical/src/physical_plan/cluster_event.rs @@ -44,6 +44,11 @@ pub enum ClusterEventOp { topic_name: String, payload: String, }, + /// Read the receiving node's tenant write marks of `group_ids`, once it + /// applied every entry the groups committed before the request. RESTORE's + /// staleness guard asks a replica of each group this way when the + /// restoring node does not replicate the group. + TenantWriteMarks { tenant_id: u64, group_ids: Vec }, } #[cfg(test)] diff --git a/nodedb-physical/src/physical_plan/collection.rs b/nodedb-physical/src/physical_plan/collection.rs index 5ab0855c4..b031081d0 100644 --- a/nodedb-physical/src/physical_plan/collection.rs +++ b/nodedb-physical/src/physical_plan/collection.rs @@ -154,4 +154,68 @@ impl PhysicalPlan { PhysicalPlan::ClusterArray(op) => Some(op.array_id().name.as_str()), } } + + /// Every user collection this plan names: each collection a committed + /// redo install writes, or else the one [`Self::collection`] reports. + /// + /// A committed-redo apply and a Calvin flush install one record that can + /// write several collections, so [`Self::collection`] reports none for + /// them. A caller that keys on a collection name uses this instead. + pub fn named_collections(&self) -> Vec<&str> { + if let PhysicalPlan::Meta( + MetaOp::ApplyTransactionRedo { collections, .. } + | MetaOp::CalvinFlush { collections, .. }, + ) = self + { + collections.iter().map(String::as_str).collect() + } else { + self.collection().into_iter().collect() + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_types::{DatabaseId, QualifiedCollection}; + + use crate::physical_plan::KvOp; + + #[test] + fn a_redo_install_names_every_collection_it_writes() { + let collections = vec!["a".to_string(), "b".to_string()]; + let redo = PhysicalPlan::Meta(MetaOp::ApplyTransactionRedo { + redo: Vec::new(), + collections: collections.clone(), + sum_targets: Vec::new(), + origin: crate::physical_plan::RedoOrigin::Commit, + }); + let flush = PhysicalPlan::Meta(MetaOp::CalvinFlush { + epoch: 1, + position: 0, + redo: Vec::new(), + collections, + sum_targets: Vec::new(), + }); + for plan in [redo, flush] { + assert_eq!(plan.collection(), None); + assert_eq!(plan.named_collections(), vec!["a", "b"]); + } + } + + #[test] + fn a_single_collection_plan_names_its_collection() { + let get = PhysicalPlan::Kv(KvOp::Get { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "users"), + key: Vec::new(), + rls_filters: Vec::new(), + surrogate_ceiling: None, + }); + assert_eq!(get.named_collections(), vec!["users"]); + assert!( + PhysicalPlan::Meta(MetaOp::Checkpoint) + .named_collections() + .is_empty() + ); + } } diff --git a/nodedb-physical/src/physical_plan/document/mod.rs b/nodedb-physical/src/physical_plan/document/mod.rs index 6a1b2bbd2..9fb7cc651 100644 --- a/nodedb-physical/src/physical_plan/document/mod.rs +++ b/nodedb-physical/src/physical_plan/document/mod.rs @@ -19,7 +19,7 @@ pub use merge_types::{MergeActionOp, MergeClauseKind as MergeClauseKindOp, Merge pub use ollp_edge::OllpPredictedEdge; pub use op::DocumentOp; pub use resolved_mutation::{DocumentResolveOutcome, DocumentResolvedMutation}; -pub use sum_target::{ResolvedSumTarget, SumTargetKey, resolved_sum_surrogate}; +pub use sum_target::{RedoSumTargets, ResolvedSumTarget, SumTargetKey, resolved_sum_surrogate}; pub use timeseries_schema::TimeseriesSchema; pub use types::{ BalancedDef, EnforcementOptions, GeneratedColumnSpec, MaterializedSumBinding, PeriodLockConfig, diff --git a/nodedb-physical/src/physical_plan/document/sum_target.rs b/nodedb-physical/src/physical_plan/document/sum_target.rs index ad9b683a5..bb619a42d 100644 --- a/nodedb-physical/src/physical_plan/document/sum_target.rs +++ b/nodedb-physical/src/physical_plan/document/sum_target.rs @@ -61,6 +61,7 @@ impl SumTargetKey { Debug, Clone, PartialEq, + Eq, serde::Serialize, serde::Deserialize, zerompk::ToMessagePack, @@ -122,6 +123,32 @@ impl ResolvedSumTarget { } } +/// The materialized-sum resolution one committed transaction's writes to one +/// SOURCE collection fold into their targets. +/// +/// A transaction redo record carries source post-images only. Every replica +/// applying it folds the source rows into their target rows, so the target +/// identities travel with the redo, keyed by source collection. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct RedoSumTargets { + /// SOURCE collection the transaction wrote. + pub collection: String, + /// Every target row the transaction's writes to `collection` resolved. + pub resolved: Vec, + /// TARGET collections whose delta travels on its own `ApplyBalanceDelta` + /// task, so the fold skips them. + pub deferred: Vec, +} + /// The surrogate `resolved` binds `target_collection`'s `join_value` to. /// /// The one lookup both planes use, so the Control Plane's "this one travels on diff --git a/nodedb-physical/src/physical_plan/kv/collection.rs b/nodedb-physical/src/physical_plan/kv/collection.rs index c309af459..2469250e8 100644 --- a/nodedb-physical/src/physical_plan/kv/collection.rs +++ b/nodedb-physical/src/physical_plan/kv/collection.rs @@ -8,8 +8,8 @@ use super::op::KvOp; impl KvOp { - /// The user collection this op targets, if any. Sorted-index ops (keyed - /// only by index name) and `ResolvedWrite` (mutations may span two + /// The user collection this op targets, if any. Sorted-index ops keyed + /// only by index name and `ResolvedWrite` (mutations may span two /// collections) return `None`. `TransferItem` reports its source. pub fn collection(&self) -> Option<&str> { match self { @@ -38,7 +38,8 @@ impl KvOp { | KvOp::RegisterSortedIndex { collection, .. } | KvOp::PredicateUpdate { collection, .. } | KvOp::PredicateDelete { collection, .. } - | KvOp::MaterializeScan { collection, .. } => Some(collection.as_str()), + | KvOp::MaterializeScan { collection, .. } + | KvOp::SortedIndexTxnRead { collection, .. } => Some(collection.as_str()), KvOp::TransferItem { source_collection, .. } => Some(source_collection.as_str()), diff --git a/nodedb-physical/src/physical_plan/kv/counter_shape.rs b/nodedb-physical/src/physical_plan/kv/counter_shape.rs new file mode 100644 index 000000000..cb06d641e --- /dev/null +++ b/nodedb-physical/src/physical_plan/kv/counter_shape.rs @@ -0,0 +1,39 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The row a KV counter atomic creates when its key is absent. + +/// What `INCR` / `INCRBYFLOAT` stores for a key that does not exist yet. +/// +/// The Data Plane does not know a collection's declared columns. The Control +/// Plane decides the shape from the catalog when it builds the op, and the +/// shape travels with the op to every path that computes the value: the live +/// handler, transaction staging, the resolve path, and WAL replay. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + Default, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum KvCounterShape { + /// A raw collection (a single `value` column, or none declared), or a + /// RESP command, where every value is a byte string. The new value is + /// stored as its decimal text. + #[default] + Raw, + /// A typed collection. The new row is `template` with `column` set to the + /// new value: the row `INSERT (key, column) VALUES (key, delta)` stores, + /// with DEFAULTs materialized. + Typed { + /// The declared numeric column the counter moves. `None` when the + /// collection declares no column of the counter's type. + column: Option, + /// A msgpack map body: the columns the insert stores other than + /// `column`. + template: Vec, + }, +} diff --git a/nodedb-physical/src/physical_plan/kv/mod.rs b/nodedb-physical/src/physical_plan/kv/mod.rs index bd8c12cb9..bbbdc278a 100644 --- a/nodedb-physical/src/physical_plan/kv/mod.rs +++ b/nodedb-physical/src/physical_plan/kv/mod.rs @@ -3,8 +3,12 @@ //! KV engine operations dispatched to the Data Plane. pub mod collection; +pub mod counter_shape; pub mod op; pub mod resolved_mutation; +pub mod sorted_read; +pub use counter_shape::KvCounterShape; pub use op::KvOp; pub use resolved_mutation::{KvResolveOutcome, KvResolvedMutation}; +pub use sorted_read::{SortedIndexRead, SortedIndexSpec}; diff --git a/nodedb-physical/src/physical_plan/kv/op.rs b/nodedb-physical/src/physical_plan/kv/op.rs index 1ccd43fee..9c392caef 100644 --- a/nodedb-physical/src/physical_plan/kv/op.rs +++ b/nodedb-physical/src/physical_plan/kv/op.rs @@ -4,7 +4,9 @@ use nodedb_types::{QualifiedCollection, RlsWriteCheck, Surrogate}; +use super::counter_shape::KvCounterShape; use super::resolved_mutation::KvResolvedMutation; +use super::sorted_read::{SortedIndexRead, SortedIndexSpec}; use crate::physical_plan::document::ReturningSpec; /// KV engine physical operations. @@ -59,6 +61,10 @@ pub enum KvOp { /// principal would show. #[serde(default)] rls_filters: Vec, + /// Sync provenance of a Lite KV push. `Some` puts the write behind + /// the sync idempotency gate. `None` for every other write. + #[serde(default)] + provenance: Option, }, /// SQL `INSERT` semantics: write only if the key does not already exist. @@ -139,6 +145,9 @@ pub enum KvOp { /// See `Put::rls_filters`. #[serde(default)] rls_filters: Vec, + /// See `Put::provenance`. + #[serde(default)] + provenance: Option, }, /// Cursor-based scan with optional filter predicate. @@ -305,8 +314,10 @@ pub enum KvOp { restart_identity: bool, }, - /// Atomic increment: init 0 if absent, `TypeMismatch` if not i64, - /// `OverflowError` on wrap. `ttl_ms > 0` sets/resets TTL; `0` preserves it. + /// Atomic increment: init 0 if absent. A raw body is decimal text in and + /// out, and a body that is not a decimal i64 is a counter fault. A typed + /// row moves its first integer column. Overflow is a counter fault, never + /// a wrap. `ttl_ms > 0` sets/resets TTL; `0` preserves it. Incr { collection: QualifiedCollection, key: Vec, @@ -318,21 +329,28 @@ pub enum KvOp { /// Write policy evaluated against the computed post-increment image /// inside the engine, not guessed by the handler. rls_write_check: RlsWriteCheck, + /// The row an absent key becomes. + shape: KvCounterShape, }, /// Atomic float increment on a numeric value. Returns new value. /// - /// Same semantics as `Incr` but for f64 values. - /// If value is not f64, returns `TypeMismatch`. + /// A raw body is decimal text in and out, added exactly. A typed row's + /// column adds in `f64`. A NaN or infinite result is a counter fault. IncrFloat { collection: QualifiedCollection, key: Vec, - delta: f64, + /// The increment as the client's decimal text, checked at the + /// protocol boundary. It is parsed once, where it is added, so no + /// digit is lost to an `f64` on the way. + delta: String, /// Stable cross-engine identity. `Surrogate::ZERO` only in tests. surrogate: Surrogate, /// Compiled row-level-security WRITE predicate — see `Incr`, whose /// engine-internal compute-and-persist this mirrors. rls_write_check: RlsWriteCheck, + /// The row an absent key becomes. + shape: KvCounterShape, }, /// Compare-and-swap: set value to `new_value` only if current equals `expected`. @@ -465,6 +483,22 @@ pub enum KvOp { primary_key: Vec, }, + /// A sorted-index read inside an explicit transaction block. + /// + /// The Data Plane answers it from a transaction-local tree. It builds the + /// tree from the collection's base rows with the transaction's staged + /// writes folded in, so the read sees the transaction's own writes and an + /// index the transaction created. Routed by `collection`, the core that + /// holds the rows. + SortedIndexTxnRead { + collection: QualifiedCollection, + index_name: String, + /// The definition of an index this transaction created and has not + /// committed. `None` reads the definition registered on the core. + pending: Option, + read: SortedIndexRead, + }, + /// Cursor-paginated raw scan for the clone materializer. /// /// Unlike `Scan`, this returns raw `(key, value)` byte pairs **plus** the diff --git a/nodedb-physical/src/physical_plan/kv/sorted_read.rs b/nodedb-physical/src/physical_plan/kv/sorted_read.rs new file mode 100644 index 000000000..6bfb5b31c --- /dev/null +++ b/nodedb-physical/src/physical_plan/kv/sorted_read.rs @@ -0,0 +1,69 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The payload of a sorted-index read inside an explicit transaction. +//! +//! A transaction must see its own DDL and its own writes. A sorted index that +//! the transaction created has no tree before COMMIT, and a committed index's +//! tree holds none of the transaction's staged writes. The Data Plane +//! therefore answers such a read from a transaction-local tree. It builds that +//! tree from the collection's base rows with the transaction's staged writes +//! folded in. + +/// The definition `CREATE SORTED INDEX` registers, carried for an index the +/// transaction created and has not committed. +/// +/// The fields match `KvOp::RegisterSortedIndex`, so the Data Plane builds the +/// transaction-local definition with the same code that builds a registered one. +#[derive( + Debug, + Clone, + PartialEq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct SortedIndexSpec { + /// Sort columns: (column_name, direction "ASC"/"DESC"). + pub sort_columns: Vec<(String, String)>, + /// Primary key column name. + pub key_column: String, + /// Window type: "none", "daily", "weekly", "monthly", or "custom". + pub window_type: String, + /// Window timestamp column (empty if window_type == "none"). + pub window_timestamp_column: String, + /// Custom window start (ms since epoch, 0 if N/A). + pub window_start_ms: u64, + /// Custom window end (ms since epoch, 0 if N/A). + pub window_end_ms: u64, +} + +/// Which sorted-index read a transaction runs. +/// +/// Each arm answers exactly like its autocommit counterpart: +/// `SortedIndexRank`, `SortedIndexTopK`, `SortedIndexRange`, +/// `SortedIndexCount` and `SortedIndexScore`. +#[derive( + Debug, + Clone, + PartialEq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum SortedIndexRead { + /// The 1-based rank of one key. + Rank { primary_key: Vec }, + /// The top `k` entries. + TopK { k: u32 }, + /// The entries whose leading sort column lies in the score range. + Range { + score_min: Option>, + score_max: Option>, + }, + /// The number of entries. + Count, + /// The sort key of one key (ZSCORE equivalent). + Score { primary_key: Vec }, +} diff --git a/nodedb-physical/src/physical_plan/meta.rs b/nodedb-physical/src/physical_plan/meta.rs index 265f23c4a..9a2bec498 100644 --- a/nodedb-physical/src/physical_plan/meta.rs +++ b/nodedb-physical/src/physical_plan/meta.rs @@ -36,15 +36,15 @@ pub enum MetaOp { target_request_id: nodedb_types::id::RequestId, }, - /// Atomic transaction batch: execute all sub-plans atomically. + /// A transaction's plans as one batch. An embedded (Lite) engine executes + /// the sub-plans atomically. An Origin core refuses it: a committed + /// transaction installs there only through its redo record + /// (`ApplyTransactionRedo`, `CalvinFlush`). /// /// `txn_id` identifies the committing session transaction whose staging - /// overlay holds the resolve-time bitemporal stamps this install must reuse - /// (so a `bitemporal=true` document put lands on the same version key the - /// redo carries, not a fresh one). `None` for install paths with no session - /// overlay to consult — Calvin (which threads its stamps in directly) and - /// procedural/test callers. Wire-additive: defaults to `None` on decode of - /// older entries. + /// overlay holds the resolve-time bitemporal stamps an install must reuse. + /// `None` for callers with no session overlay to consult. Wire-additive: + /// defaults to `None` on decode of older entries. TransactionBatch { plans: Vec, #[serde(default)] @@ -93,7 +93,16 @@ pub enum MetaOp { /// Snapshot a tenant's data from the sparse engine. /// Returns serialized `(documents, indexes)` as JSON payload. - CreateTenantSnapshot { tenant_id: u64 }, + CreateTenantSnapshot { + tenant_id: u64, + /// `Some(W)` asks the node that receives the plan to take a backup's + /// consistent cut at watermark `W` before it snapshots: every write + /// committed below `W` has its final outcome there first. The + /// receiving Control Plane takes the cut and clears the field. The + /// Data Plane never reads it. + #[serde(default)] + cut_watermark: Option, + }, /// Restore a tenant's data across all engines from a snapshot. /// `snapshot` is a MessagePack-serialized `TenantDataSnapshot`. @@ -111,10 +120,13 @@ pub enum MetaOp { /// for a lagging follower). Empty = legacy install-over-present behavior. #[serde(default)] clear_vshards: Vec, - /// (tenant_id, collection) pairs to clear before install — pre-resolved by the - /// applier from the local catalog for the cleared vShards. Empty = no clear. + /// `(database_id, tenant_id, collection)` triples to clear before + /// install — pre-resolved by the applier from the local catalog for the + /// cleared vShards. `collection` is the name the Data Plane stores the + /// collection under: database-qualified outside the default database. + /// Empty = no clear. #[serde(default)] - collections_to_clear: Vec<(u64, String)>, + collections_to_clear: Vec<(u64, u64, String)>, }, /// Purge ALL data for a tenant across every engine and cache. @@ -292,9 +304,10 @@ pub enum MetaOp { /// /// The Calvin scheduler dispatches this variant after lock acquisition for /// transactions whose read/write set is fully known at submission time (the - /// common case). The Data Plane handler executes `plans` atomically (same - /// semantics as `TransactionBatch`) and the scheduler writes a - /// `WalRecord::CalvinApplied` after a successful response. + /// common case). The Data Plane handler validates the read-set and stages + /// `plans` without mutating base. Once the global verdict is commit, the + /// scheduler resolves the staged plans into a redo record, appends it, and + /// flushes it (`CalvinResolve`, `CalvinFlush`). /// /// NOTE: This variant occupies the same msgpack positional tag as the /// original `CalvinExecute` variant it replaces, preserving wire @@ -386,22 +399,23 @@ pub enum MetaOp { is_group_leader: bool, }, - /// Rebuild all indexes (HNSW, FTS LSM, graph CSR) for a collection - /// on this core in a shadow-build + atomic-swap manner. + /// Rebuild a collection's indexes (HNSW, full-text, graph CSR) on + /// this core. /// - /// When `concurrent = true`, the build proceeds without blocking query - /// handling: a background OS thread performs the rebuild and the Data - /// Plane polls for completion on subsequent ticks, only swapping the - /// live index in at cutover. When `concurrent = false` the rebuild - /// runs inline (same semantics as the legacy Checkpoint path). + /// `index_name` narrows the rebuild to one kind: `hnsw`, `fts` or + /// `csr`. `None` rebuilds every kind the collection has. /// - /// `index_name` narrows the rebuild to a single named index when set; - /// `None` rebuilds all index types for the collection. + /// Each rebuild runs off the core while the core keeps serving reads + /// and writes. Writes made during the build are replayed onto the + /// rebuilt index at cutover, and the core swaps it in on a later tick. + /// With `concurrent = true` the core answers once the rebuilds have + /// started. With `concurrent = false` it answers once they have cut + /// over, or `DeadlineExceeded` at the request deadline; the rebuilds + /// still complete. /// - /// Returns `Response::Ok` on successful cutover, or a typed error if: - /// - another rebuild is already in progress for this collection - /// (`ErrorCode::Conflict`), or - /// - the shadow build fails (`ErrorCode::Internal`). + /// Errors when a rebuild of the collection already runs + /// (`ObjectNotInPrerequisiteState`), when a rebuild cannot start, or, + /// with `concurrent = false`, when a rebuild is discarded. RebuildIndex { collection: QualifiedCollection, index_name: Option, @@ -450,9 +464,9 @@ pub enum MetaOp { /// (unique / primary-key) immediately, computes the real affected-row /// count, and records the resulting body (or tombstone) in the overlay so /// a subsequent same-transaction read-modify-write observes it. It does - /// NOT make the write durable — the buffered plan is still replayed - /// through the real apply path inside the COMMIT `TransactionBatch`, which - /// remains the sole durable apply. Keyed by the request's `txn_id`. + /// NOT make the write durable — COMMIT resolves the overlay into the + /// transaction's redo record, and the redo install remains the sole + /// durable apply. Keyed by the request's `txn_id`. StageWrite { plan: Box }, /// Drop the per-transaction staging overlay for a completed (committed @@ -493,40 +507,42 @@ pub enum MetaOp { array_marker: u64, }, - /// Record the per-key / per-collection write versions of a committed - /// Calvin transaction's locally-applied write plans. + /// Record the per-key write versions of a committed Calvin transaction's + /// locally-applied write plans. /// - /// A Calvin apply's committed WAL LSN is known only after the apply - /// succeeds, so the apply itself cannot advance the version index. The - /// scheduler stamps that LSN onto this op's `wal_lsn` and dispatches it back - /// to the same core, which funnels `plans` through the shared write-version - /// recorder at that LSN — landing in the same shard-local WAL-LSN space the + /// The scheduler stamps the transaction's committed LSN onto this op's + /// `wal_lsn` and dispatches it back to the same core once the flush + /// completes. The core funnels `plans` through the shared write-version + /// recorder at that LSN — the same shard-local WAL-LSN space the /// single-shard fast path and read watermarks use. Records only: no base - /// mutation, no WAL append, no event emission. Wire-additive (appended last) - /// so older log entries decode unchanged. + /// mutation, no WAL append, no event emission. RecordCalvinWriteVersions { /// Tenant scope for all plans. tenant_id: TenantId, /// The locally-applied write plans whose keys' versions are recorded. plans: Vec, - /// Calvin epoch of the applied transaction. With `position` and the - /// request's vShard, keys the index-value tuples the flush staged so the - /// core drains and records them at this op's applied LSN. - epoch: u64, - /// Calvin position within the epoch (see `epoch`). - position: u32, }, - /// Flush the staged writes of a Calvin transaction to base storage. + /// Install a committed Calvin transaction's redo record on base storage. + /// + /// `CalvinExecuteStatic` STAGES the transaction's plans without mutating + /// base, and `CalvinResolve` resolves them into one redo record, which the + /// scheduler appends to the WAL as a `TransactionRedo` record. This op + /// carries that record's bytes, and the request carries its LSN. The core + /// installs it through the same passes restart replay drives: validate, + /// install with undo, then settle and cover. It then drops the staged + /// state keyed by `(epoch, position)`. /// - /// `CalvinExecuteStatic` validates and STAGES the transaction's plans into - /// the per-core commit-pending buffer without mutating base. Once the local - /// commit vote resolves to commit, the scheduler dispatches this op back to - /// the same core, which pops the staged plans keyed by `(epoch, position)` - /// and replays them through the durable apply funnel (base + side effects + - /// version recording). Absent key (already flushed/dropped) is an idempotent - /// no-op, not an error. - CalvinFlush { epoch: u64, position: u32 }, + /// `redo` is empty when the transaction wrote nothing. `collections` names + /// every collection the transaction wrote. `sum_targets` is the + /// materialized-sum resolution its document writes fold into. + CalvinFlush { + epoch: u64, + position: u32, + redo: Vec, + collections: Vec, + sum_targets: Vec, + }, /// Discard the staged writes of a Calvin transaction. /// @@ -567,10 +583,27 @@ pub enum MetaOp { /// `commit_pending` under `(epoch, position, vshard)` and the per-core /// staging overlay written under the corresponding synthetic `TxnId` /// (see `calvin_synthetic_txn_id`). Dispatched by the scheduler once the - /// local commit vote resolves to commit, in place of (or ahead of) - /// `CalvinFlush` — the flush path mutates base directly, while resolve - /// produces a durable redo record for a later install phase instead. No - /// base engine is touched during resolve. Wire-additive: appended last - /// so older log entries decode unchanged. + /// global verdict is commit, ahead of `CalvinFlush`, which installs the + /// record this op returns. No base engine is touched during resolve. CalvinResolve { epoch: u64, position: u32 }, + + /// Apply one committed transaction's resolved redo record on the core that + /// owns its vShard. + /// + /// Every replica runs this from the vShard's data-group Raft log, in log + /// order, and installs the same post-images. The write funnel appends + /// `redo` to this node's WAL as one `TransactionRedo` record before the + /// dispatch, so restart replay reproduces the apply. + /// + /// `redo` is the zerompk-encoded redo record. `collections` names every + /// collection the transaction wrote; each gets a collection-floor write + /// version at the record's LSN. `sum_targets` is the materialized-sum + /// resolution the transaction's document writes fold into their targets. + /// `origin` decides which commit-boundary checks the apply runs. + ApplyTransactionRedo { + redo: Vec, + collections: Vec, + sum_targets: Vec, + origin: super::RedoOrigin, + }, } diff --git a/nodedb-physical/src/physical_plan/mod.rs b/nodedb-physical/src/physical_plan/mod.rs index 04e8a3ce3..2f00b5eb8 100644 --- a/nodedb-physical/src/physical_plan/mod.rs +++ b/nodedb-physical/src/physical_plan/mod.rs @@ -20,6 +20,7 @@ pub mod meta; pub mod meta_calvin; pub mod plan; pub mod query; +pub mod redo_origin; pub mod rls_write_check_accessor; pub mod routing; pub mod set_op; @@ -40,18 +41,21 @@ pub use crdt::{CrdtOp, CrdtWriteVerb}; pub use document::{ BalancedDef, DocumentOp, DocumentResolveOutcome, DocumentResolvedMutation, EnforcementOptions, GeneratedColumnSpec, MaterializedSumBinding, OllpPredictedEdge, PeriodLockConfig, - RegisteredIndex, RegisteredIndexState, ResolvedSumTarget, ReturningColumns, ReturningItem, - ReturningSpec, StorageMode, SumTargetKey, TimeseriesSchema, UpdateValue, + RedoSumTargets, RegisteredIndex, RegisteredIndexState, ResolvedSumTarget, ReturningColumns, + ReturningItem, ReturningSpec, StorageMode, SumTargetKey, TimeseriesSchema, UpdateValue, resolved_sum_surrogate, }; pub use exchange::{ExchangeMode, ExchangeOp}; pub use graph::{ BatchEdge, BspSuperstepPlan, BspSuperstepResult, GraphOp, WccSuperstepPlan, WccSuperstepResult, }; -pub use kv::{KvOp, KvResolveOutcome, KvResolvedMutation}; +pub use kv::{ + KvCounterShape, KvOp, KvResolveOutcome, KvResolvedMutation, SortedIndexRead, SortedIndexSpec, +}; pub use meta::{MetaOp, SAVEPOINT_MARKER_BYTES}; pub use plan::PhysicalPlan; pub use query::{AggregateSpec, GroupKeySpec, JoinProjection, QueryOp}; +pub use redo_origin::RedoOrigin; pub use routing::plan_contains_cluster_partitioned_leaf; pub use set_op::SetOpKind; pub use sort_key::SortKeySpec; diff --git a/nodedb-physical/src/physical_plan/redo_origin.rs b/nodedb-physical/src/physical_plan/redo_origin.rs new file mode 100644 index 000000000..0a57970ce --- /dev/null +++ b/nodedb-physical/src/physical_plan/redo_origin.rs @@ -0,0 +1,27 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Where a committed redo record comes from, and so which checks its apply +//! runs. + +/// The source of a redo record a replica installs. +#[derive( + Debug, + Clone, + Copy, + PartialEq, + Eq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum RedoOrigin { + /// A transaction commit. The apply runs every commit-boundary check: + /// BALANCED, UNIQUE, and the stateless PUT and DELETE rules. + Commit, + /// A RESTORE re-installing rows a backup captured. Each row passed its + /// collection's rules when it was first written, and a bitemporal row's + /// earlier versions are history, not new writes. The apply checks only + /// that every unique value has one owner in the post-state. + Restore, +} diff --git a/nodedb-physical/src/surrogate.rs b/nodedb-physical/src/surrogate.rs index 05bd6c463..f99dbe1cd 100644 --- a/nodedb-physical/src/surrogate.rs +++ b/nodedb-physical/src/surrogate.rs @@ -8,7 +8,7 @@ //! code paths. Origin's async surrogate-fetch work stays internal to its impl //! and is hidden behind this sync facade. -use nodedb_types::{DatabaseId, Surrogate, TenantId}; +use nodedb_types::{CollectionKey, Surrogate, TenantId}; /// Errors a [`SurrogateAssigner`] may return. /// @@ -38,14 +38,14 @@ pub trait SurrogateAssigner: Send + Sync { /// allocator. Used by CLONE DATABASE to capture an AS-OF cutoff. fn current_hwm(&self) -> u32; - /// Resolve `(database_id, tenant_id, collection, pk_bytes)` to a stable - /// surrogate. Allocate on the first call; return the persisted value on - /// every subsequent call (UPSERT preserves the surrogate). + /// Resolve `(key, tenant_id, pk_bytes)` to a stable surrogate. Allocate + /// on the first call; return the persisted value on every subsequent + /// call (UPSERT preserves the surrogate). `key` carries the database and + /// the bare catalog name. fn assign( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, pk_bytes: &[u8], ) -> Result; @@ -62,8 +62,7 @@ pub trait SurrogateAssigner: Send + Sync { /// verbatim and never re-derives it. fn assign_fresh( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, ) -> Result<(Surrogate, String), SurrogateAssignError>; } diff --git a/nodedb-query/src/expr/eval.rs b/nodedb-query/src/expr/eval.rs index 04f60e2d6..e73651660 100644 --- a/nodedb-query/src/expr/eval.rs +++ b/nodedb-query/src/expr/eval.rs @@ -18,15 +18,50 @@ use super::types::SqlExpr; /// /// Mirrors the [`WindowError`](crate::window::WindowError) idiom: a small, /// `thiserror`-derived enum living next to the evaluator it describes. -/// Everything that historically folded to `Value::Null` (bad casts, wrong -/// arg counts, unknown-value coercions) keeps doing so — this type exists -/// solely for division/modulo by a zero divisor, which must surface as -/// SQLSTATE `22012` (`division_by_zero`) instead of silently evaluating to -/// `NULL`. -#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)] +/// Bad casts, wrong argument counts and unknown-value coercions fold to +/// `Value::Null`. These faults fail the statement instead: +/// - division or modulo by a zero divisor, SQLSTATE `22012`; +/// - a call to a function no evaluator implements. It never evaluates to a +/// silent `NULL`; +/// - a function argument it cannot compute on: vectors of different +/// dimensions, an argument of the wrong type, a malformed JSONPath. +/// SQLSTATE `22000`. A `NULL` argument stays `NULL` instead. +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] pub enum EvalError { #[error("division by zero")] DivisionByZero, + #[error("function {name}() has no evaluator")] + UnknownFunction { + /// The function name as called. + name: String, + }, + /// Two vector operands have different dimensions. Same message shape + /// as the vector engine's dimension error. + #[error("{function}(): vector dimension mismatch: expected {expected}, got {got}")] + VectorDimensionMismatch { + function: &'static str, + /// Dimension of the first operand. + expected: usize, + /// Dimension of the second operand. + got: usize, + }, + /// An argument holds a value of a type the function cannot compute on. + #[error("{function}(): argument {position} must be {expected}, got {got}")] + ArgumentType { + function: &'static str, + /// 1-based argument position. + position: usize, + expected: &'static str, + /// `Value::type_name` of the argument received. + got: &'static str, + }, + /// A path argument is not a supported JSONPath. + #[error("{function}(): invalid JSONPath {path:?}: {reason}")] + InvalidJsonPath { + function: &'static str, + path: String, + reason: String, + }, } /// Row scope for `SqlExpr::eval_scope`: how `Column(..)` and `OldColumn(..)` diff --git a/nodedb-query/src/functions/array.rs b/nodedb-query/src/functions/array.rs index 45298022a..9b44f625c 100644 --- a/nodedb-query/src/functions/array.rs +++ b/nodedb-query/src/functions/array.rs @@ -88,6 +88,10 @@ pub(super) fn try_eval(name: &str, args: &[Value]) -> Option { reversed.reverse(); Value::Array(reversed) } + // `ARRAY[a, b, ...]` with a non-literal element lowers to this call, + // so the array is built per row from the evaluated elements. A NULL + // element stays a NULL element. + "make_array" => Value::Array(args.to_vec()), _ => return None, }; Some(v) diff --git a/nodedb-query/src/functions/eval.rs b/nodedb-query/src/functions/eval.rs index 636c70aac..cd6e0a920 100644 --- a/nodedb-query/src/functions/eval.rs +++ b/nodedb-query/src/functions/eval.rs @@ -6,19 +6,25 @@ use nodedb_types::Value; use crate::expr::EvalError; -use super::{array, conditional, datetime, fts, id, json, math, string, system, types}; +use super::{ + array, conditional, datetime, fts, id, json, math, string, system, text_chunk, types, vector, +}; /// Evaluate a scalar function call. /// -/// Every function returns `Ok(Value::Null)` on invalid/missing arguments -/// (SQL NULL propagation semantics) — the sole exception is `mod`'s -/// zero-modulus arm (`math::try_eval`), which returns -/// `Err(EvalError::DivisionByZero)`. Every other -/// sibling module here stays `Option`-shaped internally; only the -/// `math` arm is threaded as `Option>` and the -/// rest are wrapped in `Ok` at this dispatch boundary, so a single -/// fallible arm doesn't force every scalar-function module to carry a -/// `Result` it can never actually produce. +/// A `NULL` argument gives `Ok(Value::Null)` (SQL NULL propagation). These +/// calls fail instead: +/// - `mod`'s zero-modulus arm (`math::try_eval`) returns +/// `Err(EvalError::DivisionByZero)`; +/// - a vector distance over operands of different dimensions or over a +/// non-vector operand (`vector::try_eval`); +/// - a document function given a malformed JSONPath (`json::try_eval`); +/// - a name no family module implements returns +/// `Err(EvalError::UnknownFunction)`, never a silent `NULL`. +/// +/// The fallible families (`math`, `vector`, `json`) return +/// `Option>`. The rest stay `Option`-shaped +/// and are wrapped in `Ok` at this dispatch boundary. pub fn eval_function(name: &str, args: &[Value]) -> Result { if let Some(v) = string::try_eval(name, args) { return Ok(v); @@ -35,8 +41,8 @@ pub fn eval_function(name: &str, args: &[Value]) -> Result { if let Some(v) = datetime::try_eval(name, args) { return Ok(v); } - if let Some(v) = json::try_eval(name, args) { - return Ok(v); + if let Some(r) = json::try_eval(name, args) { + return r; } if let Some(v) = types::try_eval(name, args) { return Ok(v); @@ -50,8 +56,16 @@ pub fn eval_function(name: &str, args: &[Value]) -> Result { if let Some(v) = system::try_eval(name, args) { return Ok(v); } + if let Some(r) = vector::try_eval(name, args) { + return r; + } + if let Some(v) = text_chunk::try_eval(name, args) { + return Ok(v); + } // Geo / Spatial functions — delegated to geo_functions module. - Ok(crate::geo_functions::eval_geo_function(name, args).unwrap_or(Value::Null)) + crate::geo_functions::eval_geo_function(name, args).ok_or_else(|| EvalError::UnknownFunction { + name: name.to_owned(), + }) } #[cfg(test)] @@ -69,6 +83,147 @@ mod tests { assert_eq!(err, EvalError::DivisionByZero); } + #[test] + fn an_unknown_function_is_an_error_not_null() { + let err = eval_function("no_such_function", &[Value::Integer(1)]).unwrap_err(); + assert_eq!( + err, + EvalError::UnknownFunction { + name: "no_such_function".into() + } + ); + } + + /// Registered SQL scalars with a per-row meaning dispatch to a real + /// evaluator, never to the unknown-function error. + #[test] + fn registered_row_scalars_have_evaluators() { + let doc = Value::Object( + [( + "tags".to_string(), + Value::Array(vec![Value::String("a".into())]), + )] + .into_iter() + .collect(), + ); + let path = Value::String("$.tags[0]".into()); + let vector = Value::Array(vec![Value::Float(1.0), Value::Float(0.0)]); + let calls: [(&str, Vec); 8] = [ + ("doc_get", vec![doc.clone(), path.clone()]), + ("doc_exists", vec![doc.clone(), path.clone()]), + ( + "doc_array_contains", + vec![ + doc.clone(), + Value::String("$.tags".into()), + Value::String("a".into()), + ], + ), + ("nav", vec![doc, path]), + ("vector_distance", vec![vector.clone(), vector.clone()]), + ( + "vector_cosine_distance", + vec![vector.clone(), vector.clone()], + ), + ("vector_neg_inner_product", vec![vector.clone(), vector]), + ( + "ndb_chunk_text", + vec![Value::String("abc".into()), Value::Integer(2)], + ), + ]; + for (name, args) in calls { + let value = eval_function(name, &args) + .unwrap_or_else(|e| panic!("{name}() must evaluate, got {e}")); + assert_ne!(value, Value::Null, "{name}() gave NULL"); + } + } + + fn text(s: &str) -> Value { + Value::String(s.into()) + } + + #[test] + fn like_matches_percent_and_underscore() { + assert_eq!( + eval_fn("like", vec![text("alice"), text("a%")]), + Value::Bool(true) + ); + assert_eq!( + eval_fn("like", vec![text("alice"), text("a_ice")]), + Value::Bool(true) + ); + assert_eq!( + eval_fn("like", vec![text("alice"), text("b%")]), + Value::Bool(false) + ); + assert_eq!( + eval_fn("like", vec![text("Alice"), text("a%")]), + Value::Bool(false) + ); + } + + #[test] + fn like_honours_the_escape_character() { + assert_eq!( + eval_fn("like", vec![text("50%"), text("50\\%")]), + Value::Bool(true) + ); + assert_eq!( + eval_fn("like", vec![text("500"), text("50\\%")]), + Value::Bool(false) + ); + assert_eq!( + eval_fn("like", vec![text("50%"), text("50!%"), text("!")]), + Value::Bool(true) + ); + assert_eq!( + eval_fn("like", vec![text("a\\b"), text("a\\b"), text("")]), + Value::Bool(true) + ); + } + + #[test] + fn like_with_a_null_operand_is_null() { + assert_eq!(eval_fn("like", vec![Value::Null, text("a%")]), Value::Null); + assert_eq!(eval_fn("ilike", vec![text("a"), Value::Null]), Value::Null); + } + + #[test] + fn ilike_folds_unicode_case() { + assert_eq!( + eval_fn("ilike", vec![text("ÉCOLE"), text("éc%")]), + Value::Bool(true) + ); + assert_eq!( + eval_fn("ilike", vec![text("Straße"), text("STRA%")]), + Value::Bool(true) + ); + assert_eq!( + eval_fn("like", vec![text("ÉCOLE"), text("éc%")]), + Value::Bool(false) + ); + } + + #[test] + fn not_like_negates_the_call() { + let expr = SqlExpr::Negate(Box::new(SqlExpr::Function { + name: "like".into(), + args: vec![SqlExpr::Literal(text("bob")), SqlExpr::Literal(text("a%"))], + })); + assert_eq!(expr.eval(&Value::Null).unwrap(), Value::Bool(true)); + } + + #[test] + fn make_array_keeps_every_evaluated_element() { + assert_eq!( + eval_fn( + "make_array", + vec![Value::Integer(5), Value::Null, text("x")] + ), + Value::Array(vec![Value::Integer(5), Value::Null, text("x")]) + ); + } + #[test] fn upper() { assert_eq!( diff --git a/nodedb-query/src/functions/json/dispatch.rs b/nodedb-query/src/functions/json/dispatch.rs index d5270733f..2e2dd3fc1 100644 --- a/nodedb-query/src/functions/json/dispatch.rs +++ b/nodedb-query/src/functions/json/dispatch.rs @@ -4,7 +4,28 @@ use nodedb_types::Value; -pub(in crate::functions) fn try_eval(name: &str, args: &[Value]) -> Option { +use crate::expr::EvalError; + +/// `None` when no JSON-family function has `name`. The document functions +/// can fail with a typed error; every other JSON function returns a value. +pub(in crate::functions) fn try_eval( + name: &str, + args: &[Value], +) -> Option> { + let doc_result = match name { + "doc_get" => Some(super::doc::doc_get(args)), + "doc_exists" => Some(super::doc::doc_exists(args)), + "doc_array_contains" => Some(super::doc::doc_array_contains(args)), + "nav" => Some(super::doc::nav(args)), + _ => None, + }; + if doc_result.is_some() { + return doc_result; + } + try_eval_value(name, args).map(Ok) +} + +fn try_eval_value(name: &str, args: &[Value]) -> Option { // PostgreSQL JSON operator functions (lowered from AST BinaryOp). let pg_result = match name { "pg_json_get" => { diff --git a/nodedb-query/src/functions/json/doc.rs b/nodedb-query/src/functions/json/doc.rs new file mode 100644 index 000000000..204284a2c --- /dev/null +++ b/nodedb-query/src/functions/json/doc.rs @@ -0,0 +1,271 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Document navigation functions: `doc_get`, `doc_exists`, +//! `doc_array_contains`, and `nav`. +//! +//! Each takes a document value and a path. The path is JSONPath (`$.a.b`, +//! `$.arr[0]`). A path without the leading `$` is read as `$.` plus the +//! path, so `'user.name'` and `'$.user.name'` name the same field. A document +//! held as JSON text is parsed before the walk. +//! +//! A malformed path, or a path argument that is not text, fails the +//! statement with a typed [`EvalError`]. A `NULL` path gives `NULL`. A path +//! that resolves to nothing gives the default, `NULL`, or `false`. + +use nodedb_types::Value; + +use super::path::{PathStep, parse_jsonpath, walk_path}; +use super::pg_ops::coerce_json_string; +use crate::expr::EvalError; +use crate::value_ops::coerced_eq; + +/// Parse the path argument at 1-based `position`. `Ok(None)` for a `NULL` +/// path. +fn path_steps( + function: &'static str, + position: usize, + path: &Value, +) -> Result>, EvalError> { + let text = match path { + Value::Null => return Ok(None), + Value::String(text) => text, + other => { + return Err(EvalError::ArgumentType { + function, + position, + expected: "a JSONPath text", + got: other.type_name(), + }); + } + }; + let parsed = if text.starts_with('$') { + parse_jsonpath(text) + } else { + parse_jsonpath(&format!("$.{text}")) + }; + parsed.map(Some).map_err(|e| EvalError::InvalidJsonPath { + function, + path: text.clone(), + reason: e.to_string(), + }) +} + +/// The non-null value at the path in `args[1]` inside the document in +/// `args[0]`, cloned. `Ok(None)` when the path is `NULL` or resolves to +/// nothing. +fn resolve(function: &'static str, args: &[Value]) -> Result, EvalError> { + let doc = args.first().unwrap_or(&Value::Null); + let path = args.get(1).unwrap_or(&Value::Null); + let Some(steps) = path_steps(function, 2, path)? else { + return Ok(None); + }; + let doc = coerce_json_string(doc); + Ok(match walk_path(&doc, &steps) { + Some(Value::Null) | None => None, + Some(found) => Some(found.clone()), + }) +} + +/// `doc_get(doc, path [, default])`: the value at `path`, or `default` when +/// the path is missing or null. `default` is `NULL` when omitted. +pub(super) fn doc_get(args: &[Value]) -> Result { + Ok(resolve("doc_get", args)?.unwrap_or_else(|| args.get(2).cloned().unwrap_or(Value::Null))) +} + +/// `nav(doc, path)`: the value at `path`, or `NULL`. +pub(super) fn nav(args: &[Value]) -> Result { + Ok(resolve("nav", args)?.unwrap_or(Value::Null)) +} + +/// `doc_exists(doc, path)`: whether `path` holds a non-null value. +pub(super) fn doc_exists(args: &[Value]) -> Result { + Ok(Value::Bool(resolve("doc_exists", args)?.is_some())) +} + +/// `doc_array_contains(doc, path, value)`: whether the array at `path` holds +/// an element equal to `value`. Equality coerces a numeric string to a +/// number and an ISO-8601 string to an instant. A missing path or a +/// non-array value at the path gives `false`. +pub(super) fn doc_array_contains(args: &[Value]) -> Result { + let needle = args.get(2).unwrap_or(&Value::Null); + let contains = match resolve("doc_array_contains", args)? { + Some(Value::Array(items) | Value::Set(items)) => { + items.iter().any(|item| coerced_eq(item, needle)) + } + Some(_) | None => false, + }; + Ok(Value::Bool(contains)) +} + +#[cfg(test)] +mod tests { + use super::*; + use std::collections::HashMap; + + fn obj(pairs: &[(&str, Value)]) -> Value { + Value::Object( + pairs + .iter() + .map(|(k, v)| (k.to_string(), v.clone())) + .collect::>(), + ) + } + + fn text(s: &str) -> Value { + Value::String(s.into()) + } + + fn event() -> Value { + obj(&[ + ( + "user", + obj(&[("name", text("ada")), ("email", Value::Null)]), + ), + ( + "tags", + Value::Array(vec![text("important"), text("ops"), Value::Integer(7)]), + ), + ]) + } + + #[test] + fn doc_get_reads_a_nested_field() { + assert_eq!(doc_get(&[event(), text("$.user.name")]), Ok(text("ada"))); + } + + #[test] + fn doc_get_reads_a_path_without_the_dollar() { + assert_eq!(doc_get(&[event(), text("user.name")]), Ok(text("ada"))); + } + + #[test] + fn doc_get_reads_an_array_element() { + assert_eq!(doc_get(&[event(), text("$.tags[1]")]), Ok(text("ops"))); + } + + #[test] + fn doc_get_missing_path_gives_the_default() { + assert_eq!( + doc_get(&[event(), text("$.user.age"), Value::Integer(0)]), + Ok(Value::Integer(0)) + ); + assert_eq!(doc_get(&[event(), text("$.user.age")]), Ok(Value::Null)); + } + + #[test] + fn doc_get_null_field_gives_the_default() { + assert_eq!( + doc_get(&[event(), text("$.user.email"), text("none")]), + Ok(text("none")) + ); + } + + #[test] + fn doc_get_parses_a_json_text_document() { + let doc = text(r#"{"user":{"name":"ada"}}"#); + assert_eq!(doc_get(&[doc, text("$.user.name")]), Ok(text("ada"))); + } + + #[test] + fn doc_get_malformed_path_is_an_error() { + let err = doc_get(&[event(), text("$..name"), text("d")]).unwrap_err(); + assert!( + matches!( + &err, + EvalError::InvalidJsonPath { function: "doc_get", path, .. } if path == "$..name" + ), + "{err:?}" + ); + let err = doc_get(&[event(), text("$.tags[x]")]).unwrap_err(); + assert!(matches!(err, EvalError::InvalidJsonPath { .. }), "{err:?}"); + } + + #[test] + fn doc_array_contains_malformed_path_is_an_error() { + let err = doc_array_contains(&[event(), text("$.tags["), text("ops")]).unwrap_err(); + assert!( + matches!( + err, + EvalError::InvalidJsonPath { + function: "doc_array_contains", + .. + } + ), + "{err:?}" + ); + } + + #[test] + fn a_non_text_path_is_an_argument_error() { + let err = doc_exists(&[event(), Value::Integer(3)]).unwrap_err(); + assert_eq!( + err, + EvalError::ArgumentType { + function: "doc_exists", + position: 2, + expected: "a JSONPath text", + got: "int", + } + ); + } + + #[test] + fn a_null_path_resolves_to_nothing() { + assert_eq!(doc_get(&[event(), Value::Null, text("d")]), Ok(text("d"))); + assert_eq!(doc_exists(&[event(), Value::Null]), Ok(Value::Bool(false))); + } + + #[test] + fn nav_matches_doc_get_without_a_default() { + assert_eq!(nav(&[event(), text("$.user.name")]), Ok(text("ada"))); + assert_eq!(nav(&[event(), text("$.missing")]), Ok(Value::Null)); + } + + #[test] + fn doc_exists_is_true_only_for_a_non_null_value() { + assert_eq!( + doc_exists(&[event(), text("$.user.name")]), + Ok(Value::Bool(true)) + ); + assert_eq!( + doc_exists(&[event(), text("$.user.email")]), + Ok(Value::Bool(false)) + ); + assert_eq!( + doc_exists(&[event(), text("$.nope")]), + Ok(Value::Bool(false)) + ); + assert_eq!( + doc_exists(&[Value::Null, text("$.a")]), + Ok(Value::Bool(false)) + ); + } + + #[test] + fn doc_array_contains_finds_an_element() { + assert_eq!( + doc_array_contains(&[event(), text("$.tags"), text("important")]), + Ok(Value::Bool(true)) + ); + assert_eq!( + doc_array_contains(&[event(), text("$.tags"), text("absent")]), + Ok(Value::Bool(false)) + ); + } + + #[test] + fn doc_array_contains_coerces_a_numeric_string() { + assert_eq!( + doc_array_contains(&[event(), text("$.tags"), text("7")]), + Ok(Value::Bool(true)) + ); + } + + #[test] + fn doc_array_contains_on_a_non_array_is_false() { + assert_eq!( + doc_array_contains(&[event(), text("$.user.name"), text("ada")]), + Ok(Value::Bool(false)) + ); + } +} diff --git a/nodedb-query/src/functions/json/mod.rs b/nodedb-query/src/functions/json/mod.rs index 8b5652181..3677ec16c 100644 --- a/nodedb-query/src/functions/json/mod.rs +++ b/nodedb-query/src/functions/json/mod.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: Apache-2.0 mod dispatch; +mod doc; pub(super) mod legacy; pub(crate) mod path; pub(super) mod pg_ops; diff --git a/nodedb-query/src/functions/json/pg_ops.rs b/nodedb-query/src/functions/json/pg_ops.rs index 49757a0f3..900094c3b 100644 --- a/nodedb-query/src/functions/json/pg_ops.rs +++ b/nodedb-query/src/functions/json/pg_ops.rs @@ -17,7 +17,7 @@ use nodedb_types::Value; /// /// This is a cheap path: non-string values pass through a single `matches!` /// check; strings that are not valid JSON also return quickly from the parser. -fn coerce_json_string(v: &Value) -> std::borrow::Cow<'_, Value> { +pub(super) fn coerce_json_string(v: &Value) -> std::borrow::Cow<'_, Value> { if let Value::String(s) = v && let Ok(parsed) = sonic_rs::from_str::(s) { diff --git a/nodedb-query/src/functions/mod.rs b/nodedb-query/src/functions/mod.rs index 04dcfdbb8..a7b26ec68 100644 --- a/nodedb-query/src/functions/mod.rs +++ b/nodedb-query/src/functions/mod.rs @@ -16,6 +16,8 @@ mod math; pub(crate) mod shared; mod string; mod system; +mod text_chunk; mod types; +mod vector; pub use eval::eval_function; diff --git a/nodedb-query/src/functions/string.rs b/nodedb-query/src/functions/string.rs index 84f4b80a3..bd68ef50f 100644 --- a/nodedb-query/src/functions/string.rs +++ b/nodedb-query/src/functions/string.rs @@ -3,6 +3,7 @@ //! String scalar functions. use super::shared::{num_arg, str_arg}; +use crate::scan_filter::like::{DEFAULT_LIKE_ESCAPE, sql_like_match_escaped}; use crate::value_ops::value_to_display_string; use nodedb_types::Value; @@ -48,7 +49,52 @@ pub(super) fn try_eval(name: &str, args: &[Value]) -> Option { "reverse" => { str_arg(args, 0).map_or(Value::Null, |s| Value::String(s.chars().rev().collect())) } + "like" => like(args, false), + "ilike" => like(args, true), _ => return None, }; Some(v) } + +/// `like(input, pattern[, escape])` / `ilike(...)`: SQL `LIKE` / `ILIKE`. +/// +/// - A NULL input or pattern gives NULL. +/// - A non-text scalar operand matches as its display text. +/// - `escape` defaults to `\`. An empty escape disables escaping. An escape +/// longer than one character is invalid and gives NULL. +/// +/// `NOT LIKE` is the negation of this call, so NULL stays NULL. +fn like(args: &[Value], case_insensitive: bool) -> Value { + let (Some(input), Some(pattern)) = (like_text(args.first()), like_text(args.get(1))) else { + return Value::Null; + }; + let escape = match args.get(2) { + None => Some(DEFAULT_LIKE_ESCAPE), + Some(v) => { + let Some(text) = like_text(Some(v)) else { + return Value::Null; + }; + let mut chars = text.chars(); + match (chars.next(), chars.next()) { + (None, _) => None, + (Some(c), None) => Some(c), + (Some(_), Some(_)) => return Value::Null, + } + } + }; + Value::Bool(sql_like_match_escaped( + &input, + &pattern, + case_insensitive, + escape, + )) +} + +/// The text a LIKE operand matches as. `None` for NULL or a missing operand. +fn like_text(v: Option<&Value>) -> Option { + match v? { + Value::Null => None, + Value::String(s) => Some(s.clone()), + other => Some(value_to_display_string(other)), + } +} diff --git a/nodedb-query/src/functions/text_chunk.rs b/nodedb-query/src/functions/text_chunk.rs new file mode 100644 index 000000000..35456c8af --- /dev/null +++ b/nodedb-query/src/functions/text_chunk.rs @@ -0,0 +1,90 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! `ndb_chunk_text(text, chunk_size [, overlap])` as a per-row scalar. +//! +//! The scalar form returns the chunk texts as an array, split on character +//! boundaries, the default strategy of the `SELECT * FROM NDB_CHUNK_TEXT(...)` +//! table function. `overlap` is 0 when omitted. A `NULL` or non-text `text`, +//! a `chunk_size` of 0 or less, or an `overlap` not below `chunk_size` gives +//! `NULL`. + +use nodedb_types::Value; + +use crate::chunk_text::{ChunkStrategy, chunk_text}; + +pub(super) fn try_eval(name: &str, args: &[Value]) -> Option { + if name != "ndb_chunk_text" { + return None; + } + Some(eval_chunk_text(args).unwrap_or(Value::Null)) +} + +fn eval_chunk_text(args: &[Value]) -> Option { + let text = args.first()?.as_str()?; + let chunk_size = count_arg(args.get(1)?)?; + let overlap = match args.get(2) { + Some(value) => count_arg(value)?, + None => 0, + }; + let chunks = chunk_text(text, chunk_size, overlap, ChunkStrategy::Character).ok()?; + Some(Value::Array( + chunks + .into_iter() + .map(|chunk| Value::String(chunk.text)) + .collect(), + )) +} + +fn count_arg(value: &Value) -> Option { + match value { + Value::Integer(n) => usize::try_from(*n).ok(), + _ => None, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn text(s: &str) -> Value { + Value::String(s.into()) + } + + #[test] + fn splits_into_character_chunks() { + let out = try_eval("ndb_chunk_text", &[text("abcdef"), Value::Integer(4)]); + assert_eq!(out, Some(Value::Array(vec![text("abcd"), text("ef")]))); + } + + #[test] + fn overlap_repeats_the_tail() { + let out = try_eval( + "ndb_chunk_text", + &[text("abcdef"), Value::Integer(4), Value::Integer(2)], + ); + assert_eq!(out, Some(Value::Array(vec![text("abcd"), text("cdef")]))); + } + + #[test] + fn invalid_sizes_are_null() { + assert_eq!( + try_eval("ndb_chunk_text", &[text("abc"), Value::Integer(0)]), + Some(Value::Null) + ); + assert_eq!( + try_eval( + "ndb_chunk_text", + &[text("abc"), Value::Integer(2), Value::Integer(2)] + ), + Some(Value::Null) + ); + } + + #[test] + fn null_text_is_null() { + assert_eq!( + try_eval("ndb_chunk_text", &[Value::Null, Value::Integer(2)]), + Some(Value::Null) + ); + } +} diff --git a/nodedb-query/src/functions/vector.rs b/nodedb-query/src/functions/vector.rs new file mode 100644 index 000000000..b598cd2a1 --- /dev/null +++ b/nodedb-query/src/functions/vector.rs @@ -0,0 +1,253 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Per-row vector distance functions. +//! +//! `vector_distance` (the `<->` operator), `vector_cosine_distance` (`<=>`) +//! and `vector_neg_inner_product` (`<#>`) give the same numbers a vector +//! search reports for the metric each name selects: +//! - `vector_distance`: squared Euclidean (L2) distance; +//! - `vector_cosine_distance`: `1 - cosine similarity`; +//! - `vector_neg_inner_product`: the negated dot product. +//! +//! A vector search plan serves these calls in `ORDER BY`. Every other +//! position evaluates them here, once per row. A `NULL` operand gives `NULL`. +//! Operands of different dimensions, or an operand that is not a numeric +//! vector, fail the statement with a typed [`EvalError`]. + +use nodedb_types::Value; +use nodedb_types::vector_distance::{cosine_distance, l2_squared, neg_inner_product}; + +use crate::expr::EvalError; + +/// The expected-type text an argument error names. +const NUMERIC_VECTOR: &str = "a numeric vector"; + +/// A distance between two vectors of equal dimension. +type Metric = fn(&[f32], &[f32]) -> f32; + +pub(super) fn try_eval(name: &str, args: &[Value]) -> Option> { + let (function, metric): (&'static str, Metric) = match name { + "vector_distance" => ("vector_distance", l2_squared), + "vector_cosine_distance" => ("vector_cosine_distance", cosine_distance), + "vector_neg_inner_product" => ("vector_neg_inner_product", neg_inner_product), + _ => return None, + }; + Some(eval_distance(function, metric, args)) +} + +fn eval_distance( + function: &'static str, + metric: Metric, + args: &[Value], +) -> Result { + let left = args.first().unwrap_or(&Value::Null); + let right = args.get(1).unwrap_or(&Value::Null); + if matches!(left, Value::Null) || matches!(right, Value::Null) { + return Ok(Value::Null); + } + let a = as_vector(function, 1, left)?; + let b = as_vector(function, 2, right)?; + if a.len() != b.len() { + return Err(EvalError::VectorDimensionMismatch { + function, + expected: a.len(), + got: b.len(), + }); + } + Ok(Value::Float(f64::from(metric(&a, &b)))) +} + +/// Read a vector argument: a vector value, an array of numbers, or JSON text +/// holding an array of numbers. `position` is 1-based. +fn as_vector( + function: &'static str, + position: usize, + value: &Value, +) -> Result, EvalError> { + let wrong_type = |got: &'static str| EvalError::ArgumentType { + function, + position, + expected: NUMERIC_VECTOR, + got, + }; + match value { + Value::Vector(floats) => Ok(floats.to_vec()), + Value::Array(items) => items + .iter() + .map(|item| number(item).ok_or_else(|| wrong_type(item.type_name()))) + .collect(), + Value::String(text) => sonic_rs::from_str::>(text) + .map(|parsed| parsed.into_iter().map(|x| x as f32).collect()) + .map_err(|_| wrong_type(value.type_name())), + other => Err(wrong_type(other.type_name())), + } +} + +fn number(value: &Value) -> Option { + match value { + Value::Float(f) => Some(*f as f32), + Value::Integer(i) => Some(*i as f32), + Value::Decimal(d) => { + use rust_decimal::prelude::ToPrimitive; + d.to_f32() + } + _ => None, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn vector(items: &[f64]) -> Value { + Value::Array(items.iter().map(|x| Value::Float(*x)).collect()) + } + + fn eval(name: &str, a: Value, b: Value) -> Result { + try_eval(name, &[a, b]).expect("vector function") + } + + #[test] + fn l2_is_squared_euclidean() { + assert_eq!( + eval("vector_distance", vector(&[0.0, 0.0]), vector(&[3.0, 4.0])), + Ok(Value::Float(25.0)) + ); + } + + #[test] + fn decimal_literal_elements_are_numeric() { + let decimals = Value::Array(vec![ + Value::Decimal(rust_decimal::Decimal::new(30, 1)), + Value::Decimal(rust_decimal::Decimal::new(40, 1)), + ]); + assert_eq!( + eval("vector_distance", vector(&[0.0, 0.0]), decimals), + Ok(Value::Float(25.0)) + ); + } + + #[test] + fn cosine_of_orthogonal_vectors_is_one() { + let Ok(Value::Float(d)) = eval( + "vector_cosine_distance", + vector(&[1.0, 0.0]), + vector(&[0.0, 1.0]), + ) else { + panic!("expected a float"); + }; + assert!((d - 1.0).abs() < 1e-6); + } + + #[test] + fn neg_inner_product_negates_the_dot_product() { + assert_eq!( + eval( + "vector_neg_inner_product", + vector(&[1.0, 2.0]), + vector(&[3.0, 4.0]) + ), + Ok(Value::Float(-11.0)) + ); + } + + #[test] + fn vector_values_integer_elements_and_json_text_are_vectors() { + let ints = Value::Array(vec![Value::Integer(0), Value::Integer(0)]); + assert_eq!( + eval("vector_distance", ints, Value::String("[3, 4]".into())), + Ok(Value::Float(25.0)) + ); + let packed = Value::Vector(vec![3.0_f32, 4.0].into()); + assert_eq!( + eval("vector_distance", packed, vector(&[0.0, 0.0])), + Ok(Value::Float(25.0)) + ); + } + + #[test] + fn dimension_mismatch_names_both_dimensions() { + let err = eval("vector_distance", vector(&[1.0]), vector(&[1.0, 2.0])).unwrap_err(); + assert_eq!( + err, + EvalError::VectorDimensionMismatch { + function: "vector_distance", + expected: 1, + got: 2, + } + ); + assert_eq!( + err.to_string(), + "vector_distance(): vector dimension mismatch: expected 1, got 2" + ); + } + + #[test] + fn non_vector_argument_names_its_position_and_type() { + let err = eval("vector_cosine_distance", vector(&[1.0]), Value::Integer(1)).unwrap_err(); + assert_eq!( + err, + EvalError::ArgumentType { + function: "vector_cosine_distance", + position: 2, + expected: NUMERIC_VECTOR, + got: "int", + } + ); + } + + #[test] + fn non_numeric_element_names_the_element_type() { + let mixed = Value::Array(vec![Value::Float(1.0), Value::String("x".into())]); + let err = eval("vector_distance", mixed, vector(&[1.0, 2.0])).unwrap_err(); + assert!( + matches!( + err, + EvalError::ArgumentType { + position: 1, + got: "string", + .. + } + ), + "{err:?}" + ); + } + + #[test] + fn text_that_is_not_a_vector_is_an_argument_error() { + let err = eval( + "vector_distance", + Value::String("not a vector".into()), + vector(&[1.0]), + ) + .unwrap_err(); + assert!( + matches!( + err, + EvalError::ArgumentType { + position: 1, + got: "string", + .. + } + ), + "{err:?}" + ); + } + + #[test] + fn null_operand_is_null() { + assert_eq!( + eval("vector_distance", Value::Null, vector(&[1.0])), + Ok(Value::Null) + ); + assert_eq!( + eval("vector_distance", vector(&[1.0]), Value::Null), + Ok(Value::Null) + ); + } + + #[test] + fn other_names_are_not_claimed() { + assert!(try_eval("bm25_score", &[]).is_none()); + } +} diff --git a/nodedb-query/src/msgpack_scan/kv_body.rs b/nodedb-query/src/msgpack_scan/kv_body.rs index 0e746142c..a5b22fd53 100644 --- a/nodedb-query/src/msgpack_scan/kv_body.rs +++ b/nodedb-query/src/msgpack_scan/kv_body.rs @@ -9,6 +9,8 @@ //! Decoding raw bytes as msgpack instead either fails (multi-byte value) or //! reads the first byte as a fixint and discards the rest. +use std::collections::HashMap; + use nodedb_types::{MsgpackError, NotScalar, Value, scalar_to_raw_bytes}; use crate::msgpack_scan::{map_header, write_map_header, write_str}; @@ -124,10 +126,91 @@ pub fn row_to_kv_body(row: &Value, shape: KvBodyShape) -> Result, KvBody } } +/// The fields a KV body stores for a msgpack-encoded row, and the body shape. +/// +/// This is the inverse of [`super::kv_row_msgpack`]. `key` is the entry's key. +/// A `key` field equal to it is the injected primary key and is dropped. A +/// `key` field that differs is a stored column and is kept. The remaining +/// fields pick the shape the SQL lowering picks for a fresh insert: +/// - `value` alone: [`KvBodyShape::Raw`]; +/// - any other field set: [`KvBodyShape::Map`]. +/// +/// The caller encodes the fields in its own body format. +/// [`row_to_kv_body`] is the Origin format. +pub fn kv_row_to_body_fields( + key: &str, + row: &[u8], +) -> Result<(HashMap, KvBodyShape), KvBodyError> { + let row = nodedb_types::value_from_msgpack(row)?; + let Value::Object(mut fields) = row else { + return Err(KvBodyError::RowNotObject { + kind: row.type_name(), + }); + }; + if matches!(fields.get("key"), Some(Value::String(stored)) if stored == key) { + fields.remove("key"); + } + let shape = if fields.len() == 1 && fields.contains_key("value") { + KvBodyShape::Raw + } else { + KvBodyShape::Map + }; + Ok((fields, shape)) +} + #[cfg(test)] mod tests { use super::*; - use std::collections::HashMap; + + fn body_from_row(key: &str, row: &[u8]) -> Vec { + let (fields, shape) = kv_row_to_body_fields(key, row).unwrap(); + row_to_kv_body(&Value::Object(fields), shape).unwrap() + } + + #[test] + fn a_raw_body_round_trips_through_its_row() { + for body in [b"v1".to_vec(), b"1".to_vec(), Vec::new()] { + let row = crate::msgpack_scan::kv_row_msgpack("k1", &body); + assert_eq!(body_from_row("k1", &row), body); + } + } + + #[test] + fn a_map_body_round_trips_through_its_row() { + let mut fields = HashMap::new(); + fields.insert("n".to_string(), Value::Integer(7)); + fields.insert("s".to_string(), Value::String("x".into())); + let body = row_to_kv_body(&Value::Object(fields), KvBodyShape::Map).unwrap(); + let row = crate::msgpack_scan::kv_row_msgpack("k2", &body); + assert_eq!(body_from_row("k2", &row), body); + } + + #[test] + fn a_key_column_that_differs_from_the_entry_key_is_kept() { + let mut fields = HashMap::new(); + fields.insert("key".to_string(), Value::String("stored".into())); + fields.insert("n".to_string(), Value::Integer(1)); + let body = row_to_kv_body(&Value::Object(fields), KvBodyShape::Map).unwrap(); + let row = crate::msgpack_scan::kv_row_msgpack("slot", &body); + assert_eq!(body_from_row("slot", &row), body); + } + + #[test] + fn a_row_that_is_not_a_map_is_an_error() { + let not_a_map = nodedb_types::value_to_msgpack(&Value::Integer(118)).unwrap(); + assert!(matches!( + kv_row_to_body_fields("k", ¬_a_map), + Err(KvBodyError::RowNotObject { kind: "int" }) + )); + } + + #[test] + fn bytes_that_are_not_msgpack_are_a_decode_error() { + assert!(matches!( + kv_row_to_body_fields("k", &[0x81]), + Err(KvBodyError::Decode(_)) + )); + } fn value_of(row: &Value) -> &Value { row.get("value").expect("row carries `value`") diff --git a/nodedb-query/src/msgpack_scan/mod.rs b/nodedb-query/src/msgpack_scan/mod.rs index af657f92b..939c67a8d 100644 --- a/nodedb-query/src/msgpack_scan/mod.rs +++ b/nodedb-query/src/msgpack_scan/mod.rs @@ -24,7 +24,9 @@ pub use compare::{compare_field_bytes, hash_field_bytes}; pub use field::{extract_field, extract_path}; pub use group_key::build_group_key; pub use index::FieldIndex; -pub use kv_body::{KvBodyError, KvBodyShape, kv_body_shape, kv_body_to_row, row_to_kv_body}; +pub use kv_body::{ + KvBodyError, KvBodyShape, kv_body_shape, kv_body_to_row, kv_row_to_body_fields, row_to_kv_body, +}; pub use kv_row::kv_row_msgpack; pub use reader::{ array_header, map_header, read_bin_advance, read_bool, read_f64, read_i64, read_null, read_str, diff --git a/nodedb-query/src/scan_filter/like.rs b/nodedb-query/src/scan_filter/like.rs index 7604715f9..90033bd22 100644 --- a/nodedb-query/src/scan_filter/like.rs +++ b/nodedb-query/src/scan_filter/like.rs @@ -1,60 +1,155 @@ // SPDX-License-Identifier: Apache-2.0 -/// SQL LIKE pattern matching. -/// -/// Supports `%` (zero or more characters) and `_` (exactly one character). -/// When `case_insensitive` is true, both input and pattern are lowercased (ILIKE). +//! SQL `LIKE` / `ILIKE` pattern matching: the one matcher every LIKE path +//! uses, the scan filters and the `like` / `ilike` scalar functions alike. +//! +//! - `%` matches zero or more characters. +//! - `_` matches exactly one character (a Unicode scalar value, not a byte). +//! - The escape character makes the next pattern character literal. The +//! default escape is `\`, as in PostgreSQL. A trailing escape character +//! matches itself. +//! - `ILIKE` lowercases input and pattern by Unicode rules before matching. + +/// The escape character a pattern uses when none is given. +pub const DEFAULT_LIKE_ESCAPE: char = '\\'; + +/// Match `input` against the SQL LIKE `pattern`, with `\` as the escape. pub fn sql_like_match(input: &str, pattern: &str, case_insensitive: bool) -> bool { + sql_like_match_escaped(input, pattern, case_insensitive, Some(DEFAULT_LIKE_ESCAPE)) +} + +/// Match `input` against the SQL LIKE `pattern` with the escape character +/// `escape`. `None` means the pattern has no escape character. +pub fn sql_like_match_escaped( + input: &str, + pattern: &str, + case_insensitive: bool, + escape: Option, +) -> bool { let (input, pattern) = if case_insensitive { (input.to_lowercase(), pattern.to_lowercase()) } else { - (input.to_string(), pattern.to_string()) + (input.to_owned(), pattern.to_owned()) }; + let input: Vec = input.chars().collect(); + let tokens = tokenize(&pattern, escape); + matches(&input, &tokens) +} - let input = input.as_bytes(); - let pattern = pattern.as_bytes(); - - let (mut i, mut j) = (0usize, 0usize); - let (mut star_j, mut star_i) = (usize::MAX, 0usize); +/// One element of a compiled pattern. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum Token { + /// A character that must match itself. + Char(char), + /// `_`: any one character. + AnyOne, + /// `%`: any run of characters, the empty run included. + AnyRun, +} - while i < input.len() { - if j < pattern.len() && (pattern[j] == b'_' || pattern[j] == input[i]) { - i += 1; - j += 1; - } else if j < pattern.len() && pattern[j] == b'%' { - star_j = j; - star_i = i; - j += 1; - } else if star_j != usize::MAX { - star_i += 1; - i = star_i; - j = star_j + 1; +fn tokenize(pattern: &str, escape: Option) -> Vec { + let mut tokens = Vec::with_capacity(pattern.len()); + let mut chars = pattern.chars(); + while let Some(c) = chars.next() { + let token = if Some(c) == escape { + // The escaped character is literal. A trailing escape is itself. + Token::Char(chars.next().unwrap_or(c)) } else { - return false; - } + match c { + '%' => Token::AnyRun, + '_' => Token::AnyOne, + other => Token::Char(other), + } + }; + tokens.push(token); } + tokens +} - while j < pattern.len() && pattern[j] == b'%' { - j += 1; +/// Greedy match with single-point backtracking to the last `%`. +fn matches(input: &[char], tokens: &[Token]) -> bool { + let (mut i, mut t) = (0usize, 0usize); + let mut backtrack: Option<(usize, usize)> = None; + while i < input.len() { + match tokens.get(t) { + Some(Token::AnyRun) => { + backtrack = Some((t, i)); + t += 1; + continue; + } + Some(Token::AnyOne) => { + i += 1; + t += 1; + continue; + } + Some(Token::Char(c)) if *c == input[i] => { + i += 1; + t += 1; + continue; + } + Some(Token::Char(_)) | None => {} + } + match backtrack { + Some((run_t, run_i)) => { + backtrack = Some((run_t, run_i + 1)); + i = run_i + 1; + t = run_t + 1; + } + None => return false, + } } - - j == pattern.len() + tokens[t..].iter().all(|token| *token == Token::AnyRun) } #[cfg(test)] mod tests { - use super::sql_like_match; + use super::*; #[test] fn like_basic() { assert!(sql_like_match("hello world", "%world", false)); assert!(sql_like_match("hello world", "hello%", false)); assert!(!sql_like_match("hello world", "xyz%", false)); + assert!(sql_like_match("", "%", false)); + assert!(!sql_like_match("", "_", false)); + } + + #[test] + fn underscore_matches_one_character_not_one_byte() { + assert!(sql_like_match("é", "_", false)); + assert!(sql_like_match("naïve", "na_ve", false)); + assert!(!sql_like_match("naïve", "na__ve", false)); + } + + #[test] + fn escape_makes_wildcards_literal() { + assert!(sql_like_match("100%", "100\\%", false)); + assert!(!sql_like_match("1000", "100\\%", false)); + assert!(sql_like_match("a_b", "a\\_b", false)); + assert!(!sql_like_match("axb", "a\\_b", false)); + assert!(sql_like_match("a\\", "a\\", false)); + assert!(sql_like_match_escaped("50!%", "50!!!%", false, Some('!'))); + assert!(sql_like_match_escaped("a\\b", "a\\b", false, None)); } #[test] fn ilike_case_insensitive() { assert!(sql_like_match("Hello", "hello", true)); assert!(sql_like_match("WORLD", "%world%", true)); + assert!(!sql_like_match("WORLD", "%world%", false)); + } + + #[test] + fn ilike_folds_unicode_case() { + assert!(sql_like_match("ÉCOLE", "école", true)); + assert!(sql_like_match("ΣΟΦΙΑ", "σοφια", true)); + assert!(!sql_like_match("ÉCOLE", "école", false)); + } + + #[test] + fn backtracking_finds_a_later_match() { + assert!(sql_like_match("abcabd", "%abd", false)); + assert!(sql_like_match("aaa", "%a%a", false)); + assert!(!sql_like_match("abc", "%d%", false)); } } diff --git a/nodedb-query/src/window/value_eval.rs b/nodedb-query/src/window/value_eval.rs index 1ba2da874..b4d1cbabb 100644 --- a/nodedb-query/src/window/value_eval.rs +++ b/nodedb-query/src/window/value_eval.rs @@ -28,7 +28,7 @@ pub enum WindowError { #[error("window frame error: {detail}")] BadFrame { detail: String }, - #[error("division by zero in window expression")] + #[error("window expression: {0}")] Eval(#[from] crate::expr::EvalError), } diff --git a/nodedb-sql/src/error.rs b/nodedb-sql/src/error.rs index eaa5733aa..b02210889 100644 --- a/nodedb-sql/src/error.rs +++ b/nodedb-sql/src/error.rs @@ -37,6 +37,19 @@ pub enum SqlError { )] SequencePerRowUnsupported { name: String }, + /// An index-owned search function (`bm25_score`, `text_match`, + /// `rrf_score`, `sparse_score`, ...) sits in a position the row evaluator + /// runs: the planner could not lower it into its search plan. The + /// function reads a search index and has no per-row value. + /// + /// Rendered as SQLSTATE `0A000` (feature_not_supported). + #[error( + "{name}(...) reads a search index and has no per-row value here; \ + call it in ORDER BY, in WHERE as the search predicate, or in the \ + SELECT list of a search, with a literal query" + )] + SearchFunctionOutsideSearch { name: String }, + /// A statement names a database object that does not exist — a sequence, /// most commonly. Distinct from [`SqlError::UndefinedFunction`]: the /// function exists, the object it names does not. PostgreSQL rejects the @@ -71,6 +84,13 @@ pub enum SqlError { #[error("division by zero")] DivisionByZero, + /// A constant function call received an argument it cannot compute on: + /// vectors of different dimensions, an argument of the wrong type, a + /// malformed JSONPath. Same distinction as [`SqlError::DivisionByZero`]. + /// Rendered as SQLSTATE `22000` (`data_exception`). + #[error("{detail}")] + DataException { detail: String }, + /// A LIMIT, OFFSET, or FETCH FIRST clause resolved to a value outside /// `[0, usize::MAX]`, or to an expression the planner cannot read as a /// literal. PostgreSQL rejects the same input with SQLSTATE `2201W`. diff --git a/nodedb-sql/src/lib.rs b/nodedb-sql/src/lib.rs index 62fbf1692..0f2ee0ad0 100644 --- a/nodedb-sql/src/lib.rs +++ b/nodedb-sql/src/lib.rs @@ -182,6 +182,9 @@ fn plan_statements( } } + for plan in &plans { + planner::search_scope::refuse_row_scoped_search_functions(plan, &functions)?; + } Ok(plans) } diff --git a/nodedb-sql/src/planner/const_fold.rs b/nodedb-sql/src/planner/const_fold.rs index c06a9fbbb..ec8e8f00e 100644 --- a/nodedb-sql/src/planner/const_fold.rs +++ b/nodedb-sql/src/planner/const_fold.rs @@ -389,6 +389,17 @@ pub fn fold_function_call_scoped( match nodedb_query::functions::eval_function(&name_lower, &folded_args) { Ok(result) => Ok(Some(ndb_to_sql_value(result))), Err(nodedb_query::EvalError::DivisionByZero) => Err(SqlError::DivisionByZero), + Err( + e @ (nodedb_query::EvalError::VectorDimensionMismatch { .. } + | nodedb_query::EvalError::ArgumentType { .. } + | nodedb_query::EvalError::InvalidJsonPath { .. }), + ) => Err(SqlError::DataException { + detail: e.to_string(), + }), + // A registered name some other evaluator owns (a search score): + // not foldable here. Its search plan serves it, or the plan-time + // search-scope pass refuses it. + Err(nodedb_query::EvalError::UnknownFunction { .. }) => Ok(None), } } @@ -473,6 +484,26 @@ mod tests { } } + /// A constant call whose argument the function cannot compute on fails + /// the statement at plan time instead of folding to NULL. + #[test] + fn fold_of_a_malformed_json_path_is_a_data_exception() { + let registry = FunctionRegistry::new(); + let expr = SqlExpr::Function { + name: "doc_get".into(), + args: vec![ + SqlExpr::Literal(SqlValue::String("{\"a\":1}".into())), + SqlExpr::Literal(SqlValue::String("$..a".into())), + ], + distinct: false, + }; + let err = fold_constant(&expr, ®istry).expect_err("malformed path must fail"); + assert!( + matches!(&err, SqlError::DataException { detail } if detail.contains("invalid JSONPath")), + "{err:?}" + ); + } + #[test] fn fold_current_timestamp_produces_timestamptz_once_only() { let registry = FunctionRegistry::new(); diff --git a/nodedb-sql/src/planner/dml_helpers/kv_counter.rs b/nodedb-sql/src/planner/dml_helpers/kv_counter.rs new file mode 100644 index 000000000..9498c7be6 --- /dev/null +++ b/nodedb-sql/src/planner/dml_helpers/kv_counter.rs @@ -0,0 +1,272 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The row a KV counter atomic (`KV_INCR`, `KV_INCR_FLOAT`) creates for an +//! absent key. +//! +//! The KV engine stores the bytes it is handed and does not know the declared +//! columns. So the fresh row is planned here, from the catalog, by the same +//! code a `VALUES` insert runs: `INSERT (key, column) VALUES (key, 0)`. The +//! Control Plane encodes the cells into the template the engine fills in. + +use sqlparser::ast::{self, Expr, Value, ValueWithSpan}; +use sqlparser::tokenizer::Span; + +use super::kv_insert::build_kv_insert_plan; +use super::params::KvInsertParams; +use crate::catalog::SqlCatalog; +use crate::error::{Result, SqlError}; +use crate::types::*; + +/// The kind of number a counter atomic moves. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum KvCounterKind { + /// `KV_INCR` / `KV_DECR`: an integer column. + Integer, + /// `KV_INCR_FLOAT`: an integer or float column, the same columns it moves + /// in an existing row. + Float, +} + +/// The row a counter atomic creates for an absent key. +#[derive(Debug, Clone, PartialEq)] +pub enum KvCounterFreshRow { + /// The collection holds a single `value` column, or declares none: the + /// new value is stored as raw decimal text. + Raw, + /// The collection declares typed columns. + Typed { + /// The declared column the counter moves: the first column of `kind` + /// in name order, never the primary key. An existing row picks its + /// column by the same rule. `None` when there is none. + column: Option, + /// The cells the insert stores other than `column`: DEFAULTs, and the + /// primary key when it is a named column. + cells: Vec<(String, SqlValue)>, + }, +} + +/// Plan the row `INSERT (key, column) VALUES (key, 0)` stores in the KV +/// collection `info`, for a counter of `kind`. +pub fn plan_kv_counter_fresh_row( + info: &CollectionInfo, + key: &str, + kind: KvCounterKind, + catalog: &dyn SqlCatalog, +) -> Result { + if info.engine != EngineType::KeyValue { + return Err(SqlError::Unsupported { + detail: format!( + "KV counter on '{}', which is not a key-value collection", + info.name + ), + }); + } + let pk_col = info.primary_key.as_deref().unwrap_or("key"); + let value_columns: Vec<&ColumnInfo> = info + .columns + .iter() + .filter(|c| c.name != pk_col && c.name != "key" && c.name != "ttl") + .collect(); + if value_columns.is_empty() || (value_columns.len() == 1 && value_columns[0].name == "value") { + return Ok(KvCounterFreshRow::Raw); + } + + let column = value_columns + .iter() + .filter(|c| moves_as(kind, &c.data_type)) + .map(|c| c.name.clone()) + .min(); + let Some(column) = column else { + return Ok(KvCounterFreshRow::Typed { + column: None, + cells: Vec::new(), + }); + }; + + let placeholder = match kind { + KvCounterKind::Integer => "0", + KvCounterKind::Float => "0.0", + }; + let columns = [pk_col.to_string(), column.clone()]; + let row = ast::Parens::with_empty_span(vec![ + literal(Value::SingleQuotedString(key.to_string())), + literal(Value::Number(placeholder.to_string(), false)), + ]); + let plans = build_kv_insert_plan(KvInsertParams { + collection: info.name.clone(), + columns: &columns, + rows_ast: std::slice::from_ref(&row), + intent: KvInsertIntent::Put, + on_conflict_updates: Vec::new(), + pk_col: info.primary_key.as_deref(), + declared_columns: &info.columns, + catalog, + })?; + let cells = match plans.into_iter().next() { + Some(SqlPlan::KvInsert { mut entries, .. }) if entries.len() == 1 => { + let (_, cells) = entries.remove(0); + cells + .into_iter() + .filter(|(name, _)| *name != column) + .collect() + } + _ => { + return Err(SqlError::Unsupported { + detail: "KV counter fresh row did not plan as one KV insert".into(), + }); + } + }; + Ok(KvCounterFreshRow::Typed { + column: Some(column), + cells, + }) +} + +/// Whether a declared column of `data_type` is one a counter of `kind` +/// moves. +fn moves_as(kind: KvCounterKind, data_type: &SqlDataType) -> bool { + match kind { + KvCounterKind::Integer => matches!(data_type, SqlDataType::Int64), + KvCounterKind::Float => matches!(data_type, SqlDataType::Int64 | SqlDataType::Float64), + } +} + +fn literal(value: Value) -> Expr { + Expr::Value(ValueWithSpan { + value, + span: Span::empty(), + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// A catalog with no collections and no sequence state. The DEFAULTs + /// these cases declare are literals, so no accessor is reached. + struct NoCatalog; + + impl SqlCatalog for NoCatalog { + fn get_collection( + &self, + _database_id: nodedb_types::DatabaseId, + _name: &str, + ) -> std::result::Result, crate::catalog::SqlCatalogError> { + Ok(None) + } + } + + fn column(name: &str, data_type: SqlDataType, default: Option<&str>) -> ColumnInfo { + ColumnInfo { + name: name.to_string(), + data_type, + nullable: true, + is_primary_key: false, + default: default.map(str::to_string), + raw_type: None, + int_width: None, + float_width: None, + } + } + + fn kv_info(pk: &str, columns: Vec) -> CollectionInfo { + let mut key = column(pk, SqlDataType::String, None); + key.is_primary_key = true; + let mut all = vec![key]; + all.extend(columns); + CollectionInfo { + name: "c".into(), + engine: EngineType::KeyValue, + columns: all, + primary_key: Some(pk.into()), + has_auto_tier: false, + indexes: Vec::new(), + bitemporal: false, + primary: nodedb_types::PrimaryEngine::Document, + vector_primary: None, + partition_strategy: nodedb_types::PartitionStrategy::CollectionHomed, + open_schema: CollectionInfo::open_schema_for(EngineType::KeyValue), + } + } + + #[test] + fn a_single_value_column_is_raw() { + let info = kv_info("key", vec![column("value", SqlDataType::String, None)]); + let row = plan_kv_counter_fresh_row(&info, "k", KvCounterKind::Integer, &NoCatalog) + .expect("plan"); + assert_eq!(row, KvCounterFreshRow::Raw); + } + + #[test] + fn a_typed_collection_plans_the_insert_row_with_defaults() { + let info = kv_info( + "key", + vec![ + column("n", SqlDataType::Int64, None), + column("status", SqlDataType::String, Some("'new'")), + column("score", SqlDataType::Float64, None), + ], + ); + let row = plan_kv_counter_fresh_row(&info, "k", KvCounterKind::Integer, &NoCatalog) + .expect("plan"); + let KvCounterFreshRow::Typed { column, cells } = row else { + panic!("expected a typed row"); + }; + assert_eq!(column.as_deref(), Some("n")); + assert!( + cells + .iter() + .any(|(name, value)| name == "status" && *value == SqlValue::String("new".into())), + "{cells:?}" + ); + assert!( + cells.iter().all(|(name, _)| name != "n" && name != "key"), + "{cells:?}" + ); + } + + #[test] + fn a_named_primary_key_is_kept_in_the_row() { + let info = kv_info("id", vec![column("n", SqlDataType::Int64, None)]); + let row = plan_kv_counter_fresh_row(&info, "k1", KvCounterKind::Integer, &NoCatalog) + .expect("plan"); + let KvCounterFreshRow::Typed { cells, .. } = row else { + panic!("expected a typed row"); + }; + assert!( + cells + .iter() + .any(|(name, value)| name == "id" && *value == SqlValue::String("k1".into())), + "{cells:?}" + ); + } + + #[test] + fn counters_move_the_first_column_of_their_kind_in_name_order() { + let info = kv_info( + "key", + vec![ + column("b", SqlDataType::Int64, None), + column("a", SqlDataType::Float64, None), + column("label", SqlDataType::String, None), + ], + ); + let target = |kind| match plan_kv_counter_fresh_row(&info, "k", kind, &NoCatalog) { + Ok(KvCounterFreshRow::Typed { column, .. }) => column, + other => panic!("expected a typed row, got {other:?}"), + }; + assert_eq!(target(KvCounterKind::Integer).as_deref(), Some("b")); + assert_eq!(target(KvCounterKind::Float).as_deref(), Some("a")); + + let text_only = kv_info("key", vec![column("label", SqlDataType::String, None)]); + let row = plan_kv_counter_fresh_row(&text_only, "k", KvCounterKind::Integer, &NoCatalog) + .expect("plan"); + assert_eq!( + row, + KvCounterFreshRow::Typed { + column: None, + cells: Vec::new(), + } + ); + } +} diff --git a/nodedb-sql/src/planner/dml_helpers/mod.rs b/nodedb-sql/src/planner/dml_helpers/mod.rs index 853b27b09..99a930cfa 100644 --- a/nodedb-sql/src/planner/dml_helpers/mod.rs +++ b/nodedb-sql/src/planner/dml_helpers/mod.rs @@ -9,6 +9,7 @@ //! - [`vector_primary_insert`] — vector-primary collection insert plans //! - [`vector_primary_dml`] — vector-primary collection update / delete / truncate plans //! - [`kv_insert`] — KV engine insert plans +//! - [`kv_counter`] — the row a KV counter atomic creates for an absent key //! - [`insert_select_bind`] — `INSERT ... SELECT` target-column binding //! - [`params`] — parameter structs for the helpers above @@ -16,6 +17,7 @@ mod ast_extract; mod declared_defaults; mod insert_columns; mod insert_select_bind; +mod kv_counter; mod kv_insert; mod params; mod range_check; @@ -28,6 +30,7 @@ pub(super) use ast_extract::extract_table_name_from_table_with_joins; pub(super) use declared_defaults::materialize_defaults_in_rows; pub(super) use insert_columns::resolve_insert_columns; pub(super) use insert_select_bind::bind_insert_select_columns; +pub use kv_counter::{KvCounterFreshRow, KvCounterKind, plan_kv_counter_fresh_row}; pub(super) use kv_insert::build_kv_insert_plan; pub(super) use params::KvInsertParams; pub(super) use range_check::{ diff --git a/nodedb-sql/src/planner/mod.rs b/nodedb-sql/src/planner/mod.rs index ae80ed820..3a1a72da3 100644 --- a/nodedb-sql/src/planner/mod.rs +++ b/nodedb-sql/src/planner/mod.rs @@ -31,6 +31,7 @@ pub mod lateral; pub mod merge; pub mod predicate_coerce; pub mod returning; +pub mod search_scope; pub mod select; pub use returning::resolve_returning_items; diff --git a/nodedb-sql/src/planner/search_scope.rs b/nodedb-sql/src/planner/search_scope.rs new file mode 100644 index 000000000..76a595276 --- /dev/null +++ b/nodedb-sql/src/planner/search_scope.rs @@ -0,0 +1,608 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Plan-time refusal of index-owned search functions in row-evaluated +//! positions. +//! +//! `bm25_score`, `search_score`, `text_match`, `search`, `rrf_score`, +//! `sparse_score`, `graph_score`, `multi_vector_score` and +//! `multi_vector_search` read a search index. The planner lowers each call it +//! recognises into its search plan, which serves the score as a column. A +//! call the planner could not lower stays in a filter, projection, sort key, +//! assignment or aggregate argument. The row evaluator has no index and no +//! value for it, so this pass refuses the statement at plan time. The +//! refusal does not depend on whether the collection holds rows. +//! +//! A wrapper plan (a subquery tail, aggregate, join or lateral join) over a +//! search plan is not checked for its own expressions: its projection names +//! the score column the search plan serves. + +use crate::error::{Result, SqlError}; +use crate::functions::registry::{FunctionRegistry, SearchTrigger}; +use crate::types::query::{AggregateExpr, Projection, SortKey, WindowSpec}; +use crate::types::{Filter, FilterExpr, MergePlanAction, SqlPlan}; +use crate::types_expr::SqlExpr; + +/// Refuse `plan` when an index-owned search function sits where the row +/// evaluator runs it. +pub fn refuse_row_scoped_search_functions( + plan: &SqlPlan, + functions: &FunctionRegistry, +) -> Result<()> { + Scope { functions }.plan(plan) +} + +struct Scope<'a> { + functions: &'a FunctionRegistry, +} + +impl Scope<'_> { + fn plan(&self, plan: &SqlPlan) -> Result<()> { + match plan { + SqlPlan::Scan { + filters, + projection, + sort_keys, + window_functions, + .. + } + | SqlPlan::DocumentIndexLookup { + filters, + projection, + sort_keys, + window_functions, + .. + } => { + self.filters(filters)?; + self.projection(projection)?; + self.sort_keys(sort_keys)?; + self.windows(window_functions) + } + SqlPlan::PointGet { projection, .. } | SqlPlan::RangeScan { projection, .. } => { + self.projection(projection) + } + SqlPlan::KvInsert { + on_conflict_updates, + .. + } + | SqlPlan::Upsert { + on_conflict_updates, + .. + } + | SqlPlan::VectorPrimaryInsert { + on_conflict_updates, + .. + } => self.assignments(on_conflict_updates), + SqlPlan::InsertSelect { + source, column_map, .. + } => { + self.plan(source)?; + self.assignments(column_map) + } + SqlPlan::Update { + assignments, + filters, + .. + } + | SqlPlan::VectorPrimaryUpdate { + assignments, + filters, + .. + } => { + self.assignments(assignments)?; + self.filters(filters) + } + SqlPlan::UpdateFrom { + source, + assignments, + target_filters, + .. + } => { + self.plan(source)?; + self.assignments(assignments)?; + self.filters(target_filters) + } + SqlPlan::Delete { filters, .. } | SqlPlan::VectorPrimaryDelete { filters, .. } => { + self.filters(filters) + } + SqlPlan::Join { + left, + right, + condition, + projection, + filters, + .. + } => { + self.plan(left)?; + self.plan(right)?; + if has_search_plan(left) || has_search_plan(right) { + return Ok(()); + } + if let Some(condition) = condition { + self.expr(condition)?; + } + self.projection(projection)?; + self.filters(filters) + } + SqlPlan::Aggregate { + input, + group_by, + aggregates, + having, + sort_keys, + .. + } => { + self.plan(input)?; + if has_search_plan(input) { + return Ok(()); + } + self.exprs(group_by)?; + self.aggregates(aggregates)?; + self.filters(having)?; + self.sort_keys(sort_keys) + } + SqlPlan::TimeseriesScan { + aggregates, + filters, + projection, + sort_keys, + .. + } => { + self.aggregates(aggregates)?; + self.filters(filters)?; + self.projection(projection)?; + self.sort_keys(sort_keys) + } + // A search plan serves its own score call as a column; only its + // residual filters run on the row evaluator. + SqlPlan::VectorSearch { filters, .. } | SqlPlan::TextSearch { filters, .. } => { + self.filters(filters) + } + SqlPlan::SpatialScan { + attribute_filters, .. + } => self.filters(attribute_filters), + SqlPlan::RecursiveScan { + base_filters, + recursive_filters, + .. + } => { + self.filters(base_filters)?; + self.filters(recursive_filters) + } + SqlPlan::Union { inputs, .. } => inputs.iter().try_for_each(|input| self.plan(input)), + SqlPlan::Intersect { left, right, .. } | SqlPlan::Except { left, right, .. } => { + self.plan(left)?; + self.plan(right) + } + SqlPlan::Cte { definitions, outer } => { + for (_, definition) in definitions { + self.plan(definition)?; + } + self.plan(outer) + } + SqlPlan::Subquery { + input, + filters, + projection, + window_functions, + sort_keys, + .. + } => { + self.plan(input)?; + if has_search_plan(input) { + return Ok(()); + } + self.filters(filters)?; + self.projection(projection)?; + self.windows(window_functions)?; + self.sort_keys(sort_keys) + } + SqlPlan::Merge { + source, clauses, .. + } => { + self.plan(source)?; + for clause in clauses { + self.filters(&clause.extra_predicate)?; + match &clause.action { + MergePlanAction::Update { assignments } => self.assignments(assignments)?, + MergePlanAction::Insert { values, .. } => self.exprs(values)?, + MergePlanAction::Delete | MergePlanAction::DoNothing => {} + } + } + Ok(()) + } + SqlPlan::LateralTopK { + outer, + inner_filters, + inner_order_by, + projection, + .. + } => { + self.plan(outer)?; + self.filters(inner_filters)?; + self.sort_keys(inner_order_by)?; + if has_search_plan(outer) { + return Ok(()); + } + self.projection(projection) + } + SqlPlan::LateralLoop { + outer, + inner, + projection, + .. + } => { + self.plan(outer)?; + self.plan(inner)?; + if has_search_plan(outer) || has_search_plan(inner) { + return Ok(()); + } + self.projection(projection) + } + // No row-evaluated expression: constants, literal-row writes, + // search plans that carry no residual filter, array statements, + // and DDL. + SqlPlan::ConstantResult { .. } + | SqlPlan::Insert { .. } + | SqlPlan::Truncate { .. } + | SqlPlan::TimeseriesIngest { .. } + | SqlPlan::MultiVectorSearch { .. } + | SqlPlan::SparseSearch { .. } + | SqlPlan::HybridSearch { .. } + | SqlPlan::HybridSearchTriple { .. } + | SqlPlan::RecursiveValue { .. } + | SqlPlan::CreateArray { .. } + | SqlPlan::DropArray { .. } + | SqlPlan::AlterArray { .. } + | SqlPlan::InsertArray { .. } + | SqlPlan::DeleteArray { .. } + | SqlPlan::ArraySlice { .. } + | SqlPlan::ArrayProject { .. } + | SqlPlan::ArrayAgg { .. } + | SqlPlan::ArrayElementwise { .. } + | SqlPlan::ArrayFlush { .. } + | SqlPlan::ArrayCompact { .. } + | SqlPlan::VectorPrimaryTruncate { .. } + | SqlPlan::CreateIndex { .. } + | SqlPlan::DropIndex { .. } => Ok(()), + } + } + + fn filters(&self, filters: &[Filter]) -> Result<()> { + filters + .iter() + .try_for_each(|filter| self.filter(&filter.expr)) + } + + fn filter(&self, expr: &FilterExpr) -> Result<()> { + match expr { + FilterExpr::Expr(expr) => self.expr(expr), + FilterExpr::And(children) | FilterExpr::Or(children) => self.filters(children), + FilterExpr::Not(child) => self.filter(&child.expr), + FilterExpr::Comparison { .. } + | FilterExpr::InList { .. } + | FilterExpr::Between { .. } + | FilterExpr::IsNull { .. } + | FilterExpr::IsNotNull { .. } => Ok(()), + } + } + + fn projection(&self, projection: &[Projection]) -> Result<()> { + for item in projection { + match item { + Projection::Computed { expr, .. } | Projection::CpComputed { expr, .. } => { + self.expr(expr)? + } + Projection::Column(_) | Projection::Star | Projection::QualifiedStar(_) => {} + } + } + Ok(()) + } + + fn sort_keys(&self, sort_keys: &[SortKey]) -> Result<()> { + sort_keys.iter().try_for_each(|key| self.expr(&key.expr)) + } + + fn windows(&self, windows: &[WindowSpec]) -> Result<()> { + for window in windows { + self.exprs(&window.args)?; + self.exprs(&window.partition_by)?; + self.sort_keys(&window.order_by)?; + } + Ok(()) + } + + fn aggregates(&self, aggregates: &[AggregateExpr]) -> Result<()> { + aggregates + .iter() + .try_for_each(|aggregate| self.exprs(&aggregate.args)) + } + + fn assignments(&self, assignments: &[(String, SqlExpr)]) -> Result<()> { + assignments.iter().try_for_each(|(_, expr)| self.expr(expr)) + } + + fn exprs(&self, exprs: &[SqlExpr]) -> Result<()> { + exprs.iter().try_for_each(|expr| self.expr(expr)) + } + + fn expr(&self, expr: &SqlExpr) -> Result<()> { + match first_search_function(expr, self.functions) { + Some(name) => Err(SqlError::SearchFunctionOutsideSearch { + name: name.to_owned(), + }), + None => Ok(()), + } + } +} + +/// Whether the search trigger `trigger` names a function that reads an +/// index and has no per-row value. +fn is_index_owned(trigger: SearchTrigger) -> bool { + match trigger { + SearchTrigger::MultiVectorSearch + | SearchTrigger::SparseSearch + | SearchTrigger::TextSearch + | SearchTrigger::HybridSearch + | SearchTrigger::TextMatch + | SearchTrigger::GraphSearch => true, + // The vector distances and the spatial predicates evaluate per row; + // the time bucket is a scalar; the array functions are table-valued + // and planned from FROM. + SearchTrigger::None + | SearchTrigger::VectorSearch + | SearchTrigger::SpatialDWithin + | SearchTrigger::SpatialContains + | SearchTrigger::SpatialIntersects + | SearchTrigger::SpatialWithin + | SearchTrigger::TimeBucket + | SearchTrigger::ArraySlice + | SearchTrigger::ArrayProject + | SearchTrigger::ArrayAgg + | SearchTrigger::ArrayElementwise + | SearchTrigger::ArrayFlush + | SearchTrigger::ArrayCompact => false, + } +} + +/// The first index-owned search function `expr` calls outside a subquery. +fn first_search_function<'e>(expr: &'e SqlExpr, functions: &FunctionRegistry) -> Option<&'e str> { + let find = |e: &'e SqlExpr| first_search_function(e, functions); + match expr { + SqlExpr::Function { name, args, .. } => { + if is_index_owned(functions.search_trigger(name)) { + return Some(name.as_str()); + } + args.iter().find_map(find) + } + SqlExpr::BinaryOp { left, right, .. } => find(left).or_else(|| find(right)), + SqlExpr::UnaryOp { expr, .. } + | SqlExpr::Cast { expr, .. } + | SqlExpr::IsNull { expr, .. } => find(expr), + SqlExpr::Case { + operand, + when_then, + else_expr, + } => operand + .as_deref() + .and_then(find) + .or_else(|| { + when_then + .iter() + .find_map(|(when, then)| find(when).or_else(|| find(then))) + }) + .or_else(|| else_expr.as_deref().and_then(find)), + SqlExpr::InList { expr, list, .. } => find(expr).or_else(|| list.iter().find_map(find)), + SqlExpr::Between { + expr, low, high, .. + } => find(expr).or_else(|| find(low)).or_else(|| find(high)), + SqlExpr::Like { expr, pattern, .. } => find(expr).or_else(|| find(pattern)), + SqlExpr::ArrayLiteral(items) => items.iter().find_map(find), + SqlExpr::Column { .. } | SqlExpr::Literal(_) | SqlExpr::Subquery(_) | SqlExpr::Wildcard => { + None + } + } +} + +/// Whether `plan` is, or wraps, a search plan that serves a score column. +fn has_search_plan(plan: &SqlPlan) -> bool { + match plan { + SqlPlan::VectorSearch { .. } + | SqlPlan::MultiVectorSearch { .. } + | SqlPlan::SparseSearch { .. } + | SqlPlan::TextSearch { .. } + | SqlPlan::HybridSearch { .. } + | SqlPlan::HybridSearchTriple { .. } => true, + SqlPlan::Subquery { input, .. } | SqlPlan::Aggregate { input, .. } => { + has_search_plan(input) + } + SqlPlan::Join { left, right, .. } => has_search_plan(left) || has_search_plan(right), + SqlPlan::LateralTopK { outer, .. } => has_search_plan(outer), + SqlPlan::LateralLoop { outer, inner, .. } => { + has_search_plan(outer) || has_search_plan(inner) + } + SqlPlan::Union { inputs, .. } => inputs.iter().any(has_search_plan), + SqlPlan::Intersect { left, right, .. } | SqlPlan::Except { left, right, .. } => { + has_search_plan(left) || has_search_plan(right) + } + SqlPlan::Cte { outer, .. } => has_search_plan(outer), + SqlPlan::ConstantResult { .. } + | SqlPlan::Scan { .. } + | SqlPlan::PointGet { .. } + | SqlPlan::DocumentIndexLookup { .. } + | SqlPlan::RangeScan { .. } + | SqlPlan::Insert { .. } + | SqlPlan::KvInsert { .. } + | SqlPlan::Upsert { .. } + | SqlPlan::InsertSelect { .. } + | SqlPlan::Update { .. } + | SqlPlan::UpdateFrom { .. } + | SqlPlan::Delete { .. } + | SqlPlan::Truncate { .. } + | SqlPlan::TimeseriesScan { .. } + | SqlPlan::TimeseriesIngest { .. } + | SqlPlan::SpatialScan { .. } + | SqlPlan::RecursiveScan { .. } + | SqlPlan::RecursiveValue { .. } + | SqlPlan::CreateArray { .. } + | SqlPlan::DropArray { .. } + | SqlPlan::AlterArray { .. } + | SqlPlan::InsertArray { .. } + | SqlPlan::DeleteArray { .. } + | SqlPlan::ArraySlice { .. } + | SqlPlan::ArrayProject { .. } + | SqlPlan::ArrayAgg { .. } + | SqlPlan::ArrayElementwise { .. } + | SqlPlan::ArrayFlush { .. } + | SqlPlan::ArrayCompact { .. } + | SqlPlan::Merge { .. } + | SqlPlan::VectorPrimaryInsert { .. } + | SqlPlan::VectorPrimaryDelete { .. } + | SqlPlan::VectorPrimaryTruncate { .. } + | SqlPlan::VectorPrimaryUpdate { .. } + | SqlPlan::CreateIndex { .. } + | SqlPlan::DropIndex { .. } => false, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::types::query::EngineType; + use crate::types_expr::SqlValue; + + fn call(name: &str) -> SqlExpr { + SqlExpr::Function { + name: name.into(), + args: vec![ + SqlExpr::Column { + table: None, + name: "body".into(), + }, + SqlExpr::Literal(SqlValue::String("rust".into())), + ], + distinct: false, + } + } + + fn scan(projection: Vec, filters: Vec) -> SqlPlan { + SqlPlan::Scan { + collection: "docs".into(), + alias: None, + engine: EngineType::DocumentSchemaless, + filters, + projection, + sort_keys: Vec::new(), + limit: None, + offset: 0, + distinct: false, + window_functions: Vec::new(), + temporal: Default::default(), + } + } + + fn check(plan: &SqlPlan) -> Result<()> { + refuse_row_scoped_search_functions(plan, &FunctionRegistry::new()) + } + + #[test] + fn a_score_in_a_scan_projection_is_refused() { + let plan = scan( + vec![Projection::Computed { + expr: call("bm25_score"), + alias: "s".into(), + }], + Vec::new(), + ); + assert_eq!( + check(&plan), + Err(SqlError::SearchFunctionOutsideSearch { + name: "bm25_score".into() + }) + ); + } + + #[test] + fn a_match_nested_in_a_scan_filter_is_refused() { + let nested = SqlExpr::BinaryOp { + left: Box::new(call("text_match")), + op: crate::types_expr::BinaryOp::Or, + right: Box::new(SqlExpr::Literal(SqlValue::Bool(false))), + }; + let plan = scan( + Vec::new(), + vec![Filter { + expr: FilterExpr::Expr(nested), + }], + ); + assert!(matches!( + check(&plan), + Err(SqlError::SearchFunctionOutsideSearch { .. }) + )); + } + + #[test] + fn row_scalars_in_a_scan_pass() { + let plan = scan( + vec![ + Projection::Computed { + expr: call("vector_distance"), + alias: "d".into(), + }, + Projection::Computed { + expr: call("doc_get"), + alias: "g".into(), + }, + ], + Vec::new(), + ); + assert_eq!(check(&plan), Ok(())); + } + + #[test] + fn a_search_plan_projection_serves_its_score() { + let plan = SqlPlan::TextSearch { + collection: "docs".into(), + query: crate::fts_types::FtsQuery::Plain { + text: "rust".into(), + fuzzy: true, + }, + top_k: 10, + filters: Vec::new(), + score_alias: Some("s".into()), + projection: vec![Projection::Computed { + expr: call("bm25_score"), + alias: "s".into(), + }], + }; + assert_eq!(check(&plan), Ok(())); + } + + #[test] + fn a_subquery_tail_over_a_search_plan_is_not_checked() { + let search = SqlPlan::TextSearch { + collection: "docs".into(), + query: crate::fts_types::FtsQuery::Plain { + text: "rust".into(), + fuzzy: true, + }, + top_k: 10, + filters: Vec::new(), + score_alias: Some("s".into()), + projection: Vec::new(), + }; + let plan = SqlPlan::Subquery { + input: Box::new(search), + filters: Vec::new(), + projection: vec![Projection::Computed { + expr: call("bm25_score"), + alias: "s".into(), + }], + window_functions: Vec::new(), + sort_keys: Vec::new(), + offset: 0, + distinct: false, + limit: None, + }; + assert_eq!(check(&plan), Ok(())); + } +} diff --git a/nodedb-sql/src/planner/select/order_by/projection.rs b/nodedb-sql/src/planner/select/order_by/projection.rs index 9ed17389b..9b99f3fb3 100644 --- a/nodedb-sql/src/planner/select/order_by/projection.rs +++ b/nodedb-sql/src/planner/select/order_by/projection.rs @@ -7,8 +7,8 @@ //! `bm25_score(...)` call may still appear directly in the SELECT projection. //! The canonical shape `SELECT id, rrf_score(...) AS score FROM c WHERE ... LIMIT N` //! requires this entry path because there is no ORDER BY clause to inspect. -//! Without it the score column resolves to NULL via scalar evaluation that -//! has no implementation. +//! A score call no search plan serves is refused at plan time +//! (`planner::search_scope`): it has no per-row value. //! //! Text-search shape: `SELECT id, bm25_score(field, term) FROM c ORDER BY id`. //! The plan stays a Scan after ORDER BY (non-search sort key). This pass diff --git a/nodedb-test-support/Cargo.toml b/nodedb-test-support/Cargo.toml index 716aec860..c93ed50f0 100644 --- a/nodedb-test-support/Cargo.toml +++ b/nodedb-test-support/Cargo.toml @@ -24,6 +24,7 @@ nodedb-types = { workspace = true } nodedb-wal = { workspace = true } # Runtime + wire deps used by the harness itself. +async-trait = { workspace = true } tokio = { workspace = true, features = ["test-util"] } tokio-postgres = { workspace = true } tokio-tungstenite = { workspace = true } diff --git a/nodedb-test-support/src/cluster_harness/cluster/bringup.rs b/nodedb-test-support/src/cluster_harness/cluster/bringup.rs index a29a8212a..cda7141b3 100644 --- a/nodedb-test-support/src/cluster_harness/cluster/bringup.rs +++ b/nodedb-test-support/src/cluster_harness/cluster/bringup.rs @@ -1,9 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 -//! The shared 3-node bringup body (`spawn_three_inner`) and its -//! post-join convergence barriers: topology size, rolling-upgrade -//! compat-mode exit, metadata-group leader stability, and per-group -//! Raft leader stability. +//! The shared 3-node bringup body (`spawn_three_inner`). The post-join +//! convergence barriers live in `ready`. use std::time::Duration; @@ -12,7 +10,6 @@ use nodedb_types::config::tuning::ClusterTransportTuning; use super::TestCluster; use super::types::ClusterSpawnConfig; use crate::cluster_harness::node::TestClusterNode; -use crate::cluster_harness::wait::wait_for; impl TestCluster { /// Shared 3-node spawn body. Threads an optional Raft @@ -74,152 +71,7 @@ impl TestCluster { spawn_config: config, }; - wait_for( - "all 3 nodes report topology_size == 3", - Duration::from_secs(30), - Duration::from_millis(50), - || cluster.nodes.iter().all(|n| n.topology_size() == 3), - ) - .await; - - // CRITICAL: wait for every node to exit rolling-upgrade - // compat mode before letting the test issue any DDL. - // - // `metadata_proposer::propose_catalog_entry` consults - // `cluster_version_view().can_activate_feature(DISTRIBUTED_CATALOG_VERSION)` - // and, while even one node still reports a lower wire - // version, returns `Ok(0)` without going through the raft - // group. The pgwire DDL handlers (CREATE USER, etc.) then - // fall through to a LEGACY path that writes the record - // directly on the proposing node — **with zero - // replication** to followers. Any subsequent - // `has_active_user` check on a follower returns false and - // the test flakes. - // - // Topology has three members the moment the join request - // completes, but the `wire_version` field on each node's - // topology entry is updated asynchronously by the gossip - // path. That's why `topology_size == 3` converges fast yet - // `can_activate_feature(...)` can still be false for - // several hundred milliseconds afterwards. Waiting here - // closes the window deterministically — no retries, no - // flakes, no compat-mode fallback silently breaking - // replication. - wait_for( - "all 3 nodes exit rolling-upgrade compat mode", - Duration::from_secs(30), - Duration::from_millis(20), - || { - cluster.nodes.iter().all(|n| { - n.shared.cluster_version_view().can_activate_feature( - nodedb::control::rolling_upgrade::DISTRIBUTED_CATALOG_VERSION, - ) - }) - }, - ) - .await; - - // CRITICAL: wait for the metadata Raft group to elect a leader - // and for every node's local view to agree on the same leader id. - // - // Topology convergence + rolling-upgrade exit only guarantees - // membership and wire version are agreed; they say nothing about - // election state. Under heavy host load (e.g. running this test - // immediately after another full-suite cluster test exits and - // the unit-test pool ramps back up), the initial Raft heartbeat - // window can be missed and the first `acquire`/`propose` issued - // by the test races a re-election — surfacing as - // `raft error: not leader (leader hint: None)` from a - // descriptor-lease or DDL call. - // - // Waiting until every node reports the same non-zero leader id - // closes the window deterministically. Symmetric to the - // rolling-upgrade wait above: no retries, no flakes, no - // wasted CI minutes on cleanup of a doomed cluster bringup. - wait_for( - "metadata group has stable leader visible on every node", - Duration::from_secs(30), - Duration::from_millis(20), - || { - let leaders: Vec = cluster - .nodes - .iter() - .map(|n| n.metadata_group_leader()) - .collect(); - let first = leaders[0]; - first != 0 && leaders.iter().all(|&l| l == first) - }, - ) - .await; - - // CRITICAL: wait for EVERY data Raft group to elect a stable - // leader visible on every node. Without this barrier, the - // first data-group write after `spawn_three()` returns can - // race a still-electing group: - // - // 1. Proposer's local `propose()` runs on a node that thinks - // it's leader (stale routing-table hint), gets an Ok back - // with a `log_index` that was never actually committed. - // 2. `ProposeTracker::register((group_id, log_index))`. - // 3. Some unrelated entry that *does* commit at that index - // (e.g., a leadership-change no-op) fires `tracker.complete`, - // waking the waiter with `Ok([])` even though the user's - // `INSERT` row was never replicated. - // 4. `simple_query` returns success; the row is permanently - // lost. - // - // The metadata-group-only wait above is insufficient because - // data groups elect independently and lag the metadata group - // by hundreds of milliseconds under load. Waiting until every - // group on every node reports a non-zero leader closes the - // window deterministically. - wait_for( - "every Raft group has a stable leader visible on every node", - Duration::from_secs(30), - Duration::from_millis(20), - || { - // Snapshot every node's per-group leader view. A group - // is "ready" iff every node reports the same non-zero - // leader for it. - let per_node: Vec> = cluster - .nodes - .iter() - .map(|n| n.all_group_leaders()) - .collect(); - if per_node.iter().any(|v| v.is_empty()) { - return false; - } - // The Calvin sequencer group is an internal Raft group that is - // not part of the data/metadata routing topology. Cluster - // readiness for data operations does not depend on it, and its - // leader is surfaced to the observer on a slower/independent path - // than the routing groups — so gating general cluster startup on - // it makes every test (Calvin or not) flake when the sequencer - // group's observed leader lags. Calvin tests gate on the - // sequencer separately (`wait_for_sequencer_leader`). Exclude it - // from the general readiness gate. - let group_ids: std::collections::BTreeSet = per_node - .iter() - .flat_map(|v| v.iter().map(|(gid, _)| *gid)) - .filter(|gid| *gid != nodedb_cluster::calvin::SEQUENCER_GROUP_ID) - .collect(); - if group_ids.is_empty() { - return false; - } - group_ids.iter().all(|gid| { - let leaders: Vec = per_node - .iter() - .filter_map(|v| v.iter().find(|(g, _)| g == gid).map(|(_, l)| *l)) - .collect(); - if leaders.len() != per_node.len() { - return false; - } - let first = leaders[0]; - first != 0 && leaders.iter().all(|&l| l == first) - }) - }, - ) - .await; + cluster.await_ready().await; Ok(cluster) } diff --git a/nodedb-test-support/src/cluster_harness/cluster/mod.rs b/nodedb-test-support/src/cluster_harness/cluster/mod.rs index 899e9ade0..dcafb7432 100644 --- a/nodedb-test-support/src/cluster_harness/cluster/mod.rs +++ b/nodedb-test-support/src/cluster_harness/cluster/mod.rs @@ -4,11 +4,14 @@ //! //! [`TestCluster`] + [`ClusterSpawnConfig`] type definitions live in //! [`types`]; spawn-variant convenience wrappers in [`spawn_variants`]; -//! the heavy bringup/convergence body in [`bringup`]; post-spawn -//! membership + DDL helpers in [`membership`]. +//! the bringup body in [`bringup`]; the convergence barriers in [`ready`]; +//! the in-place restart in [`restart`]; post-spawn membership + DDL helpers +//! in [`membership`]. mod bringup; mod membership; +mod ready; +mod restart; mod spawn_variants; mod types; diff --git a/nodedb-test-support/src/cluster_harness/cluster/ready.rs b/nodedb-test-support/src/cluster_harness/cluster/ready.rs new file mode 100644 index 000000000..0dca12245 --- /dev/null +++ b/nodedb-test-support/src/cluster_harness/cluster/ready.rs @@ -0,0 +1,162 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The convergence barriers a cluster passes before a test issues anything: +//! topology size, rolling-upgrade compat-mode exit, metadata-group leader +//! stability, and per-group Raft leader stability. A fresh bringup and an +//! in-place restart both wait here. + +use std::time::Duration; + +use super::TestCluster; +use crate::cluster_harness::wait::wait_for; + +impl TestCluster { + /// Wait until every node agrees on the topology, every node left compat + /// mode, and every Raft group has one leader every node sees. + pub(super) async fn await_ready(&self) { + let node_count = self.nodes.len(); + wait_for( + "every node reports the full topology", + Duration::from_secs(30), + Duration::from_millis(50), + || self.nodes.iter().all(|n| n.topology_size() == node_count), + ) + .await; + + // CRITICAL: wait for every node to exit rolling-upgrade + // compat mode before letting the test issue any DDL. + // + // `metadata_proposer::propose_catalog_entry` consults + // `cluster_version_view().can_activate_feature(DISTRIBUTED_CATALOG_VERSION)` + // and, while even one node still reports a lower wire + // version, returns `Ok(0)` without going through the raft + // group. The pgwire DDL handlers (CREATE USER, etc.) then + // fall through to a LEGACY path that writes the record + // directly on the proposing node — **with zero + // replication** to followers. Any subsequent + // `has_active_user` check on a follower returns false and + // the test flakes. + // + // Topology has three members the moment the join request + // completes, but the `wire_version` field on each node's + // topology entry is updated asynchronously by the gossip + // path. That's why `topology_size == 3` converges fast yet + // `can_activate_feature(...)` can still be false for + // several hundred milliseconds afterwards. Waiting here + // closes the window deterministically — no retries, no + // flakes, no compat-mode fallback silently breaking + // replication. + wait_for( + "every node exits rolling-upgrade compat mode", + Duration::from_secs(30), + Duration::from_millis(20), + || { + self.nodes.iter().all(|n| { + n.shared.cluster_version_view().can_activate_feature( + nodedb::control::rolling_upgrade::DISTRIBUTED_CATALOG_VERSION, + ) + }) + }, + ) + .await; + + // CRITICAL: wait for the metadata Raft group to elect a leader + // and for every node's local view to agree on the same leader id. + // + // Topology convergence + rolling-upgrade exit only guarantees + // membership and wire version are agreed; they say nothing about + // election state. Under heavy host load (e.g. running this test + // immediately after another full-suite cluster test exits and + // the unit-test pool ramps back up), the initial Raft heartbeat + // window can be missed and the first `acquire`/`propose` issued + // by the test races a re-election — surfacing as + // `raft error: not leader (leader hint: None)` from a + // descriptor-lease or DDL call. + // + // Waiting until every node reports the same non-zero leader id + // closes the window deterministically. Symmetric to the + // rolling-upgrade wait above: no retries, no flakes, no + // wasted CI minutes on cleanup of a doomed cluster bringup. + wait_for( + "metadata group has stable leader visible on every node", + Duration::from_secs(30), + Duration::from_millis(20), + || { + let leaders: Vec = self + .nodes + .iter() + .map(|n| n.metadata_group_leader()) + .collect(); + let first = leaders[0]; + first != 0 && leaders.iter().all(|&l| l == first) + }, + ) + .await; + + // CRITICAL: wait for EVERY data Raft group to elect a stable + // leader visible on every node. Without this barrier, the + // first data-group write after `spawn_three()` returns can + // race a still-electing group: + // + // 1. Proposer's local `propose()` runs on a node that thinks + // it's leader (stale routing-table hint), gets an Ok back + // with a `log_index` that was never actually committed. + // 2. `ProposeTracker::register((group_id, log_index))`. + // 3. Some unrelated entry that *does* commit at that index + // (e.g., a leadership-change no-op) fires `tracker.complete`, + // waking the waiter with `Ok([])` even though the user's + // `INSERT` row was never replicated. + // 4. `simple_query` returns success; the row is permanently + // lost. + // + // The metadata-group-only wait above is insufficient because + // data groups elect independently and lag the metadata group + // by hundreds of milliseconds under load. Waiting until every + // group on every node reports a non-zero leader closes the + // window deterministically. + wait_for( + "every Raft group has a stable leader visible on every node", + Duration::from_secs(30), + Duration::from_millis(20), + || { + // Snapshot every node's per-group leader view. A group + // is "ready" iff every node reports the same non-zero + // leader for it. + let per_node: Vec> = + self.nodes.iter().map(|n| n.all_group_leaders()).collect(); + if per_node.iter().any(|v| v.is_empty()) { + return false; + } + // The Calvin sequencer group is an internal Raft group that is + // not part of the data/metadata routing topology. Cluster + // readiness for data operations does not depend on it, and its + // leader is surfaced to the observer on a slower/independent path + // than the routing groups — so gating general cluster startup on + // it makes every test (Calvin or not) flake when the sequencer + // group's observed leader lags. Calvin tests gate on the + // sequencer separately (`wait_for_sequencer_leader`). Exclude it + // from the general readiness gate. + let group_ids: std::collections::BTreeSet = per_node + .iter() + .flat_map(|v| v.iter().map(|(gid, _)| *gid)) + .filter(|gid| *gid != nodedb_cluster::calvin::SEQUENCER_GROUP_ID) + .collect(); + if group_ids.is_empty() { + return false; + } + group_ids.iter().all(|gid| { + let leaders: Vec = per_node + .iter() + .filter_map(|v| v.iter().find(|(g, _)| g == gid).map(|(_, l)| *l)) + .collect(); + if leaders.len() != per_node.len() { + return false; + } + let first = leaders[0]; + first != 0 && leaders.iter().all(|&l| l == first) + }) + }, + ) + .await; + } +} diff --git a/nodedb-test-support/src/cluster_harness/cluster/restart.rs b/nodedb-test-support/src/cluster_harness/cluster/restart.rs new file mode 100644 index 000000000..a46d6ea2e --- /dev/null +++ b/nodedb-test-support/src/cluster_harness/cluster/restart.rs @@ -0,0 +1,39 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Restart every node of a [`TestCluster`] in place. + +use super::TestCluster; +use crate::cluster_harness::node::TestClusterNode; + +impl TestCluster { + /// Stop every node, then bring every one back on its node id, listen + /// address and data directory, and wait until the cluster is ready. + /// + /// Every node stops before any restarts, so nothing survives in memory: + /// what each node serves afterwards comes from its own disk. The nodes + /// restart together, since each Raft group needs a quorum to elect. + pub async fn restart_all(self) -> Result> { + let TestCluster { + nodes, + spawn_config, + } = self; + let mut stopped = Vec::with_capacity(nodes.len()); + for node in nodes { + stopped.push(node.stop_for_restart().await?); + } + let seeds: Vec = + stopped.iter().map(|node| node.listen_addr()).collect(); + let nodes = futures::future::try_join_all( + stopped + .into_iter() + .map(|node| TestClusterNode::restart(node, seeds.clone(), &spawn_config)), + ) + .await?; + let cluster = TestCluster { + nodes, + spawn_config, + }; + cluster.await_ready().await; + Ok(cluster) + } +} diff --git a/nodedb-test-support/src/cluster_harness/cluster/spawn_variants.rs b/nodedb-test-support/src/cluster_harness/cluster/spawn_variants.rs index 6e9dd08e4..f67106008 100644 --- a/nodedb-test-support/src/cluster_harness/cluster/spawn_variants.rs +++ b/nodedb-test-support/src/cluster_harness/cluster/spawn_variants.rs @@ -165,6 +165,26 @@ impl TestCluster { .await } + /// Spawn a 3-node cluster whose data groups each place + /// `replication_factor` of the three nodes. With a factor below 3 some + /// node replicates no copy of a group, so a test can act on a node that + /// never applied the group's writes. + /// + /// Uses the standard fast-election tuning and 1 Data-Plane core per node. + pub async fn spawn_three_with_replication_factor( + replication_factor: usize, + ) -> Result> { + Self::spawn_three_inner( + fast_cluster_tuning(), + nodedb_types::config::tuning::GraphTuning::default(), + nodedb_types::config::tuning::QueryTuning::default(), + 1, + None, + replication_factor, + ) + .await + } + /// Spawn a 3-node cluster with custom cluster-transport, graph engine tuning, /// query execution tuning, and a specific core count per node. /// diff --git a/nodedb-test-support/src/cluster_harness/node/inspect/crdt.rs b/nodedb-test-support/src/cluster_harness/node/inspect/crdt.rs index e633b3130..0c2519152 100644 --- a/nodedb-test-support/src/cluster_harness/node/inspect/crdt.rs +++ b/nodedb-test-support/src/cluster_harness/node/inspect/crdt.rs @@ -34,8 +34,7 @@ impl TestClusterNode { let request_id = RequestId::new(HARNESS_REQUEST_ID.fetch_add(1, Ordering::Relaxed)); let vshard_id = VShardId::new(nodedb_cluster::routing::vshard_for_collection( - DatabaseId::DEFAULT, - collection, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection), )); let request = Request { request_id, @@ -107,8 +106,7 @@ impl TestClusterNode { let request_id = RequestId::new(HARNESS_REQUEST_ID.fetch_add(1, Ordering::Relaxed)); let vshard_id = VShardId::new(nodedb_cluster::routing::vshard_for_collection( - DatabaseId::DEFAULT, - collection, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection), )); let request = Request { request_id, diff --git a/nodedb-test-support/src/cluster_harness/node/inspect/snapshot.rs b/nodedb-test-support/src/cluster_harness/node/inspect/snapshot.rs index 2d8878e84..412d5d6a7 100644 --- a/nodedb-test-support/src/cluster_harness/node/inspect/snapshot.rs +++ b/nodedb-test-support/src/cluster_harness/node/inspect/snapshot.rs @@ -36,7 +36,9 @@ impl TestClusterNode { /// error or non-Ok response. pub async fn create_tenant_snapshot(&self, tenant: TenantId) -> Vec { let request_id = RequestId::new(SNAPSHOT_REQUEST_ID.fetch_add(1, Ordering::Relaxed)); - let vshard_id = VShardId::new(vshard_for_collection(DatabaseId::DEFAULT, "__system")); + let vshard_id = VShardId::new(vshard_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "__system"), + )); let request = Request { request_id, tenant_id: tenant, @@ -44,6 +46,7 @@ impl TestClusterNode { vshard_id, plan: PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { tenant_id: tenant.as_u64(), + cut_watermark: None, }), deadline: std::time::Instant::now() + std::time::Duration::from_secs(5), priority: Priority::Normal, @@ -97,7 +100,9 @@ impl TestClusterNode { /// `Ok`. pub async fn restore_tenant_snapshot(&self, snapshot_bytes: Vec) -> bool { let request_id = RequestId::new(SNAPSHOT_REQUEST_ID.fetch_add(1, Ordering::Relaxed)); - let vshard_id = VShardId::new(vshard_for_collection(DatabaseId::DEFAULT, "__system")); + let vshard_id = VShardId::new(vshard_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "__system"), + )); let request = Request { request_id, tenant_id: TenantId::new(0), diff --git a/nodedb-test-support/src/cluster_harness/node/inspect/topology.rs b/nodedb-test-support/src/cluster_harness/node/inspect/topology.rs index 1a155b0e1..f01c71164 100644 --- a/nodedb-test-support/src/cluster_harness/node/inspect/topology.rs +++ b/nodedb-test-support/src/cluster_harness/node/inspect/topology.rs @@ -145,6 +145,27 @@ impl TestClusterNode { .any(|g| g.group_id == group_id) } + /// True iff this node is a current replica of data group `group_id`: it + /// has the group mounted, and its own routing table lists it as a voter or + /// learner of the group. + /// + /// [`Self::hosts_data_group`] alone does not answer this. A node removed + /// from a group keeps its mounted replica: a join adds the joiner as a + /// learner of every group, and placement convergence later removes the + /// nodes outside the group's placement from its membership only. + pub fn replicates_data_group(&self, group_id: u64) -> bool { + if !self.hosts_data_group(group_id) { + return false; + } + let Some(routing) = self.shared.cluster_routing.as_ref() else { + return false; + }; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + routing.group_info(group_id).is_some_and(|info| { + info.members.contains(&self.node_id) || info.learners.contains(&self.node_id) + }) + } + /// The local `snapshot_index` for `group_id` from this node's own Raft /// state, or `0` if the group isn't hosted here. /// @@ -200,8 +221,9 @@ impl TestClusterNode { /// group mapping yet (e.g. before the CREATE COLLECTION DDL has /// propagated). pub fn group_id_for_collection(&self, collection: &str) -> Option { - let vshard = - nodedb_cluster::routing::vshard_for_collection(DatabaseId::DEFAULT, collection); + let vshard = nodedb_cluster::routing::vshard_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection), + ); self.shared .cluster_routing .as_ref()? diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/mod.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/mod.rs index 50cce7a13..dc0df838d 100644 --- a/nodedb-test-support/src/cluster_harness/node/lifecycle/mod.rs +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/mod.rs @@ -27,9 +27,11 @@ //! //! Struct definition in [`types`]; thin `spawn*` convenience wrappers in //! [`spawn_variants`]; the full spawn body in [`spawn_full`]; query -//! execution + shutdown + `Drop` teardown in [`teardown`]. +//! execution + shutdown + `Drop` teardown in [`teardown`]; the in-place +//! restart in [`restart`]. mod client_slot; +mod restart; mod spawn_full; mod spawn_variants; mod teardown; diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/restart.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/restart.rs new file mode 100644 index 000000000..8893a4fba --- /dev/null +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/restart.rs @@ -0,0 +1,102 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! In-place restart of a [`TestClusterNode`]: the same node id, the same QUIC +//! listen address, and the same data directory. + +use std::net::SocketAddr; +use std::time::Duration; + +use crate::cluster_harness::cluster::ClusterSpawnConfig; + +use super::types::{DataDir, TestClusterNode}; + +/// A node stopped for an in-place restart. It owns the node's data +/// directory, which outlives the stopped node. +pub(crate) struct StoppedNode { + node_id: u64, + listen_addr: SocketAddr, + data_dir: tempfile::TempDir, +} + +impl StoppedNode { + /// The QUIC address the node listened on, which its peers still hold. + pub(crate) fn listen_addr(&self) -> SocketAddr { + self.listen_addr + } +} + +impl TestClusterNode { + /// Stop the node, flush its WAL and release every file handle, keeping + /// its data directory for [`Self::restart`]. + pub(crate) async fn stop_for_restart( + mut self, + ) -> Result> { + let DataDir::Owned(data_dir) = std::mem::replace(&mut self._data_dir, DataDir::Borrowed) + else { + return Err(format!( + "node {} runs on a caller-supplied data directory, which an in-place \ + restart does not own", + self.node_id + ) + .into()); + }; + let (node_id, listen_addr) = (self.node_id, self.listen_addr); + self.graceful_shutdown_wal_only().await; + await_port_released(node_id, listen_addr).await?; + Ok(StoppedNode { + node_id, + listen_addr, + data_dir, + }) + } + + /// Bring `stopped` back on its node id, listen address and data. Its + /// persisted cluster state makes it rejoin as the same member. + pub(crate) async fn restart( + stopped: StoppedNode, + seed_nodes: Vec, + config: &ClusterSpawnConfig, + ) -> Result> { + let StoppedNode { + node_id, + listen_addr, + data_dir, + } = stopped; + let mut node = Self::spawn_with_full_config_at( + node_id, + seed_nodes, + config, + Some(data_dir.path().to_path_buf()), + Some(listen_addr), + ) + .await?; + node._data_dir = DataDir::Owned(data_dir); + Ok(node) + } +} + +/// Wait until `listen_addr`'s UDP port is free: the stopped node's QUIC +/// endpoint releases its socket once its last handle dropped, after the +/// close drained every connection. +async fn await_port_released( + node_id: u64, + listen_addr: SocketAddr, +) -> Result<(), Box> { + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + loop { + match tokio::net::UdpSocket::bind(listen_addr).await { + Ok(probe) => { + drop(probe); + return Ok(()); + } + Err(error) if tokio::time::Instant::now() >= deadline => { + return Err(format!( + "node {node_id} stopped, but its QUIC port {listen_addr} is still bound \ + after 10s ({error}): a task still holds the node's transport" + ) + .into()); + } + Err(_) => tokio::time::sleep(Duration::from_millis(20)).await, + } + } +} diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs index 64cb89d60..5c0498945 100644 --- a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs @@ -11,7 +11,7 @@ use std::path::PathBuf; use std::sync::Arc; use std::time::Duration; -use nodedb::bridge::dispatch::Dispatcher; +use nodedb::bridge::dispatch::{DATA_PLANE_QUEUE_CAPACITY, Dispatcher}; use nodedb::config::auth::AuthMode; use nodedb::config::server::ClusterSettings; use nodedb::control::server::pgwire::listener::PgListener; @@ -33,7 +33,7 @@ impl TestClusterNode { seed_nodes: Vec, config: &ClusterSpawnConfig, ) -> Result> { - Self::spawn_with_full_config_at(node_id, seed_nodes, config, None).await + Self::spawn_with_full_config_at(node_id, seed_nodes, config, None, None).await } /// Lowest-level cluster-node spawn. In addition to the tuning knobs of @@ -67,11 +67,16 @@ impl TestClusterNode { /// before this parameter existed); on a reopened directory it rebuilds /// in-memory-only structures (e.g. the vector HNSW index) from the /// persisted `TransactionRedo` / `Put` / etc. records. + /// + /// `listen_override`: `None` binds the QUIC transport on an ephemeral + /// port. `Some(addr)` binds it on `addr`, the address a restarted node's + /// peers already hold for it. pub(crate) async fn spawn_with_full_config_at( node_id: u64, seed_nodes: Vec, config: &ClusterSpawnConfig, data_dir_path_override: Option, + listen_override: Option, ) -> Result> { // Every cluster node funnels through here, so installing tracing at // this one point means no test has to opt in to see server-side logs. @@ -101,7 +106,7 @@ impl TestClusterNode { )?); let wal_records: Arc<[nodedb_wal::WalRecord]> = Arc::from(wal.replay()?.into_boxed_slice()); let replay_tombstones = nodedb_wal::extract_tombstones(&wal_records).unwrap(); - let (dispatcher, data_sides) = Dispatcher::new(num_cores, 1024); + let (dispatcher, data_sides) = Dispatcher::new(num_cores, DATA_PLANE_QUEUE_CAPACITY); let (event_producers, event_consumers) = create_event_bus(num_cores); // Credential store backed by the system catalog — required for @@ -143,9 +148,13 @@ impl TestClusterNode { } else { // Pre-bind the QUIC transport on a random port so we know the // listen address before wiring seeds / cluster settings. + let bind_addr = match listen_override { + Some(addr) => addr, + None => "127.0.0.1:0".parse()?, + }; let transport = Arc::new(nodedb_cluster::NexarTransport::new( node_id, - "127.0.0.1:0".parse()?, + bind_addr, nodedb_cluster::TransportCredentials::Insecure, )?); let listen_addr = transport.local_addr(); @@ -253,6 +262,7 @@ impl TestClusterNode { let core_handle = crate::core_loop_runner::spawn_core_loop(crate::core_loop_runner::CoreLoopSpawn { idx, + num_cores, data_side, core_dir: data_dir_path.clone(), core_array_catalog: shared.array_catalog.clone(), @@ -440,6 +450,19 @@ impl TestClusterNode { let _ = connection.await; }); + // The node plans permission-checked statements only under an + // authorization lease, as a production node opens its gateway only + // once it holds one. + if let Some(timing) = shared.authorization_fence.timing() { + nodedb::control::security::auth_lease::await_planning_admitted( + &shared, + Duration::from_secs(15), + timing.renew_every, + ) + .await + .map_err(|e| format!("node {node_id}: {e}"))?; + } + Ok(Self { node_id, listen_addr, diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_variants.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_variants.rs index d223e1684..dffccc7ca 100644 --- a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_variants.rs +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_variants.rs @@ -159,6 +159,6 @@ impl TestClusterNode { replication_factor: 1, single_node_calvin: true, }; - Self::spawn_with_full_config_at(1, vec![], &config, data_dir_path).await + Self::spawn_with_full_config_at(1, vec![], &config, data_dir_path, None).await } } diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/teardown.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/teardown.rs index 0e8c5c555..0c77dc508 100644 --- a/nodedb-test-support/src/cluster_harness/node/lifecycle/teardown.rs +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/teardown.rs @@ -136,6 +136,20 @@ impl TestClusterNode { } } + // Close the QUIC endpoint. Its driver owns the UDP socket and runs + // until every connection is gone, and a live peer keeps its + // connection open until the idle timeout. Without the close, a + // restart on the same address finds the port still bound. + if let Some(transport) = self.shared.cluster_transport.clone() + && !transport.close(Duration::from_secs(2)).await + { + eprintln!( + "graceful_shutdown_wal_only: node {} peers did not acknowledge the transport \ + close within 2s", + self.node_id + ); + } + // `start_raft` fans out to background tasks (raft apply loop, tick // loop, sequencer service, RPC server, health monitor, per-vShard // Calvin schedulers, reconcile loop) that each hold an diff --git a/nodedb-test-support/src/core_loop_runner.rs b/nodedb-test-support/src/core_loop_runner.rs index fab331129..a17642c8a 100644 --- a/nodedb-test-support/src/core_loop_runner.rs +++ b/nodedb-test-support/src/core_loop_runner.rs @@ -61,6 +61,10 @@ pub struct WalReplay { pub struct CoreLoopSpawn { /// Core index within the data plane (0-based). pub idx: usize, + /// Total number of Data-Plane cores in this node. The committed-redo apply + /// routes each record to `vshard_id % num_cores`, the same rule the + /// dispatcher routes requests by. + pub num_cores: usize, /// SPSC bridge endpoints for this core. pub data_side: CoreChannelDataSide, /// Storage directory shared with the rest of the harness. @@ -109,6 +113,7 @@ pub struct CoreLoopSpawn { pub fn spawn_core_loop(spawn: CoreLoopSpawn) -> tokio::task::JoinHandle<()> { let CoreLoopSpawn { idx, + num_cores, data_side, core_dir, core_array_catalog, @@ -138,6 +143,7 @@ pub fn spawn_core_loop(spawn: CoreLoopSpawn) -> tokio::task::JoinHandle<()> { ) .expect("CoreLoop::open_with_array_catalog"); core.set_event_producer(event_producer); + core.set_num_cores(num_cores); core.set_query_tuning(query_tuning); core.set_graph_tuning(graph_tuning); if let Some(m) = core_metrics { @@ -150,10 +156,10 @@ pub fn spawn_core_loop(spawn: CoreLoopSpawn) -> tokio::task::JoinHandle<()> { if let Some(WalReplay { records, tombstones, - num_cores, + num_cores: replay_num_cores, }) = replay { - core.replay_all_wal(&records, num_cores, &tombstones); + core.replay_all_wal(&records, replay_num_cores, &tombstones); } while matches!( stop_rx.try_recv(), diff --git a/nodedb-test-support/src/lib.rs b/nodedb-test-support/src/lib.rs index 16e370094..6a79b6b09 100644 --- a/nodedb-test-support/src/lib.rs +++ b/nodedb-test-support/src/lib.rs @@ -15,6 +15,7 @@ pub mod pgwire_harness; pub mod sync_client; pub mod test_tracing; pub mod tx_batch_helpers; +pub mod tx_commit; use nodedb::event::cdc::event::CdcEvent; use nodedb_types::DatabaseId; @@ -53,5 +54,6 @@ pub fn make_cdc_event( field_diffs: None, system_time_ms: None, valid_time_ms: None, + source: nodedb::event::EventSource::User, } } diff --git a/nodedb-test-support/src/native_harness/server.rs b/nodedb-test-support/src/native_harness/server.rs index e20af7293..3bf526426 100644 --- a/nodedb-test-support/src/native_harness/server.rs +++ b/nodedb-test-support/src/native_harness/server.rs @@ -77,6 +77,8 @@ impl NativeTestServer { let shared = SharedState::new_with_credentials(dispatcher, Arc::clone(&wal), credentials, false) .expect("build shared state"); + // The same gateway install production boot runs. + nodedb::bootstrap::state_wiring::install_gateway(&shared); let data_side = data_sides.into_iter().next().expect("data side"); let core_dir = dir.path().to_path_buf(); @@ -98,6 +100,7 @@ impl NativeTestServer { ) .expect("open core"); core.set_event_producer(event_producer); + core.set_num_cores(1); if let Some(m) = core_metrics { core.set_metrics(m); } diff --git a/nodedb-test-support/src/pgwire_harness/mod.rs b/nodedb-test-support/src/pgwire_harness/mod.rs index f38c81cd9..47aa90af5 100644 --- a/nodedb-test-support/src/pgwire_harness/mod.rs +++ b/nodedb-test-support/src/pgwire_harness/mod.rs @@ -8,6 +8,7 @@ mod multicore; mod query; pub mod raw_pgwire; +mod read_gate; mod restart; mod start; mod support; diff --git a/nodedb-test-support/src/pgwire_harness/multicore.rs b/nodedb-test-support/src/pgwire_harness/multicore.rs index caf2206ee..0f71f6873 100644 --- a/nodedb-test-support/src/pgwire_harness/multicore.rs +++ b/nodedb-test-support/src/pgwire_harness/multicore.rs @@ -51,6 +51,9 @@ impl TestServer { s.governor = init_test_memory_governor(); } let shared = shared; + // The same gateway install production boot runs, after every + // `Arc::get_mut` above. + nodedb::bootstrap::state_wiring::install_gateway(&shared); let mut core_stop_txs = Vec::new(); let mut core_handles = Vec::new(); @@ -61,6 +64,7 @@ impl TestServer { let core_handle = crate::core_loop_runner::spawn_core_loop(crate::core_loop_runner::CoreLoopSpawn { idx, + num_cores, data_side, core_dir: dir.path().to_path_buf(), core_array_catalog: shared.array_catalog.clone(), diff --git a/nodedb-test-support/src/pgwire_harness/read_gate.rs b/nodedb-test-support/src/pgwire_harness/read_gate.rs new file mode 100644 index 000000000..e501b7646 --- /dev/null +++ b/nodedb-test-support/src/pgwire_harness/read_gate.rs @@ -0,0 +1,73 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The Raft read gate for a harness that runs as a single-node cluster. +//! +//! Production `start_raft` publishes a gate backed by the Raft loop. A +//! harness started with a routing table runs no Raft loop. Without a gate, +//! every linearizable read refuses with "no leader is currently serving +//! this range". This gate answers both questions the way a group with one +//! voter answers them. + +use std::sync::{Arc, RwLock}; +use std::time::Duration; + +use nodedb::control::cluster::{RaftReadGate, ReadIndexRefusal}; +use nodedb::control::state::SharedState; +use nodedb_cluster::RoutingTable; + +/// Read gate for groups whose only voter is this node. +struct SingleVoterReadGate { + node_id: u64, + routing: Arc>, +} + +impl SingleVoterReadGate { + /// Whether this node is the sole voter and the leader of `group_id`. + fn is_sole_leader(&self, group_id: u64) -> bool { + let routing = self.routing.read().unwrap_or_else(|p| p.into_inner()); + routing + .group_info(group_id) + .is_some_and(|info| info.leader == self.node_id && info.members == [self.node_id]) + } +} + +#[async_trait::async_trait] +impl RaftReadGate for SingleVoterReadGate { + /// A sole voter is its own quorum, so it confirms leadership at once. + /// + /// The harness keeps no Raft log, so the read index is `0`. The caller + /// serves the read from local state. + async fn confirm_leader( + &self, + group_id: u64, + _timeout: Duration, + ) -> Result { + if self.is_sole_leader(group_id) { + Ok(0) + } else { + Err(ReadIndexRefusal::NotLeader) + } + } + + /// A sole voter holds the only copy, so it is never behind. + fn within_staleness_bound(&self, group_id: u64, _max_staleness: Duration) -> bool { + self.is_sole_leader(group_id) + } +} + +/// Publish the single-voter read gate when `shared` carries a routing table. +/// +/// `raft_read_gate` is a `OnceLock`, so this runs once, after every +/// `Arc::get_mut` install. +pub(super) fn install_single_voter_read_gate(shared: &SharedState) { + let Some(routing) = shared.cluster_routing.as_ref() else { + return; + }; + let gate: Arc = Arc::new(SingleVoterReadGate { + node_id: shared.node_id, + routing: Arc::clone(routing), + }); + if shared.raft_read_gate.set(gate).is_err() { + panic!("harness raft_read_gate installed twice"); + } +} diff --git a/nodedb-test-support/src/pgwire_harness/restart.rs b/nodedb-test-support/src/pgwire_harness/restart.rs index cc815fcc3..4a0e1f7bf 100644 --- a/nodedb-test-support/src/pgwire_harness/restart.rs +++ b/nodedb-test-support/src/pgwire_harness/restart.rs @@ -200,6 +200,9 @@ impl TestServer { s.governor = init_test_memory_governor(); } let shared = shared; + // The same gateway install production boot runs, after every + // `Arc::get_mut` above. + nodedb::bootstrap::state_wiring::install_gateway(&shared); nodedb::bootstrap::credentials::replay_surrogate_wal( &shared, &wal_records, @@ -234,6 +237,8 @@ impl TestServer { let core_handle = crate::core_loop_runner::spawn_core_loop(crate::core_loop_runner::CoreLoopSpawn { idx, + // Single-core harness (`Dispatcher::new(1, ..)`). + num_cores: 1, data_side, core_dir: dir_path.to_path_buf(), core_array_catalog: shared.array_catalog.clone(), @@ -317,6 +322,12 @@ impl TestServer { shutdown_bus: shutdown_bus.clone(), }); + // Load grants and hierarchy edges before the listener opens, as the + // production boot does once the data groups replayed. + nodedb::bootstrap::permission_tree_load::load_permission_trees(&shared) + .await + .expect("permission tree load on restart"); + let pg_listener = PgListener::bind("127.0.0.1:0".parse().unwrap()) .await .unwrap(); diff --git a/nodedb-test-support/src/pgwire_harness/start.rs b/nodedb-test-support/src/pgwire_harness/start.rs index ccd2044df..782b8a4a0 100644 --- a/nodedb-test-support/src/pgwire_harness/start.rs +++ b/nodedb-test-support/src/pgwire_harness/start.rs @@ -13,7 +13,10 @@ use nodedb::control::state::SharedState; use nodedb::event::{EventPlane, EventPlaneConfig, create_event_bus}; use nodedb::wal::WalManager; -use super::support::{bind_http_listener, bind_native_listener, init_test_memory_governor}; +use super::read_gate::install_single_voter_read_gate; +use super::support::{ + bind_http_listener, bind_native_listener, init_test_memory_governor, single_routing_leader, +}; use super::types::{TestClient, TestDataDir, TestServer}; /// Knobs for spawning a `TestServer`. `Default` reproduces the historical @@ -228,6 +231,10 @@ impl TestServer { s.backup_kek = Some(Arc::new([0x42u8; 32])); s.governor = init_test_memory_governor(); if let Some(routing) = cfg.routing { + // Production takes `node_id` from the cluster handle that owns + // the routing table. A single-node server runs as the one node + // that leads every group in it, so the gateway routes locally. + s.node_id = single_routing_leader(&routing); s.cluster_routing = Some(std::sync::Arc::new(std::sync::RwLock::new(routing))); } s.jwks_registry = cfg.jwks_registry; @@ -240,6 +247,11 @@ impl TestServer { ); } let shared = shared; + // The same gateway install production boot runs, after every + // `Arc::get_mut` above. + nodedb::bootstrap::state_wiring::install_gateway(&shared); + // Production `start_raft` publishes the read gate for a routed node. + install_single_voter_read_gate(&shared); // Data Plane core. Share the SharedState's array_catalog so DDL // mutations made by the SQL converter are visible to the handler @@ -254,6 +266,8 @@ impl TestServer { let core_handle = crate::core_loop_runner::spawn_core_loop(crate::core_loop_runner::CoreLoopSpawn { idx, + // Single-core harness (`Dispatcher::new(1, ..)`). + num_cores: 1, data_side, core_dir: dir.path().to_path_buf(), core_array_catalog: shared.array_catalog.clone(), diff --git a/nodedb-test-support/src/pgwire_harness/support.rs b/nodedb-test-support/src/pgwire_harness/support.rs index f150057f5..13f7de794 100644 --- a/nodedb-test-support/src/pgwire_harness/support.rs +++ b/nodedb-test-support/src/pgwire_harness/support.rs @@ -27,6 +27,24 @@ pub(super) fn init_test_memory_governor() -> Arc { nodedb::memory::init_governor(ceiling, &budgets).expect("harness governor config is valid") } +/// The one node that leads every group in a single-node routing table. +/// +/// Panics when the table names no leader, or more than one: a single-node +/// harness cannot serve a group another node leads. +pub(super) fn single_routing_leader(routing: &nodedb_cluster::RoutingTable) -> u64 { + let mut leaders: Vec = routing + .group_members() + .values() + .map(|group| group.leader) + .collect(); + leaders.sort_unstable(); + leaders.dedup(); + match leaders.as_slice() { + [leader] if *leader != 0 => *leader, + other => panic!("single-node harness routing must name one leader, got {other:?}"), + } +} + /// Bind a native (MessagePack) protocol listener on `127.0.0.1:0` and /// spawn its accept loop. Returns the listener's local port plus the /// handle to await on shutdown. diff --git a/nodedb-test-support/src/tx_batch_helpers.rs b/nodedb-test-support/src/tx_batch_helpers.rs index 0230b6f37..4ab576dca 100644 --- a/nodedb-test-support/src/tx_batch_helpers.rs +++ b/nodedb-test-support/src/tx_batch_helpers.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Plan-builder helpers shared by transaction batch cross-engine tests. +//! Plan builders and a single-core commit driver shared by the transaction +//! cross-engine tests. #![allow(dead_code)] use std::sync::Arc; @@ -15,6 +16,8 @@ use nodedb_physical::physical_plan::{ AggregateSpec, ColumnarInsertIntent, ColumnarOp, CrdtOp, DocumentOp, GraphOp, KvOp, PhysicalPlan, QueryOp, TimeseriesOp, VectorOp, }; + +pub use crate::tx_commit::commit_plans; use nodedb_types::OrdinalClock; // ── Core setup ────────────────────────────────────────────────────────────── @@ -69,10 +72,8 @@ pub fn send_ok( rx: &mut Consumer, plan: PhysicalPlan, ) -> Vec { - tx.try_push(BridgeRequest { - inner: make_request(plan), - }) - .unwrap(); + tx.try_push(BridgeRequest::unfloored(make_request(plan))) + .unwrap(); core.tick(); let resp = rx.try_pop().unwrap(); assert_eq!( @@ -90,10 +91,8 @@ pub fn send_raw( rx: &mut Consumer, plan: PhysicalPlan, ) -> nodedb::bridge::envelope::Response { - tx.try_push(BridgeRequest { - inner: make_request(plan), - }) - .unwrap(); + tx.try_push(BridgeRequest::unfloored(make_request(plan))) + .unwrap(); core.tick(); rx.try_pop().unwrap().inner } @@ -110,58 +109,6 @@ fn qualify(collection: &str) -> nodedb_types::QualifiedCollection { nodedb_types::QualifiedCollection::new(nodedb_types::DatabaseId::DEFAULT, collection) } -pub fn vector_set_params(collection: &str) -> PhysicalPlan { - PhysicalPlan::Vector(VectorOp::SetParams { - collection: qualify(collection), - field_name: String::new(), - dim: 0, - m: 16, - ef_construction: 200, - metric: "cosine".into(), - index_type: String::new(), - pq_m: 0, - ivf_cells: 0, - ivf_nprobe: 0, - }) -} - -pub fn vector_seed(collection: &str) -> PhysicalPlan { - PhysicalPlan::Vector(VectorOp::Insert { - collection: qualify(collection), - vector: vec![1.0, 2.0, 3.0], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: None, - provenance: None, - }) -} - -pub fn vector_insert_ok(collection: &str) -> PhysicalPlan { - PhysicalPlan::Vector(VectorOp::Insert { - collection: qualify(collection), - vector: vec![0.5, 0.5, 0.5], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::new(101), - pk_bytes: None, - provenance: None, - }) -} - -/// Always fails: dim mismatch (index expects dim=3). -pub fn vector_fail(collection: &str) -> PhysicalPlan { - PhysicalPlan::Vector(VectorOp::Insert { - collection: qualify(collection), - vector: vec![1.0, 2.0], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: None, - provenance: None, - }) -} - pub fn doc_put(collection: &str, doc_id: &str, val: &[u8]) -> PhysicalPlan { PhysicalPlan::Document(DocumentOp::PointPut { collection: qualify(collection), @@ -233,6 +180,7 @@ pub fn kv_put(key: &[u8], value: &[u8]) -> PhysicalPlan { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }) } @@ -254,7 +202,9 @@ pub fn columnar_insert(collection: &str, id: &str, val: i64) -> PhysicalPlan { format: "msgpack".into(), intent: ColumnarInsertIntent::Insert, on_conflict_updates: Vec::new(), - surrogates: Vec::new(), + // One surrogate per row, as the planner binds it: a transaction + // stages each columnar row under its surrogate. + surrogates: vec![row_surrogate(id)], schema_bytes: Vec::new(), provenance: None, wal_lsn: None, @@ -266,6 +216,15 @@ pub fn columnar_insert(collection: &str, id: &str, val: i64) -> PhysicalPlan { }) } +/// A stable surrogate for the row whose primary key is `id`. Never zero, and +/// distinct from the small fixed surrogates the other helpers use. +fn row_surrogate(id: &str) -> nodedb_types::Surrogate { + let hash = id.bytes().fold(2_166_136_261u32, |h, b| { + (h ^ u32::from(b)).wrapping_mul(16_777_619) + }); + nodedb_types::Surrogate::new(hash | 0x8000_0000) +} + pub fn columnar_count(collection: &str) -> PhysicalPlan { PhysicalPlan::Query(QueryOp::Aggregate { collection: qualify(collection), @@ -336,6 +295,62 @@ pub fn crdt_apply(collection: &str, doc_id: &str) -> PhysicalPlan { }) } +/// A vector-primary direct insert of a 3-dimensional vector. +pub fn vector_direct_insert(collection: &str, surrogate: u32) -> PhysicalPlan { + let mut payload = std::collections::HashMap::new(); + payload.insert( + "id".to_string(), + nodedb_types::Value::String(format!("r{surrogate}")), + ); + PhysicalPlan::Vector(VectorOp::DirectInsert { + collection: qualify(collection), + field: "vec".into(), + surrogate: nodedb_types::Surrogate::new(surrogate), + pk_bytes: format!("r{surrogate}").into_bytes(), + vector: vec![0.5, 0.5, 0.5], + payload: zerompk::to_msgpack_vec(&payload).unwrap(), + quantization: nodedb_types::VectorQuantization::None, + storage_dtype: nodedb_types::VectorStorageDtype::F32, + payload_indexes: Vec::new(), + returning: None, + rls_filters: Vec::new(), + }) +} + +/// A CRDT row write of `{"title": title}`. +pub fn crdt_upsert(collection: &str, doc_id: &str, surrogate: u32) -> PhysicalPlan { + PhysicalPlan::Crdt(CrdtOp::DocUpsert { + collection: qualify(collection), + document_id: doc_id.into(), + fields_json: r#"{"title":"staged"}"#.into(), + surrogate: nodedb_types::Surrogate::new(surrogate), + partial: false, + verb: nodedb_physical::physical_plan::CrdtWriteVerb::Insert, + returning: None, + rls_filters: Vec::new(), + }) +} + +/// `plans` followed by two inserts of one key: the transaction's staging +/// refuses the second as a unique violation, so the transaction commits +/// none of `plans`. +pub fn with_unique_refusal(mut plans: Vec) -> Vec { + let insert = PhysicalPlan::Document(DocumentOp::PointInsert { + collection: qualify("refusal_probe"), + document_id: "dup".into(), + value: nodedb_types::json_to_msgpack(&serde_json::json!({"n": 1})).unwrap(), + surrogate: nodedb_types::Surrogate::new(9_001), + if_absent: false, + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + deferred_sum_targets: Vec::new(), + }); + plans.push(insert.clone()); + plans.push(insert); + plans +} + // ── Assertion helpers ───────────────────────────────────────────────────────── /// Assert that a KV key is absent (NotFound or empty payload). diff --git a/nodedb-test-support/src/tx_commit.rs b/nodedb-test-support/src/tx_commit.rs new file mode 100644 index 000000000..62bc8fade --- /dev/null +++ b/nodedb-test-support/src/tx_commit.rs @@ -0,0 +1,107 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A single-core commit driver: stage, resolve, and install a transaction. + +use nodedb::bridge::dispatch::{BridgeRequest, BridgeResponse}; +use nodedb::bridge::envelope::{Request, Response, Status}; +use nodedb::control::wal_replication::transaction_redo::collections::written_collections; +use nodedb::control::wal_replication::transaction_redo::sum_targets::redo_sum_targets; +use nodedb::data::executor::core_loop::CoreLoop; +use nodedb::types::Lsn; +use nodedb_bridge::buffer::{Consumer, Producer}; +use nodedb_physical::physical_plan::{MetaOp, PhysicalPlan}; + +use crate::tx_batch_helpers::make_request; + +fn send_request( + core: &mut CoreLoop, + tx: &mut Producer, + rx: &mut Consumer, + request: Request, +) -> Response { + tx.try_push(BridgeRequest::unfloored(request)).unwrap(); + core.tick(); + rx.try_pop().unwrap().inner +} + +/// Commit `plans` as one transaction the way a session COMMIT does on its +/// core: stage each plan under the transaction `lsn` names, resolve the +/// transaction into its redo record, and install the record at `lsn`. +/// +/// Returns the install's response, or the first staging or resolve refusal. +/// The staging overlay is released on every path. +pub fn commit_plans( + core: &mut CoreLoop, + tx: &mut Producer, + rx: &mut Consumer, + plans: Vec, + lsn: u64, +) -> Response { + let txn_id = nodedb_types::id::TxnId::new(lsn); + let in_txn = |plan: PhysicalPlan| { + let mut request = make_request(plan); + request.txn_id = Some(txn_id); + request + }; + let mut refusal = None; + for plan in &plans { + let staged = send_request( + core, + tx, + rx, + in_txn(PhysicalPlan::Meta(MetaOp::StageWrite { + plan: Box::new(plan.clone()), + })), + ); + if staged.status != Status::Ok { + refusal = Some(staged); + break; + } + } + let resolved = match refusal { + Some(refusal) => Err(refusal), + None => { + let resolved = send_request( + core, + tx, + rx, + in_txn(PhysicalPlan::Meta(MetaOp::ResolveTxn { + txn_id, + plans: plans.clone(), + })), + ); + if resolved.status == Status::Ok { + Ok(resolved.payload.to_vec()) + } else { + Err(resolved) + } + } + }; + let response = match resolved { + Ok(redo) => { + let mut install = make_request(PhysicalPlan::Meta(MetaOp::ApplyTransactionRedo { + redo, + collections: written_collections(&plans), + sum_targets: redo_sum_targets(&plans), + origin: nodedb_physical::physical_plan::RedoOrigin::Commit, + })); + install.wal_lsn = Some(Lsn::new(lsn)); + send_request(core, tx, rx, install) + } + Err(refusal) => refusal, + }; + // A session COMMIT releases the overlay once the install answered. + let dropped = send_request( + core, + tx, + rx, + in_txn(PhysicalPlan::Meta(MetaOp::DropTxnOverlay { txn_id })), + ); + assert_eq!( + dropped.status, + Status::Ok, + "drop overlay: {:?}", + dropped.error_code + ); + response +} diff --git a/nodedb-types/src/backup_envelope/crypto.rs b/nodedb-types/src/backup_envelope/crypto.rs index 8f9dbfce2..5e0eff7a0 100644 --- a/nodedb-types/src/backup_envelope/crypto.rs +++ b/nodedb-types/src/backup_envelope/crypto.rs @@ -2,7 +2,7 @@ //! Per-backup DEK + KEK wrapping for encrypted backup envelopes. //! -//! ## Wire layout after the 52-byte HEADER (version byte = 1): +//! ## Wire layout after the 52-byte HEADER (version byte = [`VERSION`]): //! //! ```text //! ┌─ CRYPTO BLOCK (68 bytes) ──────────────────────────────────────────────┐ @@ -56,7 +56,7 @@ use super::types::{Envelope, EnvelopeError, EnvelopeMeta, Section, read2, read4, use super::types::{HEADER_LEN, MAGIC, TRAILER_LEN, VERSION}; use super::write::{EnvelopeWriter, write_header}; -/// Size of the crypto block inserted after the header in version-2 envelopes. +/// Size of the crypto block inserted after the header in encrypted envelopes. /// /// Layout: kek_fingerprint(8) + dek_nonce(12) + wrapped_dek(48) = 68 bytes. const CRYPTO_BLOCK_LEN: usize = 68; @@ -117,7 +117,7 @@ fn aes_decrypt( // ── EnvelopeWriter extension ───────────────────────────────────────────────── impl EnvelopeWriter { - /// Finalize with encryption. Produces a version-1 encrypted envelope. + /// Finalize with encryption. Produces an encrypted envelope of [`VERSION`]. /// /// - Generates a random 32-byte DEK via `getrandom`. /// - Wraps the DEK with the KEK using AES-256-GCM (random 12-byte nonce). @@ -180,7 +180,7 @@ impl EnvelopeWriter { // ── Decryption ──────────────────────────────────────────────────────────────── -/// Parse and decrypt an encrypted backup envelope (version 1 with crypto block). +/// Parse and decrypt an encrypted backup envelope ([`VERSION`] with crypto block). /// /// Verifies the KEK fingerprint before attempting decryption, surfacing /// [`EnvelopeError::WrongBackupKek`] when the presented key does not match diff --git a/nodedb-types/src/backup_envelope/mod.rs b/nodedb-types/src/backup_envelope/mod.rs index f5d947170..0f2d8816e 100644 --- a/nodedb-types/src/backup_envelope/mod.rs +++ b/nodedb-types/src/backup_envelope/mod.rs @@ -9,11 +9,11 @@ pub use crypto::parse_encrypted; pub use read::parse; pub use types::{ DEFAULT_MAX_SECTION_BYTES, DEFAULT_MAX_TOTAL_BYTES, HEADER_LEN, MAGIC, - SECTION_ORIGIN_CATALOG_ROWS, SECTION_ORIGIN_SOURCE_TOMBSTONES, SECTION_ORIGIN_SURROGATE_PK, - SECTION_OVERHEAD, TRAILER_LEN, VERSION, + SECTION_ORIGIN_CATALOG_ROWS, SECTION_ORIGIN_DATABASES, SECTION_ORIGIN_SOURCE_TOMBSTONES, + SECTION_ORIGIN_SURROGATE_PK, SECTION_OVERHEAD, TRAILER_LEN, VERSION, }; pub use types::{ - Envelope, EnvelopeError, EnvelopeMeta, Section, SourceTombstoneEntry, StoredCollectionBlob, - SurrogateBindBlob, + DatabaseBlob, DatabaseDataSection, Envelope, EnvelopeError, EnvelopeMeta, Section, + SourceTombstoneEntry, StoredCollectionBlob, SurrogateBindBlob, }; pub use write::EnvelopeWriter; diff --git a/nodedb-types/src/backup_envelope/read.rs b/nodedb-types/src/backup_envelope/read.rs index 3a0354b3a..fd038a4e3 100644 --- a/nodedb-types/src/backup_envelope/read.rs +++ b/nodedb-types/src/backup_envelope/read.rs @@ -11,7 +11,7 @@ use super::types::{HEADER_LEN, MAGIC, SECTION_OVERHEAD, TRAILER_LEN, VERSION}; /// Parse and fully validate a plaintext backup envelope. /// -/// Rejects bytes that do not carry version 1 in the header. +/// Rejects bytes that do not carry [`VERSION`] in the header. /// Use [`crate::backup_envelope::parse_encrypted`] for encrypted envelopes. pub fn parse(bytes: &[u8], max_total: u64) -> Result { if bytes.len() as u64 > max_total { @@ -308,7 +308,7 @@ mod tests { assert!(parse(truncated, DEFAULT_MAX_TOTAL_BYTES).is_err()); } - /// Asserts `NDBB` magic at [0..4], VERSION == 1 at [4], and that the + /// Asserts `NDBB` magic at [0..4], VERSION == 2 at [4], and that the /// header CRC at [48..52] covers header bytes [0..48]. #[test] fn golden_backup_envelope_format() { @@ -319,9 +319,9 @@ mod tests { // Magic at [0..4]. assert_eq!(&bytes[0..4], MAGIC.as_slice(), "magic mismatch"); - // VERSION == 1 at [4]. + // VERSION == 2 at [4]. assert_eq!(bytes[4], VERSION, "version mismatch"); - assert_eq!(bytes[4], 1u8, "expected VERSION == 1"); + assert_eq!(bytes[4], 2u8, "expected VERSION == 2"); // Header CRC at [48..52] covers [0..48]. assert!(bytes.len() >= HEADER_LEN, "envelope too short for header"); diff --git a/nodedb-types/src/backup_envelope/types.rs b/nodedb-types/src/backup_envelope/types.rs index b61d8003b..1f514a943 100644 --- a/nodedb-types/src/backup_envelope/types.rs +++ b/nodedb-types/src/backup_envelope/types.rs @@ -7,9 +7,13 @@ use thiserror::Error; pub const MAGIC: &[u8; 4] = b"NDBB"; /// Backup envelope version stamped in byte 4 of every envelope header. -/// Both plaintext and encrypted envelopes use version 1; the presence -/// of the crypto block (68 bytes after the header) distinguishes them. -pub const VERSION: u8 = 1; +/// Plaintext and encrypted envelopes carry the same version. The crypto +/// block (68 bytes after the header) distinguishes them. +/// +/// Version 2 scopes every section body to a database: data sections carry a +/// [`DatabaseDataSection`], and the metadata sections name the database of +/// each entry. A version-1 envelope is refused. +pub const VERSION: u8 = 2; /// Header is fixed-size — 52 bytes (48 framed + 4 crc). /// @@ -40,6 +44,39 @@ pub const SECTION_ORIGIN_SOURCE_TOMBSTONES: u64 = 0xFFFF_FFFF_FFFF_FFF1; /// point-lookups (`WHERE id=`). The body is a msgpack-encoded /// `Vec`. pub const SECTION_ORIGIN_SURROGATE_PK: u64 = 0xFFFF_FFFF_FFFF_FFF2; +/// Section carrying every database the tenant has collections in. The body +/// is a msgpack-encoded `Vec`. Restore reads it first: every +/// other section names its database by the id recorded here. +pub const SECTION_ORIGIN_DATABASES: u64 = 0xFFFF_FFFF_FFFF_FFF3; + +/// One database of the backed-up tenant, carried in a +/// `SECTION_ORIGIN_DATABASES` section. +/// +/// `database_id` is the id on the source cluster. Restore maps it to the +/// destination database of the same `name`, and creates that database when +/// the destination has none. +#[derive(Debug, Clone, PartialEq, Eq, zerompk::ToMessagePack, zerompk::FromMessagePack)] +pub struct DatabaseBlob { + pub database_id: u64, + pub name: String, + /// zerompk-encoded `DatabaseDescriptor` from the `nodedb` crate: the + /// database's settings. + pub descriptor: Vec, + /// The database's own quota, when one is set. + pub database_quota: Option, + /// The tenant's quota inside this database, when one is set. + pub tenant_quota: Option, +} + +/// Body of a per-node data section: one database's slice of the tenant +/// snapshot a source node took. +#[derive(Debug, Clone, PartialEq, Eq, zerompk::ToMessagePack, zerompk::FromMessagePack)] +pub struct DatabaseDataSection { + /// Source id of the database the snapshot covers. + pub database_id: u64, + /// zerompk-encoded `TenantDataSnapshot` from the `nodedb` crate. + pub snapshot: Vec, +} /// Single catalog-row entry in a catalog-rows section. The outer /// container is `Vec` msgpack-encoded into the @@ -48,6 +85,8 @@ pub const SECTION_ORIGIN_SURROGATE_PK: u64 = 0xFFFF_FFFF_FFFF_FFF2; /// depend on the `nodedb` catalog types, so the blob is opaque here. #[derive(Debug, Clone, PartialEq, Eq, zerompk::ToMessagePack, zerompk::FromMessagePack)] pub struct StoredCollectionBlob { + /// Source id of the database the collection lives in. + pub database_id: u64, pub name: String, /// zerompk-encoded `StoredCollection`. pub bytes: Vec, @@ -59,6 +98,8 @@ pub struct StoredCollectionBlob { /// resurrect. #[derive(Debug, Clone, PartialEq, Eq, zerompk::ToMessagePack, zerompk::FromMessagePack)] pub struct SourceTombstoneEntry { + /// Source id of the database the collection lived in. + pub database_id: u64, pub collection: String, pub purge_lsn: u64, } @@ -66,11 +107,14 @@ pub struct SourceTombstoneEntry { /// Single PK→surrogate binding carried in a `SECTION_ORIGIN_SURROGATE_PK` /// section. The outer container is `Vec` msgpack-encoded into /// the section body. Mirrors one row of the source catalog's `surrogate_pk_v3` -/// table for one `(tenant_id, collection)`; rebound on the restore side so PK -/// point-lookups resolve. +/// table for one `(database_id, tenant_id, collection)`; rebound on the restore +/// side so PK point-lookups resolve. #[derive(Debug, Clone, PartialEq, Eq, zerompk::ToMessagePack, zerompk::FromMessagePack)] pub struct SurrogateBindBlob { + /// Source id of the database the collection lives in. + pub database_id: u64, pub tenant_id: u64, + /// Bare catalog name of the collection. pub collection: String, pub pk: Vec, pub surrogate: u32, @@ -153,3 +197,95 @@ pub fn read4(s: &[u8]) -> [u8; 4] { pub fn read8(s: &[u8]) -> [u8; 8] { [s[0], s[1], s[2], s[3], s[4], s[5], s[6], s[7]] } + +#[cfg(test)] +mod tests { + use super::*; + use crate::backup_envelope::{parse_encrypted, write::EnvelopeWriter}; + + const KEK: [u8; 32] = [0x5Au8; 32]; + + fn meta() -> EnvelopeMeta { + EnvelopeMeta { + tenant_id: 3, + source_vshard_count: 1024, + hash_seed: 0, + snapshot_watermark: 17, + } + } + + /// A database-scoped data section and the database list survive the + /// encrypted envelope intact, each database under its own id. + #[test] + fn database_sections_round_trip_through_an_encrypted_envelope() { + let databases = vec![ + DatabaseBlob { + database_id: 0, + name: "default".into(), + descriptor: vec![1], + database_quota: None, + tenant_quota: None, + }, + DatabaseBlob { + database_id: 1025, + name: "sales".into(), + descriptor: vec![2], + database_quota: Some(crate::QuotaRecord::DEFAULT), + tenant_quota: None, + }, + ]; + let data = DatabaseDataSection { + database_id: 1025, + snapshot: vec![9, 8, 7], + }; + let mut writer = EnvelopeWriter::new(meta()); + writer + .push_section( + SECTION_ORIGIN_DATABASES, + zerompk::to_msgpack_vec(&databases).expect("encode databases"), + ) + .expect("push databases"); + writer + .push_section(7, zerompk::to_msgpack_vec(&data).expect("encode data")) + .expect("push data"); + let bytes = writer.finalize_encrypted(&KEK).expect("encrypt"); + + let env = parse_encrypted(&bytes, DEFAULT_MAX_TOTAL_BYTES, &KEK).expect("parse"); + let decoded: Vec = + zerompk::from_msgpack(&env.sections[0].body).expect("decode databases"); + assert_eq!(decoded, databases); + let decoded: DatabaseDataSection = + zerompk::from_msgpack(&env.sections[1].body).expect("decode data"); + assert_eq!(decoded, data); + } + + /// The database id sits inside the encrypted body, so the AEAD tag + /// covers it: flipping a ciphertext byte fails the parse. + #[test] + fn a_tampered_database_section_fails_authentication() { + let data = DatabaseDataSection { + database_id: 1025, + snapshot: vec![1, 2, 3, 4], + }; + let mut writer = EnvelopeWriter::new(meta()); + writer + .push_section(7, zerompk::to_msgpack_vec(&data).expect("encode data")) + .expect("push data"); + let mut bytes = writer.finalize_encrypted(&KEK).expect("encrypt"); + // Header, crypto block, origin, length and nonce precede the body. + let body_start = HEADER_LEN + 68 + 8 + 4 + 12; + bytes[body_start] ^= 0xFF; + // Recompute the body and trailer CRCs so only the AEAD tag can refuse. + let body_len = u32::from_le_bytes(read4(&bytes[HEADER_LEN + 68 + 8..])) as usize; + let body_end = body_start + body_len; + let body_crc = crc32c::crc32c(&bytes[body_start..body_end]); + bytes[body_end..body_end + 4].copy_from_slice(&body_crc.to_le_bytes()); + let trailer_start = bytes.len() - TRAILER_LEN; + let trailer_crc = crc32c::crc32c(&bytes[..trailer_start]); + bytes[trailer_start..].copy_from_slice(&trailer_crc.to_le_bytes()); + assert_eq!( + parse_encrypted(&bytes, DEFAULT_MAX_TOTAL_BYTES, &KEK), + Err(EnvelopeError::DecryptionFailed) + ); + } +} diff --git a/nodedb-types/src/backup_envelope/write.rs b/nodedb-types/src/backup_envelope/write.rs index 4b15b7cb7..8c313e30d 100644 --- a/nodedb-types/src/backup_envelope/write.rs +++ b/nodedb-types/src/backup_envelope/write.rs @@ -59,7 +59,7 @@ impl EnvelopeWriter { Ok(()) } - /// Finalize without encryption. Produces a version-1 envelope. + /// Finalize without encryption. Produces a plaintext envelope of [`VERSION`]. pub fn finalize(self) -> Vec { let mut out = Vec::with_capacity(self.framed_size as usize); write_header(&mut out, &self.meta, self.sections.len() as u16, VERSION); diff --git a/nodedb-types/src/columnar/dml_wal_record.rs b/nodedb-types/src/columnar/dml_wal_record.rs index 277dded8c..09b06a9f9 100644 --- a/nodedb-types/src/columnar/dml_wal_record.rs +++ b/nodedb-types/src/columnar/dml_wal_record.rs @@ -109,6 +109,7 @@ mod tests { payload: vec![1, 2, 3], provenance: None, surrogates: Vec::new(), + conflict_policy: Vec::new(), }; let bytes = zerompk::to_msgpack_vec(&rec).expect("encode"); let as_dml_record: Result = zerompk::from_msgpack(&bytes); diff --git a/nodedb-types/src/columnar/image_wal_record.rs b/nodedb-types/src/columnar/image_wal_record.rs new file mode 100644 index 000000000..f67ef95a9 --- /dev/null +++ b/nodedb-types/src/columnar/image_wal_record.rs @@ -0,0 +1,114 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Columnar row-image WAL record payload. +//! +//! A committed transaction's columnar writes travel as the final row images +//! the transaction staged and showed its own reads. Each entry names one row +//! by its cross-engine surrogate. It carries the primary key of the base row +//! the transaction replaced or removed, and the image the row holds after the +//! commit. Replay installs the image verbatim. It never re-runs an +//! `ON CONFLICT` merge, a SET list or a predicate against the replaying +//! node's state. +//! +//! Rides `RecordType::TimeseriesBatch`, disambiguated from the other columnar +//! record shapes by `kind = "columnar_image"`. + +use serde::{Deserialize, Serialize}; + +/// The `kind` tag every [`ColumnarImageWalRecord`] carries. +pub const COLUMNAR_IMAGE_KIND: &str = "columnar_image"; + +/// One row inside a [`ColumnarImageWalRecord`]. +#[derive( + Debug, + Clone, + PartialEq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +#[msgpack(map)] +pub struct ColumnarImageWalRow { + /// The row's cross-engine surrogate. + pub surrogate: u32, + /// MessagePack-encoded primary key of the base row this write replaces or + /// removes. Empty when the row had no base row the write displaced by key + /// (an insert or an `ON CONFLICT` upsert, which the image's own key + /// overwrites). + pub prior_pk_msgpack: Vec, + /// MessagePack-encoded post-image (`Value::Object`, column name to value, + /// bitemporal columns included). Empty for a delete. + pub image_msgpack: Vec, +} + +/// Map-encoded columnar row-image WAL record. +#[derive( + Debug, + Clone, + PartialEq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +#[msgpack(map)] +pub struct ColumnarImageWalRecord { + /// Record kind tag. Always [`COLUMNAR_IMAGE_KIND`]. + pub kind: String, + /// Target collection name. + pub collection: String, + /// The catalog schema the writing plan carried (`ColumnarSchema`, + /// MessagePack). Empty when the plan carried none. Replay uses it to + /// create the engine on a node that holds no row of the collection yet. + pub schema_bytes: Vec, + /// Every row the transaction wrote, in surrogate order. + pub rows: Vec, +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::columnar::{ColumnarDmlWalRecord, ColumnarResolvedDmlWalRecord}; + + fn record() -> ColumnarImageWalRecord { + ColumnarImageWalRecord { + kind: COLUMNAR_IMAGE_KIND.to_string(), + collection: "events".to_string(), + schema_bytes: vec![9], + rows: vec![ + ColumnarImageWalRow { + surrogate: 7, + prior_pk_msgpack: vec![1], + image_msgpack: vec![2, 3], + }, + ColumnarImageWalRow { + surrogate: 8, + prior_pk_msgpack: vec![4], + image_msgpack: Vec::new(), + }, + ], + } + } + + #[test] + fn round_trips_every_row() { + let rec = record(); + let bytes = zerompk::to_msgpack_vec(&rec).expect("encode"); + let decoded: ColumnarImageWalRecord = zerompk::from_msgpack(&bytes).expect("decode"); + assert_eq!(decoded, rec); + } + + #[test] + fn does_not_decode_as_the_dml_record_shapes() { + let bytes = zerompk::to_msgpack_vec(&record()).expect("encode"); + let as_dml = zerompk::from_msgpack::(&bytes); + assert!(as_dml.map(|r| r.kind != "columnar_dml").unwrap_or(true)); + let as_resolved = zerompk::from_msgpack::(&bytes); + assert!( + as_resolved + .map(|r| r.kind != "columnar_resolved_dml") + .unwrap_or(true) + ); + } +} diff --git a/nodedb-types/src/columnar/mod.rs b/nodedb-types/src/columnar/mod.rs index f6534228b..5b2bca556 100644 --- a/nodedb-types/src/columnar/mod.rs +++ b/nodedb-types/src/columnar/mod.rs @@ -6,6 +6,7 @@ pub mod column_type; pub mod declared_type_keyword; pub mod dml_wal_record; pub mod float_width; +pub mod image_wal_record; pub mod int_width; pub mod profile; pub mod resolved_dml_wal_record; @@ -19,6 +20,7 @@ pub use column_type::ColumnType; pub use declared_type_keyword::declared_type_matches; pub use dml_wal_record::ColumnarDmlWalRecord; pub use float_width::FloatWidth; +pub use image_wal_record::{COLUMNAR_IMAGE_KIND, ColumnarImageWalRecord, ColumnarImageWalRow}; pub use int_width::IntWidth; pub use profile::{ColumnarProfile, DocumentMode}; pub use resolved_dml_wal_record::{ColumnarResolvedDmlWalRecord, ColumnarResolvedDmlWalRow}; diff --git a/nodedb-types/src/columnar/wal_record.rs b/nodedb-types/src/columnar/wal_record.rs index 5b5232013..856342d20 100644 --- a/nodedb-types/src/columnar/wal_record.rs +++ b/nodedb-types/src/columnar/wal_record.rs @@ -60,6 +60,13 @@ pub struct ColumnarWalRecord { #[serde(default)] #[msgpack(default)] pub surrogates: Vec, + /// What a row whose primary key already exists does, encoded by the + /// writer that knows the insert's conflict intent and `ON CONFLICT` + /// assignments. Empty for a plain insert, which replaces the row. Replay + /// decides each row the way the live insert did. + #[serde(default)] + #[msgpack(default)] + pub conflict_policy: Vec, } #[cfg(test)] @@ -80,6 +87,7 @@ mod tests { payload: vec![1, 2, 3, 4], provenance: Some(prov.clone()), surrogates: vec![Surrogate::new(10), Surrogate::new(11), Surrogate::new(12)], + conflict_policy: Vec::new(), }; let bytes = zerompk::to_msgpack_vec(&rec).expect("encode ColumnarWalRecord"); @@ -104,6 +112,7 @@ mod tests { payload: vec![9], provenance: None, surrogates: Vec::new(), + conflict_policy: Vec::new(), }; let bytes = zerompk::to_msgpack_vec(&rec).expect("encode"); let decoded: ColumnarWalRecord = zerompk::from_msgpack(&bytes).expect("decode"); diff --git a/nodedb-types/src/error/code.rs b/nodedb-types/src/error/code.rs index 2ff0461b7..0d0a3343b 100644 --- a/nodedb-types/src/error/code.rs +++ b/nodedb-types/src/error/code.rs @@ -10,180 +10,214 @@ use serde::{Deserialize, Serialize}; #[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] pub struct ErrorCode(pub u16); -impl ErrorCode { +/// Defines each named code once, and [`ErrorCode::ALL`] from the same list, +/// so the list of every code cannot miss a constant. +macro_rules! error_codes { + ( $( $(#[$meta:meta])* $name:ident = $value:literal; )* ) => { + impl ErrorCode { + $( + $(#[$meta])* + pub const $name: Self = Self($value); + )* + + /// Every named code, in declaration order. + pub const ALL: &'static [ErrorCode] = &[ $( Self::$name ),* ]; + } + }; +} + +error_codes! { // Write path (1000–1099) - pub const CONSTRAINT_VIOLATION: Self = Self(1000); - pub const WRITE_CONFLICT: Self = Self(1001); - pub const DEADLINE_EXCEEDED: Self = Self(1002); - pub const PREVALIDATION_REJECTED: Self = Self(1003); - pub const APPEND_ONLY_VIOLATION: Self = Self(1010); - pub const BALANCE_VIOLATION: Self = Self(1011); - pub const PERIOD_LOCKED: Self = Self(1012); - pub const STATE_TRANSITION_VIOLATION: Self = Self(1013); - pub const TRANSITION_CHECK_VIOLATION: Self = Self(1014); - pub const RETENTION_VIOLATION: Self = Self(1015); - pub const LEGAL_HOLD_ACTIVE: Self = Self(1016); + CONSTRAINT_VIOLATION = 1000; + WRITE_CONFLICT = 1001; + DEADLINE_EXCEEDED = 1002; + PREVALIDATION_REJECTED = 1003; + APPEND_ONLY_VIOLATION = 1010; + BALANCE_VIOLATION = 1011; + PERIOD_LOCKED = 1012; + STATE_TRANSITION_VIOLATION = 1013; + TRANSITION_CHECK_VIOLATION = 1014; + RETENTION_VIOLATION = 1015; + LEGAL_HOLD_ACTIVE = 1016; /// A period-lock reference row exists but does not carry the /// configured `status_column` — a misconfigured column name, not a /// locked period. - pub const PERIOD_LOCK_MISCONFIGURED: Self = Self(1017); - pub const TYPE_MISMATCH: Self = Self(1020); - pub const OVERFLOW: Self = Self(1021); - pub const INSUFFICIENT_BALANCE: Self = Self(1022); - pub const RATE_EXCEEDED: Self = Self(1023); - pub const TYPE_GUARD_VIOLATION: Self = Self(1024); + PERIOD_LOCK_MISCONFIGURED = 1017; + TYPE_MISMATCH = 1020; + OVERFLOW = 1021; + INSUFFICIENT_BALANCE = 1022; + RATE_EXCEEDED = 1023; + TYPE_GUARD_VIOLATION = 1024; + /// A transaction rolled back for a reason other than a serialization + /// conflict, such as a participant error. The client retries it. + TRANSACTION_ROLLBACK = 1030; + /// The statement cannot run inside an explicit transaction block. + ACTIVE_SQL_TRANSACTION = 1031; // Read path (1100–1199) - pub const COLLECTION_NOT_FOUND: Self = Self(1100); - pub const DOCUMENT_NOT_FOUND: Self = Self(1101); - pub const COLLECTION_DRAINING: Self = Self(1102); - pub const COLLECTION_DEACTIVATED: Self = Self(1103); + COLLECTION_NOT_FOUND = 1100; + DOCUMENT_NOT_FOUND = 1101; + COLLECTION_DRAINING = 1102; + COLLECTION_DEACTIVATED = 1103; /// The named database does not exist. - pub const DATABASE_NOT_FOUND: Self = Self(1110); + DATABASE_NOT_FOUND = 1110; /// A named catalog object (type, role, index, alert, …) other than a /// collection or database does not exist. Generic: use the object name /// in the message for specifics. - pub const UNDEFINED_OBJECT: Self = Self(1111); + UNDEFINED_OBJECT = 1111; /// A named catalog object already exists under that name. - pub const ALREADY_EXISTS: Self = Self(1112); + ALREADY_EXISTS = 1112; /// The target object exists but is not in a state that accepts this /// operation (locked, busy, mid-transition). - pub const OBJECT_NOT_READY: Self = Self(1113); + OBJECT_NOT_READY = 1113; /// A requested value/record does not exist. Generic: use for lookups /// that don't fit `DOCUMENT_NOT_FOUND`'s collection/id shape. - pub const NOT_FOUND: Self = Self(1114); + NOT_FOUND = 1114; + /// A drop or revoke refused because other objects still depend on the + /// named object. + DEPENDENT_OBJECTS_EXIST = 1115; // Query (1200–1299) - pub const PLAN_ERROR: Self = Self(1200); - pub const FAN_OUT_EXCEEDED: Self = Self(1201); - pub const SQL_NOT_ENABLED: Self = Self(1202); + PLAN_ERROR = 1200; + FAN_OUT_EXCEEDED = 1201; + SQL_NOT_ENABLED = 1202; /// A function call names no registered scalar/aggregate/window function. - pub const UNDEFINED_FUNCTION: Self = Self(1203); + UNDEFINED_FUNCTION = 1203; /// Expression evaluation divided or took a modulus by zero. - pub const DIVISION_BY_ZERO: Self = Self(1204); + DIVISION_BY_ZERO = 1204; /// A LIMIT/OFFSET/FETCH bound resolved outside `[0, usize::MAX]`. - pub const INVALID_LIMIT_VALUE: Self = Self(1205); + INVALID_LIMIT_VALUE = 1205; /// A column reference names no column of any relation in scope. - pub const UNDEFINED_COLUMN: Self = Self(1206); + UNDEFINED_COLUMN = 1206; /// A bare column name resolves against more than one relation in scope. - pub const AMBIGUOUS_COLUMN: Self = Self(1207); + AMBIGUOUS_COLUMN = 1207; + /// A function received a value it cannot compute on: a vector of the + /// wrong dimension, an argument of the wrong shape, a malformed path. + DATA_EXCEPTION = 1208; + /// A statement exceeded a server limit on its own size or depth: a + /// recursion depth, a per-transaction staging budget. + PROGRAM_LIMIT_EXCEEDED = 1209; // Engine ops (1300–1399) - pub const ARRAY: Self = Self(1300); + ARRAY = 1300; // Quota (1400–1499) /// The proposed quota allocation would push the sum of all database quotas /// past the configured global ceiling, or the sum of all tenant quotas past /// the database ceiling. - pub const QUOTA_OVERCOMMIT: Self = Self(1400); + QUOTA_OVERCOMMIT = 1400; /// A request was rejected because the calling tenant has exhausted its quota /// (QPS, memory, connections, or storage). - pub const TENANT_QUOTA_EXCEEDED: Self = Self(1401); + TENANT_QUOTA_EXCEEDED = 1401; /// A request was rejected because the target database has exhausted its quota. - pub const DATABASE_QUOTA_EXCEEDED: Self = Self(1402); + DATABASE_QUOTA_EXCEEDED = 1402; /// The server is under global resource pressure and cannot accept new requests. - pub const SERVER_OVERLOAD: Self = Self(1403); + SERVER_OVERLOAD = 1403; // Clone (1500–1599) /// A `CLONE DATABASE` would exceed the maximum clone chain depth of 8. - pub const CLONE_DEPTH_EXCEEDED: Self = Self(1500); + CLONE_DEPTH_EXCEEDED = 1500; /// A mirror database cannot be cloned; promote the mirror first. - pub const CANNOT_CLONE_MIRROR: Self = Self(1501); + CANNOT_CLONE_MIRROR = 1501; /// The source database cannot be dropped while clones depend on it. - pub const CLONE_DEPENDENCY: Self = Self(1502); + CLONE_DEPENDENCY = 1502; /// A bitemporal `AS OF` query timestamp predates the clone's creation LSN. - pub const CLONE_PREDATES_QUERY_TIME: Self = Self(1503); + CLONE_PREDATES_QUERY_TIME = 1503; /// A write targeted a `Shadowed`/`Materializing` clone collection whose /// engine has no copy-on-write support; `MATERIALIZE` the clone first. - pub const CLONE_WRITE_REQUIRES_MATERIALIZE: Self = Self(1504); + CLONE_WRITE_REQUIRES_MATERIALIZE = 1504; // Mirror (1700–1799) /// Write attempted on a mirror database that has not yet been promoted. - pub const MIRROR_READ_ONLY: Self = Self(1700); + MIRROR_READ_ONLY = 1700; /// Strong consistency read requested on a mirror; mirrors cannot serve /// strong reads. The client should retry against the source cluster. - pub const STALE_READ_NOT_LEADER: Self = Self(1701); + STALE_READ_NOT_LEADER = 1701; /// Operation requires the mirror to be promoted, but it has not been. - pub const MIRROR_NOT_PROMOTED: Self = Self(1702); + MIRROR_NOT_PROMOTED = 1702; /// `DROP DATABASE` targeted the built-in `default` database, which is /// immutable. Shares SQLSTATE `0A000` with `SQL_NOT_ENABLED` and /// `CANNOT_CLONE_MIRROR`, so it must be constructed explicitly rather /// than derived from the bare SQLSTATE string. - pub const CANNOT_DROP_DEFAULT_DATABASE: Self = Self(1710); + CANNOT_DROP_DEFAULT_DATABASE = 1710; // Move Tenant (1600–1699) /// `MOVE TENANT` drain phase timed out; source left unchanged. - pub const MOVE_TENANT_DRAIN_TIMEOUT: Self = Self(1600); + MOVE_TENANT_DRAIN_TIMEOUT = 1600; /// `MOVE TENANT` pre-flight failed; collection schema incompatibility. - pub const MOVE_TENANT_PREFLIGHT_FAILED: Self = Self(1601); + MOVE_TENANT_PREFLIGHT_FAILED = 1601; /// `MOVE TENANT` snapshot phase failed; source left unchanged. - pub const MOVE_TENANT_SNAPSHOT_FAILED: Self = Self(1602); + MOVE_TENANT_SNAPSHOT_FAILED = 1602; /// `MOVE TENANT` cutover phase failed; source still holds the data. - pub const MOVE_TENANT_CUTOVER_FAILED: Self = Self(1603); + MOVE_TENANT_CUTOVER_FAILED = 1603; /// Tenant is already at the target database; `MOVE TENANT` was a no-op. - pub const MOVE_TENANT_ALREADY_AT_TARGET: Self = Self(1604); + MOVE_TENANT_ALREADY_AT_TARGET = 1604; // Backup / Restore (1800–1899) /// RESTORE targeted a tenant different from the one the envelope belongs to. - pub const BACKUP_TENANT_MISMATCH: Self = Self(1800); + BACKUP_TENANT_MISMATCH = 1800; /// Backup envelope did not decrypt under this server's configured backup KEK. - pub const BACKUP_KEY_MISMATCH: Self = Self(1801); + BACKUP_KEY_MISMATCH = 1801; // Auth / Security (2000–2099) - pub const AUTHORIZATION_DENIED: Self = Self(2000); - pub const AUTH_EXPIRED: Self = Self(2001); + AUTHORIZATION_DENIED = 2000; + AUTH_EXPIRED = 2001; + /// Authentication failed: a wrong password, an unknown user, or missing + /// credentials. One code for all of them, so a client cannot tell them + /// apart. + AUTHENTICATION_FAILED = 2002; /// Vector insert or index rejected because the vector dimension exceeds the /// tenant's `max_vector_dim` quota. - pub const TENANT_VECTOR_DIM_EXCEEDED: Self = Self(2010); + TENANT_VECTOR_DIM_EXCEEDED = 2010; /// Graph traversal rejected because the requested depth exceeds the tenant's /// `max_graph_depth` quota. - pub const TENANT_GRAPH_DEPTH_EXCEEDED: Self = Self(2011); + TENANT_GRAPH_DEPTH_EXCEEDED = 2011; // Protocol handshake (2100–2199) - pub const HANDSHAKE_FAILED: Self = Self(2100); + HANDSHAKE_FAILED = 2100; // Sync (3000–3099) - pub const SYNC_CONNECTION_FAILED: Self = Self(3000); - pub const SYNC_DELTA_REJECTED: Self = Self(3001); - pub const SHAPE_SUBSCRIPTION_FAILED: Self = Self(3002); + SYNC_CONNECTION_FAILED = 3000; + SYNC_DELTA_REJECTED = 3001; + SHAPE_SUBSCRIPTION_FAILED = 3002; // Storage (4000–4099) - pub const STORAGE: Self = Self(4000); - pub const SEGMENT_CORRUPTED: Self = Self(4001); - pub const COLD_STORAGE: Self = Self(4002); + STORAGE = 4000; + SEGMENT_CORRUPTED = 4001; + COLD_STORAGE = 4002; // WAL (4100–4199) - pub const WAL: Self = Self(4100); + WAL = 4100; // Serialization (4200–4299) - pub const SERIALIZATION: Self = Self(4200); - pub const CODEC: Self = Self(4201); + SERIALIZATION = 4200; + CODEC = 4201; // Config (5000–5099) - pub const CONFIG: Self = Self(5000); - pub const BAD_REQUEST: Self = Self(5001); + CONFIG = 5000; + BAD_REQUEST = 5001; // Cluster (6000–6099) - pub const NO_LEADER: Self = Self(6000); - pub const NOT_LEADER: Self = Self(6001); - pub const MIGRATION_IN_PROGRESS: Self = Self(6002); - pub const NODE_UNREACHABLE: Self = Self(6003); - pub const CLUSTER: Self = Self(6010); + NO_LEADER = 6000; + NOT_LEADER = 6001; + MIGRATION_IN_PROGRESS = 6002; + NODE_UNREACHABLE = 6003; + CLUSTER = 6010; // Memory (7000–7099) - pub const MEMORY_EXHAUSTED: Self = Self(7000); + MEMORY_EXHAUSTED = 7000; // Encryption (8000–8099) - pub const ENCRYPTION: Self = Self(8000); + ENCRYPTION = 8000; // Internal (9000–9099) - pub const INTERNAL: Self = Self(9000); - pub const BRIDGE: Self = Self(9001); - pub const DISPATCH: Self = Self(9002); + INTERNAL = 9000; + BRIDGE = 9001; + DISPATCH = 9002; } impl fmt::Display for ErrorCode { @@ -202,4 +236,14 @@ mod tests { assert_eq!(ErrorCode::INTERNAL.to_string(), "NDB-9000"); assert_eq!(ErrorCode::WAL.to_string(), "NDB-4100"); } + + #[test] + fn every_code_has_a_distinct_value() { + let mut seen = std::collections::HashSet::new(); + for code in ErrorCode::ALL { + assert!(seen.insert(code.0), "{code} is defined twice"); + } + assert!(ErrorCode::ALL.contains(&ErrorCode::DATABASE_QUOTA_EXCEEDED)); + assert!(ErrorCode::ALL.contains(&ErrorCode::DISPATCH)); + } } diff --git a/nodedb-types/src/error/code_table.rs b/nodedb-types/src/error/code_table.rs index f9fdfdc14..7bd5cfe6c 100644 --- a/nodedb-types/src/error/code_table.rs +++ b/nodedb-types/src/error/code_table.rs @@ -71,6 +71,8 @@ error_code_table! { OVERFLOW => Overflow { collection: String::new() }, INSUFFICIENT_BALANCE => InsufficientBalance { collection: String::new() }, RATE_EXCEEDED => RateExceeded { gate: String::new() }, + TRANSACTION_ROLLBACK => TransactionRollback { detail: message.to_owned() }, + ACTIVE_SQL_TRANSACTION => ActiveSqlTransaction { detail: message.to_owned() }, // Read path. COLLECTION_NOT_FOUND => CollectionNotFound { collection: String::new() }, @@ -79,6 +81,7 @@ error_code_table! { ALREADY_EXISTS => AlreadyExists { object: String::new() }, OBJECT_NOT_READY => ObjectNotReady { object: String::new() }, NOT_FOUND => NotFound { detail: message.to_owned() }, + DEPENDENT_OBJECTS_EXIST => DependentObjectsExist { object: String::new() }, DOCUMENT_NOT_FOUND => DocumentNotFound { collection: String::new(), document_id: String::new() }, COLLECTION_DRAINING => CollectionDraining { collection: String::new() }, COLLECTION_DEACTIVATED => CollectionDeactivated { @@ -95,11 +98,14 @@ error_code_table! { UNDEFINED_COLUMN => UndefinedColumn { column: String::new() }, AMBIGUOUS_COLUMN => AmbiguousColumn { column: String::new() }, DIVISION_BY_ZERO => DivisionByZero, + DATA_EXCEPTION => DataException { detail: message.to_owned() }, + PROGRAM_LIMIT_EXCEEDED => ProgramLimitExceeded { detail: message.to_owned() }, INVALID_LIMIT_VALUE => InvalidLimitValue { clause: "remote".into(), value: message.to_owned() }, // Auth / tenant quota. AUTHORIZATION_DENIED => AuthorizationDenied { resource: String::new() }, AUTH_EXPIRED => AuthExpired, + AUTHENTICATION_FAILED => AuthenticationFailed, TENANT_VECTOR_DIM_EXCEEDED => TenantVectorDimExceeded { dim: 0, limit: 0 }, TENANT_GRAPH_DEPTH_EXCEEDED => TenantGraphDepthExceeded { depth: 0, limit: 0 }, @@ -237,6 +243,11 @@ mod tests { ErrorCode::CANNOT_DROP_DEFAULT_DATABASE, ErrorCode::COLLECTION_DEACTIVATED, ErrorCode::ARRAY, + ErrorCode::PROGRAM_LIMIT_EXCEEDED, + ErrorCode::TRANSACTION_ROLLBACK, + ErrorCode::ACTIVE_SQL_TRANSACTION, + ErrorCode::DEPENDENT_OBJECTS_EXIST, + ErrorCode::AUTHENTICATION_FAILED, ErrorCode::QUOTA_OVERCOMMIT, ErrorCode::CLONE_DEPTH_EXCEEDED, ErrorCode::CLONE_WRITE_REQUIRES_MATERIALIZE, diff --git a/nodedb-types/src/error/ctors/from_wire.rs b/nodedb-types/src/error/ctors/from_wire.rs index 2cc5b13b1..67badbe63 100644 --- a/nodedb-types/src/error/ctors/from_wire.rs +++ b/nodedb-types/src/error/ctors/from_wire.rs @@ -49,6 +49,28 @@ impl NodeDbError { cause: None, } } + + /// Rebuild a typed error from a wire frame that also carries the + /// structured details the server held. + /// + /// The carried details are used when they belong to `code`, so a client + /// reads the collection, gate, or document the server named. Details of + /// another category, or none, fall back to [`NodeDbError::from_wire`]. + pub fn from_wire_with_details( + code: ErrorCode, + message: impl Into, + details: Option, + ) -> Self { + match details { + Some(details) if details.code() == code => Self { + code, + message: message.into(), + details, + cause: None, + }, + _ => Self::from_wire(code, message), + } + } } /// Map a numeric code onto the details variant that carries its category, @@ -114,6 +136,40 @@ mod tests { } } + #[test] + fn carried_details_keep_the_collection() { + let e = NodeDbError::from_wire_with_details( + ErrorCode::OVERFLOW, + "increment or decrement would overflow on counters", + Some(ErrorDetails::Overflow { + collection: "counters".into(), + }), + ); + assert_eq!( + e.details(), + &ErrorDetails::Overflow { + collection: "counters".into() + } + ); + } + + #[test] + fn details_of_another_category_fall_back_to_the_code() { + let e = NodeDbError::from_wire_with_details( + ErrorCode::OVERFLOW, + "overflow", + Some(ErrorDetails::TypeMismatch { + collection: "counters".into(), + }), + ); + assert_eq!( + e.details(), + &ErrorDetails::Overflow { + collection: String::new() + } + ); + } + #[test] fn genuinely_unmapped_codes_still_fall_back_to_internal() { assert!(NodeDbError::from_wire(ErrorCode(65000), "x").is_internal()); diff --git a/nodedb-types/src/error/ctors/read_query_auth.rs b/nodedb-types/src/error/ctors/read_query_auth.rs index 9ac2d70a5..19c769a07 100644 --- a/nodedb-types/src/error/ctors/read_query_auth.rs +++ b/nodedb-types/src/error/ctors/read_query_auth.rs @@ -147,6 +147,20 @@ impl NodeDbError { } } + /// A drop or revoke refused because other objects still depend on + /// `object`. SQLSTATE `2BP01` (`dependent_objects_still_exist`). + /// `message` is the full message and names the dependents. + pub fn dependent_objects_exist(object: impl Into, message: impl Into) -> Self { + Self { + code: ErrorCode::DEPENDENT_OBJECTS_EXIST, + message: message.into(), + details: ErrorDetails::DependentObjectsExist { + object: object.into(), + }, + cause: None, + } + } + /// An object exists but a prerequisite step has not run, such as `currval` /// before this session called `nextval`. Renders as SQLSTATE `55000` /// (`object_not_in_prerequisite_state`). @@ -199,6 +213,32 @@ impl NodeDbError { } } + /// A function received a value it cannot compute on: a vector of the + /// wrong dimension, an argument of the wrong shape, a malformed path. + /// SQLSTATE `22000` (`data_exception`). `detail` is the full message. + pub fn data_exception(detail: impl Into) -> Self { + let detail = detail.into(); + Self { + code: ErrorCode::DATA_EXCEPTION, + message: detail.clone(), + details: ErrorDetails::DataException { detail }, + cause: None, + } + } + + /// A statement exceeded a server limit on its own size or depth: a + /// recursion depth, a per-transaction staging budget. SQLSTATE `54000` + /// (`program_limit_exceeded`). `detail` is the full message. + pub fn program_limit_exceeded(detail: impl Into) -> Self { + let detail = detail.into(); + Self { + code: ErrorCode::PROGRAM_LIMIT_EXCEEDED, + message: detail.clone(), + details: ErrorDetails::ProgramLimitExceeded { detail }, + cause: None, + } + } + /// A LIMIT/OFFSET/FETCH bound did not resolve to `[0, usize::MAX]`. /// Distinct from `plan_error` so clients match the code, SQLSTATE /// `2201W`, instead of parsing the message. @@ -223,6 +263,17 @@ impl NodeDbError { } } + /// Credentials were rejected. `detail` must not say whether the user + /// exists. + pub fn authentication_failed(detail: impl fmt::Display) -> Self { + Self { + code: ErrorCode::AUTHENTICATION_FAILED, + message: format!("authentication failed: {detail}"), + details: ErrorDetails::AuthenticationFailed, + cause: None, + } + } + pub fn auth_expired(detail: impl fmt::Display) -> Self { Self { code: ErrorCode::AUTH_EXPIRED, diff --git a/nodedb-types/src/error/ctors/write_path.rs b/nodedb-types/src/error/ctors/write_path.rs index 0970ccacd..3fa68fac7 100644 --- a/nodedb-types/src/error/ctors/write_path.rs +++ b/nodedb-types/src/error/ctors/write_path.rs @@ -203,6 +203,41 @@ impl NodeDbError { } } + /// A KV counter atomic (`INCR`, `INCRBYFLOAT`) refused on `collection`. + /// + /// `fault` is the client text, e.g. `value is not an integer or out of + /// range`. The message is `"{fault} on {collection}"`, the text the SQL + /// surfaces send. `out_of_range` picks the class, and both classes are + /// the data-exception class (`22`) the SQL surfaces send: + /// + /// - `OVERFLOW` for a result out of range (SQLSTATE `22003`). + /// - `DATA_EXCEPTION` for a stored value that does not parse (SQLSTATE + /// `22P02`). `TYPE_MISMATCH` is the class for a key that holds the + /// wrong kind of value (SQLSTATE `42846`), a different condition. + pub fn kv_counter_fault( + collection: impl Into, + fault: impl fmt::Display, + out_of_range: bool, + ) -> Self { + let collection = collection.into(); + let message = format!("{fault} on {collection}"); + if out_of_range { + Self { + code: ErrorCode::OVERFLOW, + message, + details: ErrorDetails::Overflow { collection }, + cause: None, + } + } else { + Self { + code: ErrorCode::DATA_EXCEPTION, + message: message.clone(), + details: ErrorDetails::DataException { detail: message }, + cause: None, + } + } + } + pub fn insufficient_balance(collection: impl Into, detail: impl fmt::Display) -> Self { let collection = collection.into(); Self { @@ -222,4 +257,30 @@ impl NodeDbError { cause: None, } } + + /// A transaction rolled back for a reason other than a serialization + /// conflict. SQLSTATE `40000` (`transaction_rollback`). The client + /// retries it. `detail` is the full message. + pub fn transaction_rollback(detail: impl Into) -> Self { + let detail = detail.into(); + Self { + code: ErrorCode::TRANSACTION_ROLLBACK, + message: detail.clone(), + details: ErrorDetails::TransactionRollback { detail }, + cause: None, + } + } + + /// The statement cannot run inside an explicit transaction block. + /// SQLSTATE `25001` (`active_sql_transaction`). `detail` is the full + /// message. + pub fn active_sql_transaction(detail: impl Into) -> Self { + let detail = detail.into(); + Self { + code: ErrorCode::ACTIVE_SQL_TRANSACTION, + message: detail.clone(), + details: ErrorDetails::ActiveSqlTransaction { detail }, + cause: None, + } + } } diff --git a/nodedb-types/src/error/details.rs b/nodedb-types/src/error/details.rs index fbf0e4fc4..04af07aea 100644 --- a/nodedb-types/src/error/details.rs +++ b/nodedb-types/src/error/details.rs @@ -62,6 +62,14 @@ pub enum ErrorDetails { InsufficientBalance { collection: String }, #[serde(rename = "rate_exceeded")] RateExceeded { gate: String }, + /// A transaction rolled back for a reason other than a serialization + /// conflict. `detail` names the reason. The client retries it. + #[serde(rename = "transaction_rollback")] + TransactionRollback { detail: String }, + /// The statement cannot run inside an explicit transaction block. + /// `detail` names the statement or the refused operation. + #[serde(rename = "active_sql_transaction")] + ActiveSqlTransaction { detail: String }, // Read path #[serde(rename = "collection_not_found")] @@ -89,6 +97,10 @@ pub enum ErrorDetails { /// collection/document shape of `DocumentNotFound`. #[serde(rename = "not_found")] NotFound { detail: String }, + /// A drop or revoke refused because other objects still depend on + /// `object`. The message lists the dependents. + #[serde(rename = "dependent_objects_exist")] + DependentObjectsExist { object: String }, #[serde(rename = "collection_draining")] CollectionDraining { collection: String }, #[serde(rename = "collection_deactivated")] @@ -123,6 +135,14 @@ pub enum ErrorDetails { /// Expression evaluation divided or took a modulus by zero. #[serde(rename = "division_by_zero")] DivisionByZero, + /// A function received a value it cannot compute on. `detail` names + /// the function and the offending value. + #[serde(rename = "data_exception")] + DataException { detail: String }, + /// A statement exceeded a server limit on its own size or depth. + /// `detail` names the limit. + #[serde(rename = "program_limit_exceeded")] + ProgramLimitExceeded { detail: String }, /// A LIMIT/OFFSET/FETCH bound resolved outside `[0, usize::MAX]`. #[serde(rename = "invalid_limit_value")] InvalidLimitValue { clause: String, value: String }, @@ -132,6 +152,10 @@ pub enum ErrorDetails { AuthorizationDenied { resource: String }, #[serde(rename = "auth_expired")] AuthExpired, + /// Credentials were rejected. Carries nothing that tells a wrong + /// password from an unknown user. + #[serde(rename = "authentication_failed")] + AuthenticationFailed, /// Tenant quota: vector dimension exceeds `max_vector_dim`. #[serde(rename = "tenant_vector_dim_exceeded")] TenantVectorDimExceeded { dim: u32, limit: u32 }, diff --git a/nodedb-types/src/error/msgpack/constants.rs b/nodedb-types/src/error/msgpack/constants.rs index 9627de4b7..893da1017 100644 --- a/nodedb-types/src/error/msgpack/constants.rs +++ b/nodedb-types/src/error/msgpack/constants.rs @@ -85,6 +85,12 @@ // | 79 | UndefinedColumn | // | 80 | AmbiguousColumn | // | 81 | PeriodLockMisconfigured | +// | 82 | DataException | +// | 83 | ProgramLimitExceeded | +// | 84 | TransactionRollback | +// | 85 | ActiveSqlTransaction | +// | 86 | DependentObjectsExist | +// | 87 | AuthenticationFailed | pub(super) const TAG_CONSTRAINT_VIOLATION: u16 = 1; pub(super) const TAG_WRITE_CONFLICT: u16 = 2; @@ -167,3 +173,9 @@ pub(super) const TAG_INVALID_LIMIT_VALUE: u16 = 78; pub(super) const TAG_UNDEFINED_COLUMN: u16 = 79; pub(super) const TAG_AMBIGUOUS_COLUMN: u16 = 80; pub(super) const TAG_PERIOD_LOCK_MISCONFIGURED: u16 = 81; +pub(super) const TAG_DATA_EXCEPTION: u16 = 82; +pub(super) const TAG_PROGRAM_LIMIT_EXCEEDED: u16 = 83; +pub(super) const TAG_TRANSACTION_ROLLBACK: u16 = 84; +pub(super) const TAG_ACTIVE_SQL_TRANSACTION: u16 = 85; +pub(super) const TAG_DEPENDENT_OBJECTS_EXIST: u16 = 86; +pub(super) const TAG_AUTHENTICATION_FAILED: u16 = 87; diff --git a/nodedb-types/src/error/msgpack/decode/from_messagepack.rs b/nodedb-types/src/error/msgpack/decode/from_messagepack.rs index 5666fcd4d..94ce8ad8f 100644 --- a/nodedb-types/src/error/msgpack/decode/from_messagepack.rs +++ b/nodedb-types/src/error/msgpack/decode/from_messagepack.rs @@ -96,6 +96,14 @@ impl<'a> FromMessagePack<'a> for ErrorDetails { let (gate,) = read1_str(reader, field_count)?; Ok(ErrorDetails::RateExceeded { gate }) } + TAG_TRANSACTION_ROLLBACK => { + let (detail,) = read1_str(reader, field_count)?; + Ok(ErrorDetails::TransactionRollback { detail }) + } + TAG_ACTIVE_SQL_TRANSACTION => { + let (detail,) = read1_str(reader, field_count)?; + Ok(ErrorDetails::ActiveSqlTransaction { detail }) + } TAG_COLLECTION_NOT_FOUND => { let (collection,) = read1_str(reader, field_count)?; Ok(ErrorDetails::CollectionNotFound { collection }) @@ -155,6 +163,14 @@ impl<'a> FromMessagePack<'a> for ErrorDetails { skip_fields(reader, field_count)?; Ok(ErrorDetails::DivisionByZero) } + TAG_DATA_EXCEPTION => { + let (detail,) = read1_str(reader, field_count)?; + Ok(ErrorDetails::DataException { detail }) + } + TAG_PROGRAM_LIMIT_EXCEEDED => { + let (detail,) = read1_str(reader, field_count)?; + Ok(ErrorDetails::ProgramLimitExceeded { detail }) + } TAG_INVALID_LIMIT_VALUE => { let (clause, value) = read2_str(reader, field_count)?; Ok(ErrorDetails::InvalidLimitValue { clause, value }) @@ -167,6 +183,10 @@ impl<'a> FromMessagePack<'a> for ErrorDetails { skip_fields(reader, field_count)?; Ok(ErrorDetails::AuthExpired) } + TAG_AUTHENTICATION_FAILED => { + skip_fields(reader, field_count)?; + Ok(ErrorDetails::AuthenticationFailed) + } TAG_SYNC_CONNECTION_FAILED => { skip_fields(reader, field_count)?; Ok(ErrorDetails::SyncConnectionFailed) @@ -387,6 +407,10 @@ impl<'a> FromMessagePack<'a> for ErrorDetails { let (detail,) = read1_str(reader, field_count)?; Ok(ErrorDetails::NotFound { detail }) } + TAG_DEPENDENT_OBJECTS_EXIST => { + let (object,) = read1_str(reader, field_count)?; + Ok(ErrorDetails::DependentObjectsExist { object }) + } TAG_CANNOT_DROP_DEFAULT_DATABASE => { skip_fields(reader, field_count)?; Ok(ErrorDetails::CannotDropDefaultDatabase) @@ -420,6 +444,7 @@ mod tests { ErrorDetails::SqlNotEnabled, ErrorDetails::DivisionByZero, ErrorDetails::AuthExpired, + ErrorDetails::AuthenticationFailed, ErrorDetails::SyncConnectionFailed, ErrorDetails::Config, ErrorDetails::BadRequest, @@ -500,6 +525,62 @@ mod tests { assert_eq!(roundtrip(&v), v); } + #[test] + fn data_exception_roundtrip() { + let v = ErrorDetails::DataException { + detail: "vector_distance(): vector dimension mismatch: expected 3, got 2".into(), + }; + assert_eq!(roundtrip(&v), v); + } + + #[test] + fn program_limit_exceeded_roundtrip() { + let v = ErrorDetails::ProgramLimitExceeded { + detail: "WITH RECURSIVE CTE 'walk' exceeded max recursion depth 100".into(), + }; + assert_eq!(roundtrip(&v), v); + } + + #[test] + fn transaction_rollback_roundtrip() { + let v = ErrorDetails::TransactionRollback { + detail: "a participant vShard returned an error".into(), + }; + assert_eq!(roundtrip(&v), v); + } + + #[test] + fn active_sql_transaction_roundtrip() { + let v = ErrorDetails::ActiveSqlTransaction { + detail: "VACUUM cannot run inside a transaction block".into(), + }; + assert_eq!(roundtrip(&v), v); + } + + #[test] + fn dependent_objects_exist_roundtrip() { + let v = ErrorDetails::DependentObjectsExist { + object: "role \"auditor\"".into(), + }; + assert_eq!(roundtrip(&v), v); + } + + /// The details of each transaction and dependency code decode to the + /// same variant, which answers the same numeric code. + #[test] + fn transaction_and_dependency_details_keep_their_code() { + use crate::error::NodeDbError; + for e in [ + NodeDbError::transaction_rollback("participant failed"), + NodeDbError::active_sql_transaction("VACUUM"), + NodeDbError::dependent_objects_exist("role \"r\"", "held by users: alice"), + ] { + let back = roundtrip(e.details()); + assert_eq!(&back, e.details()); + assert_eq!(back.code(), e.code()); + } + } + #[test] fn bridge_enriched_roundtrip() { let v = ErrorDetails::Bridge { diff --git a/nodedb-types/src/error/msgpack/encode.rs b/nodedb-types/src/error/msgpack/encode.rs index dfb983f48..77f26b7e1 100644 --- a/nodedb-types/src/error/msgpack/encode.rs +++ b/nodedb-types/src/error/msgpack/encode.rs @@ -136,6 +136,12 @@ impl ToMessagePack for ErrorDetails { write1(writer, TAG_INSUFFICIENT_BALANCE, collection) } ErrorDetails::RateExceeded { gate } => write1(writer, TAG_RATE_EXCEEDED, gate), + ErrorDetails::TransactionRollback { detail } => { + write1(writer, TAG_TRANSACTION_ROLLBACK, detail) + } + ErrorDetails::ActiveSqlTransaction { detail } => { + write1(writer, TAG_ACTIVE_SQL_TRANSACTION, detail) + } ErrorDetails::CollectionNotFound { collection } => { write1(writer, TAG_COLLECTION_NOT_FOUND, collection) } @@ -178,6 +184,10 @@ impl ToMessagePack for ErrorDetails { write1(writer, TAG_AMBIGUOUS_COLUMN, column) } ErrorDetails::DivisionByZero => write_unit(writer, TAG_DIVISION_BY_ZERO), + ErrorDetails::DataException { detail } => write1(writer, TAG_DATA_EXCEPTION, detail), + ErrorDetails::ProgramLimitExceeded { detail } => { + write1(writer, TAG_PROGRAM_LIMIT_EXCEEDED, detail) + } ErrorDetails::InvalidLimitValue { clause, value } => { write2(writer, TAG_INVALID_LIMIT_VALUE, clause, value) } @@ -185,6 +195,7 @@ impl ToMessagePack for ErrorDetails { write1(writer, TAG_AUTHORIZATION_DENIED, resource) } ErrorDetails::AuthExpired => write_unit(writer, TAG_AUTH_EXPIRED), + ErrorDetails::AuthenticationFailed => write_unit(writer, TAG_AUTHENTICATION_FAILED), ErrorDetails::HandshakeFailed { server_code } => { write1(writer, TAG_HANDSHAKE_FAILED, server_code) } @@ -325,6 +336,9 @@ impl ToMessagePack for ErrorDetails { ErrorDetails::AlreadyExists { object } => write1(writer, TAG_ALREADY_EXISTS, object), ErrorDetails::ObjectNotReady { object } => write1(writer, TAG_OBJECT_NOT_READY, object), ErrorDetails::NotFound { detail } => write1(writer, TAG_NOT_FOUND, detail), + ErrorDetails::DependentObjectsExist { object } => { + write1(writer, TAG_DEPENDENT_OBJECTS_EXIST, object) + } ErrorDetails::CannotDropDefaultDatabase => { write_unit(writer, TAG_CANNOT_DROP_DEFAULT_DATABASE) } diff --git a/nodedb-types/src/error/sqlstate.rs b/nodedb-types/src/error/sqlstate.rs index cc8a8899c..79aa86d17 100644 --- a/nodedb-types/src/error/sqlstate.rs +++ b/nodedb-types/src/error/sqlstate.rs @@ -49,6 +49,11 @@ pub const CANNOT_DROP_DEFAULT_DATABASE: AmbiguousSqlstate = AmbiguousSqlstate("0 // ── Class 22 — Data Exception ──────────────────────────────────────────────── +/// `22000` — `data_exception` (a function received a value it cannot compute +/// on: a vector of the wrong dimension, an argument of the wrong shape, a +/// malformed JSONPath) +pub const DATA_EXCEPTION: &str = "22000"; + /// `22003` — `numeric_value_out_of_range` pub const NUMERIC_VALUE_OUT_OF_RANGE: &str = "22003"; @@ -60,6 +65,10 @@ pub const DIVISION_BY_ZERO: &str = "22012"; /// negative, fractional, non-numeric, or wider than `usize`) pub const INVALID_LIMIT_VALUE: &str = "2201W"; +/// `22P02` — `invalid_text_representation` (stored text that does not +/// parse as the number an operation reads, e.g. `INCR` on `"abc"`) +pub const INVALID_TEXT_REPRESENTATION: &str = "22P02"; + /// `22023` — `invalid_parameter_value` (a `SET` value the parameter's own /// grammar refuses, e.g. `SET statement_timeout = 'later'`) pub const INVALID_PARAMETER_VALUE: &str = "22023"; @@ -69,6 +78,10 @@ pub const INVALID_PARAMETER_VALUE: &str = "22023"; /// `23000` — `integrity_constraint_violation` (generic) pub const INTEGRITY_CONSTRAINT_VIOLATION: &str = "23000"; +/// `428C9` — `generated_always` (a write names a generated column; its +/// value is computed from other columns) +pub const GENERATED_ALWAYS: &str = "428C9"; + /// `23502` — `not_null_violation` pub const NOT_NULL_VIOLATION: &str = "23502"; @@ -111,9 +124,35 @@ pub const PERIOD_LOCK_MISCONFIGURED: &str = "23609"; // ── Class 28 — Invalid Authorization Specification ─────────────────────────── -/// `28000` — `invalid_authorization_specification` (no valid credentials) +/// `28000` — `invalid_authorization_specification`: no valid credentials. +/// The default meaning, a credential failure. Its code is +/// `AUTHENTICATION_FAILED`, the code `INVALID_PASSWORD` has too, so a client +/// cannot tell a wrong password from an unknown user. pub const INVALID_AUTHORIZATION: &str = "28000"; +/// `28P01` — `invalid_password`: a credential failure. Same code as +/// `INVALID_AUTHORIZATION`. +pub const INVALID_PASSWORD: &str = "28P01"; + +/// `28000` — the session's bearer token expired, and the client must +/// re-authenticate. Its code is `AUTH_EXPIRED`. Ambiguous with the +/// credential-failure meaning of `28000` — see [`AmbiguousSqlstate`]. +pub const AUTH_TOKEN_EXPIRED: AmbiguousSqlstate = AmbiguousSqlstate("28000"); + +/// `active_sql_transaction`: the statement cannot run inside a transaction +/// block. +pub const ACTIVE_SQL_TRANSACTION: &str = "25001"; + +/// `25006` — `read_only_sql_transaction`: a write targeted a read-only +/// database, such as an unpromoted mirror. +pub const READ_ONLY_SQL_TRANSACTION: &str = "25006"; + +// ── Class 2B — Dependent Privilege Descriptors Still Exist ─────────────────── + +/// `2BP01` — `dependent_objects_still_exist`: a DROP names an object that +/// other objects depend on. +pub const DEPENDENT_OBJECTS_STILL_EXIST: &str = "2BP01"; + // ── Class 3D — Invalid Catalog Name ────────────────────────────────────────── /// `3D000` — `invalid_catalog_name` (the selected database does not exist) @@ -140,6 +179,9 @@ pub const SYNTAX_ERROR: &str = "42601"; /// carry, e.g. `SET nonsense = 1` or `SHOW nonsense`) pub const UNDEFINED_OBJECT: &str = "42704"; +/// `42710` — `duplicate_object`: a named catalog object already exists. +pub const DUPLICATE_OBJECT: &str = "42710"; + /// `42703` — `undefined_column` (a column reference that resolves against no /// relation in scope) pub const UNDEFINED_COLUMN: &str = "42703"; @@ -192,8 +234,10 @@ pub const LOCK_NOT_AVAILABLE: &str = "55P03"; // ── Class 57 — Operator Intervention ───────────────────────────────────────── -/// `57014` — `query_canceled` (deadline exceeded) -pub const QUERY_CANCELED: &str = "57014"; +/// `57014` — `query_canceled`: a deadline passed, or a user or statement +/// cancelled the query. No meaning is the default, so a bare `57014` never +/// classifies as the retriable deadline code — see [`AmbiguousSqlstate`]. +pub const QUERY_CANCELED: AmbiguousSqlstate = AmbiguousSqlstate("57014"); /// `57P03` — `cannot_connect_now` (collection is draining) pub const CANNOT_CONNECT_NOW: &str = "57P03"; @@ -285,14 +329,15 @@ pub const BACKUP_TENANT_MISMATCH: &str = "22023"; /// `28000` — NodeDB extension: the backup envelope did not decrypt under this /// server's configured `backup_encryption` key. Aliased to /// `invalid_authorization_specification`, the same class used for any other -/// presented-key mismatch. -pub const BACKUP_KEY_MISMATCH: &str = "28000"; +/// presented-key mismatch. Ambiguous with the credential-failure meaning of +/// `28000` — see [`AmbiguousSqlstate`]. +pub const BACKUP_KEY_MISMATCH: AmbiguousSqlstate = AmbiguousSqlstate("28000"); // ── Move Tenant DDL (Class 55 / 57) ───────────────────────────────────────── /// `57014` — `query_canceled`: drain phase timed out; client should re-try after /// ensuring the tenant has no active connections on the source database. -/// Ambiguous with `QUERY_CANCELED` (deadline exceeded) — see +/// Ambiguous with the other meanings of `QUERY_CANCELED` — see /// [`AmbiguousSqlstate`]. pub const MOVE_TENANT_DRAIN_TIMEOUT: AmbiguousSqlstate = AmbiguousSqlstate("57014"); @@ -322,6 +367,13 @@ pub const MOVE_TENANT_ALREADY_AT_TARGET: AmbiguousSqlstate = AmbiguousSqlstate(" /// reach the target core (bridge closed, core panic, timeout). pub const CONNECTION_FAILURE: &str = "08006"; +/// `08004` — `sqlserver_rejected_establishment_of_sqlconnection`: the server +/// refused to establish a sync shape subscription. +pub const SERVER_REJECTED_ESTABLISHMENT: &str = "08004"; + +/// `08P01` — `protocol_violation`: the protocol handshake failed. +pub const PROTOCOL_VIOLATION: &str = "08P01"; + // ── Class 58 — System Error ────────────────────────────────────────────────── /// `58030` — `io_error`: a WAL append or filesystem operation failed. @@ -343,10 +395,13 @@ mod tests { WARNING, NO_DATA, FEATURE_NOT_SUPPORTED, + DATA_EXCEPTION, NUMERIC_VALUE_OUT_OF_RANGE, DIVISION_BY_ZERO, INVALID_LIMIT_VALUE, + INVALID_TEXT_REPRESENTATION, INTEGRITY_CONSTRAINT_VIOLATION, + GENERATED_ALWAYS, NOT_NULL_VIOLATION, FOREIGN_KEY_VIOLATION, UNIQUE_VIOLATION, @@ -361,6 +416,10 @@ mod tests { LEGAL_HOLD_ACTIVE, TYPE_GUARD_VIOLATION, INVALID_AUTHORIZATION, + INVALID_PASSWORD, + ACTIVE_SQL_TRANSACTION, + READ_ONLY_SQL_TRANSACTION, + DEPENDENT_OBJECTS_STILL_EXIST, SERIALIZATION_FAILURE, INSUFFICIENT_PRIVILEGE, SYNTAX_ERROR, @@ -378,7 +437,6 @@ mod tests { STATEMENT_TOO_COMPLEX, OBJECT_NOT_IN_PREREQUISITE_STATE, LOCK_NOT_AVAILABLE, - QUERY_CANCELED, CANNOT_CONNECT_NOW, DATABASE_DROPPED, INTERNAL_ERROR, @@ -389,6 +447,9 @@ mod tests { CLONE_PREDATES_QUERY_TIME, STALE_READ_NOT_LEADER, CONNECTION_FAILURE, + SERVER_REJECTED_ESTABLISHMENT, + PROTOCOL_VIOLATION, + DUPLICATE_OBJECT, IO_ERROR, ]; for code in &codes { @@ -400,6 +461,9 @@ mod tests { } let ambiguous = [ + QUERY_CANCELED, + AUTH_TOKEN_EXPIRED, + BACKUP_KEY_MISMATCH, CANNOT_DROP_DEFAULT_DATABASE, CANNOT_CLONE_MIRROR, CLONE_DEPENDENCY, @@ -426,7 +490,7 @@ mod tests { assert_eq!(AMBIGUOUS_COLUMN, "42702"); assert_eq!(UNDEFINED_TABLE, "42P01"); assert_eq!(INSUFFICIENT_PRIVILEGE, "42501"); - assert_eq!(QUERY_CANCELED, "57014"); + assert_eq!(QUERY_CANCELED.0, "57014"); assert_eq!(INTERNAL_ERROR, "XX000"); assert_eq!(FEATURE_NOT_SUPPORTED, "0A000"); } diff --git a/nodedb-types/src/error/types.rs b/nodedb-types/src/error/types.rs index be809a036..3baf7beb6 100644 --- a/nodedb-types/src/error/types.rs +++ b/nodedb-types/src/error/types.rs @@ -68,6 +68,7 @@ impl NodeDbError { matches!( self.details, ErrorDetails::WriteConflict { .. } + | ErrorDetails::TransactionRollback { .. } | ErrorDetails::DeadlineExceeded | ErrorDetails::NoLeader | ErrorDetails::NotLeader { .. } @@ -96,12 +97,17 @@ impl NodeDbError { | ErrorDetails::DocumentNotFound { .. } | ErrorDetails::AuthorizationDenied { .. } | ErrorDetails::AuthExpired + | ErrorDetails::AuthenticationFailed | ErrorDetails::Config | ErrorDetails::SqlNotEnabled | ErrorDetails::UndefinedFunction { .. } | ErrorDetails::UndefinedColumn { .. } | ErrorDetails::AmbiguousColumn { .. } | ErrorDetails::DivisionByZero + | ErrorDetails::DataException { .. } + | ErrorDetails::ProgramLimitExceeded { .. } + | ErrorDetails::ActiveSqlTransaction { .. } + | ErrorDetails::DependentObjectsExist { .. } | ErrorDetails::InvalidLimitValue { .. } | ErrorDetails::BackupTenantMismatch { .. } | ErrorDetails::BackupKeyMismatch @@ -234,6 +240,26 @@ mod tests { assert!(!NodeDbError::internal("oops").is_client_error()); } + /// A participant rollback is retriable. A statement refused inside a + /// transaction block and a drop refused by dependents are client errors. + #[test] + fn transaction_and_dependency_codes_classify() { + let rollback = NodeDbError::transaction_rollback("participant failed"); + assert!(rollback.is_retriable()); + assert!(!rollback.is_client_error()); + assert_eq!(rollback.code(), ErrorCode::TRANSACTION_ROLLBACK); + + let in_block = NodeDbError::active_sql_transaction("VACUUM"); + assert!(in_block.is_client_error()); + assert!(!in_block.is_retriable()); + assert_eq!(in_block.code(), ErrorCode::ACTIVE_SQL_TRANSACTION); + + let dependents = NodeDbError::dependent_objects_exist("role \"r\"", "held by users"); + assert!(dependents.is_client_error()); + assert!(!dependents.is_retriable()); + assert_eq!(dependents.code(), ErrorCode::DEPENDENT_OBJECTS_EXIST); + } + #[test] fn json_serialization() { let e = NodeDbError::collection_not_found("users"); diff --git a/nodedb-types/src/fail_point.rs b/nodedb-types/src/fail_point.rs index 1d8bab0a3..88948da0e 100644 --- a/nodedb-types/src/fail_point.rs +++ b/nodedb-types/src/fail_point.rs @@ -15,6 +15,10 @@ //! [`fail_point_err!`] can honour this — the call site supplies the //! mapping into its own error type, so no crate has to know about //! anyone else's. +//! - `WaitForFile(path)`: park the call site until `path` exists, so a test +//! releases it at the moment it chooses. Only an async call site can +//! honour this: it awaits the file with the crate's own timer and parks +//! only its own task. A synchronous call site refuses it. //! //! The framework is deliberately tiny — no fail-rs dep, no parsing of env //! vars, no list of probabilities. Tests install actions explicitly via @@ -47,6 +51,9 @@ mod imp { /// Return an error from the injected call site, carrying this detail. /// Ignored by bare `fail_point!` — use `fail_point_err!`. Fail(String), + /// Park the call site until this file exists. Only an async call site + /// that looks the action up with [`lookup`] can honour it. + WaitForFile(std::path::PathBuf), } /// Environment variable read once, the first time any fail point is @@ -54,7 +61,8 @@ mod imp { /// in-process `set` API cannot reach a server the test only supervises. /// /// Format: comma-separated `name=action`, where action is `panic`, - /// `sleep()`, or `fail()`. For example: + /// `abort`, `sleep()`, `fail()`, or `wait_file()`. + /// For example: /// `NODEDB_FAILPOINTS='checkpoint::after_marker_before_truncate=panic'` pub const FAILPOINTS_ENV: &str = "NODEDB_FAILPOINTS"; @@ -85,6 +93,11 @@ mod imp { rest if rest.starts_with("fail(") && rest.ends_with(')') => { FailAction::Fail(rest["fail(".len()..rest.len() - 1].to_string()) } + rest if rest.starts_with("wait_file(") && rest.ends_with(')') => { + FailAction::WaitForFile(std::path::PathBuf::from( + &rest["wait_file(".len()..rest.len() - 1], + )) + } other => panic!("{FAILPOINTS_ENV} entry {entry:?} has unknown action {other:?}"), }; actions.insert(name.trim().to_string(), action); @@ -134,6 +147,12 @@ mod imp { "fail_point {name} installed Fail({detail}) but the call site cannot return an error — use fail_point_err!" ) } + // Blocking the thread would stall every task that shares it. + FailAction::WaitForFile(path) => panic!( + "fail_point {name} installed WaitForFile({}) but the call site is \ + synchronous — only an async call site can park", + path.display() + ), } } } @@ -157,6 +176,11 @@ mod imp { std::thread::sleep(d); None } + Some(FailAction::WaitForFile(path)) => panic!( + "fail_point {name} installed WaitForFile({}) but the call site is synchronous \ + — only an async call site can park", + path.display() + ), None => None, } } @@ -194,10 +218,10 @@ pub use imp::{FAILPOINTS_ENV, FailAction, FailGuard, clear, eval, eval_fail, loo /// Inject a fail point. Expands to nothing without the `failpoints` feature. /// /// Usage in production code: -/// `nodedb_types::fail_point!("transaction_batch::between_subapply");` +/// `nodedb_types::fail_point!("calvin_static::during_overlay_stage");` /// /// Tests opt in by enabling the feature and installing actions: -/// `fail_point::set("transaction_batch::between_subapply", +/// `fail_point::set("calvin_static::during_overlay_stage", /// fail_point::FailAction::Panic);` #[macro_export] macro_rules! fail_point { @@ -279,6 +303,25 @@ mod tests { )); } + #[test] + fn env_spec_parses_a_file_gate() { + let actions = super::imp::parse_env(Some("d::gate=wait_file(/tmp/release-d)")); + assert!(matches!( + actions.get("d::gate"), + Some(FailAction::WaitForFile(path)) if path == std::path::Path::new("/tmp/release-d") + )); + } + + #[test] + #[should_panic(expected = "only an async call site can park")] + fn a_file_gate_at_a_synchronous_call_site_is_loud() { + let _g = FailGuard::install( + "nodedb::test::gate_at_sync", + FailAction::WaitForFile(std::path::PathBuf::from("/nonexistent")), + ); + eval("nodedb::test::gate_at_sync"); + } + #[test] fn empty_env_spec_arms_nothing() { assert!(super::imp::parse_env(None).is_empty()); diff --git a/nodedb-types/src/id/collection_key.rs b/nodedb-types/src/id/collection_key.rs new file mode 100644 index 000000000..7dfac715d --- /dev/null +++ b/nodedb-types/src/id/collection_key.rs @@ -0,0 +1,209 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Canonical collection identity for placement and surrogate identity. +//! +//! A collection's vShard and its surrogate bindings are keyed by the pair +//! `(database_id, bare_name)`. The bare name is the name the catalog keys the +//! collection by. The database-qualified form `"{database_id}/{name}"` names +//! the same collection, but it folds the database into the string a second +//! time. Hashing it gives a different vShard than hashing the bare name. +//! +//! [`CollectionKey`] is the only input the vShard hash and the surrogate +//! allocator accept. It has no `From<&str>`. A caller builds it from a bare +//! catalog name with [`CollectionKey::from_bare`], or from a qualified name +//! with [`CollectionKey::from_qualified`], which strips the qualifier. + +use super::{DatabaseId, QualifiedCollection}; + +/// Error returned when a qualified collection name does not carry the +/// qualifier of the database it is resolved in. +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +pub enum CollectionKeyError { + /// The name lacks the `"{database_id}/"` prefix that + /// [`QualifiedCollection::new`] writes for a non-default database. + #[error( + "collection name '{name}' is not qualified for database {database_id}; \ + expected the prefix '{database_id}/'" + )] + NotQualified { + /// Raw id of the database the name was resolved in. + database_id: u64, + /// The rejected name. + name: String, + }, +} + +/// The canonical `(database_id, bare_name)` identity of a collection. +/// +/// Borrowed: it never allocates. Build it with [`Self::from_bare`] from a +/// catalog name, or with [`Self::from_qualified`] / +/// [`Self::from_qualified_str`] from a database-qualified name. +/// +/// No `From<&str>` impl exists, so a raw string never converts implicitly: +/// +/// ```compile_fail +/// use nodedb_types::CollectionKey; +/// let key: CollectionKey<'_> = "1024/users".into(); +/// ``` +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub struct CollectionKey<'a> { + database_id: DatabaseId, + name: &'a str, +} + +impl<'a> CollectionKey<'a> { + /// Build a key from the bare name the catalog keys the collection by. + /// + /// `name` must be the catalog name, never a `"{database_id}/{name}"` + /// string. A qualified name goes through [`Self::from_qualified`]. + pub fn from_bare(database_id: DatabaseId, name: &'a str) -> Self { + Self { database_id, name } + } + + /// Build a key from a [`QualifiedCollection`], stripping its qualifier. + pub fn from_qualified( + database_id: DatabaseId, + qualified: &'a QualifiedCollection, + ) -> Result { + Self::from_qualified_str(database_id, qualified.as_str()) + } + + /// Build a key from a database-qualified name carried as a string on a + /// plan, a WAL record, or the wire. + /// + /// The default database stores names unqualified, so the name is taken + /// as-is. The empty name marks a plan with no routing collection and is + /// the same in both forms, so it is also taken as-is. Any other name in + /// any other database requires the exact `"{database_id}/"` prefix + /// [`QualifiedCollection::new`] writes. Exactly one prefix is stripped, + /// so a bare name that itself contains `/` survives intact. + pub fn from_qualified_str( + database_id: DatabaseId, + qualified: &'a str, + ) -> Result { + if database_id == DatabaseId::DEFAULT || qualified.is_empty() { + return Ok(Self::from_bare(database_id, qualified)); + } + let not_qualified = || CollectionKeyError::NotQualified { + database_id: database_id.as_u64(), + name: qualified.to_owned(), + }; + let (head, bare) = qualified.split_once('/').ok_or_else(not_qualified)?; + if head.is_empty() || !head.bytes().all(|b| b.is_ascii_digit()) { + return Err(not_qualified()); + } + match head.parse::() { + Ok(raw) if raw == database_id.as_u64() && !head.starts_with('0') => { + Ok(Self::from_bare(database_id, bare)) + } + _ => Err(not_qualified()), + } + } + + /// The database the collection lives in. + pub fn database_id(&self) -> DatabaseId { + self.database_id + } + + /// The bare catalog name. + pub fn name(&self) -> &'a str { + self.name + } + + /// The database-qualified name storage engines key data by. + pub fn qualified(&self) -> QualifiedCollection { + QualifiedCollection::new(self.database_id, self.name) + } + + /// The vShard this collection homes to. + pub fn vshard(&self) -> super::VShardId { + super::VShardId::from_collection(*self) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + const DB: DatabaseId = DatabaseId::new(1024); + + #[test] + fn qualified_input_yields_the_bare_key() { + let qualified = QualifiedCollection::new(DB, "users"); + let from_qualified = CollectionKey::from_qualified(DB, &qualified).expect("qualified"); + assert_eq!(from_qualified.name(), "users"); + assert_eq!(from_qualified, CollectionKey::from_bare(DB, "users")); + assert_eq!( + from_qualified.vshard(), + CollectionKey::from_bare(DB, "users").vshard() + ); + } + + #[test] + fn qualified_string_cannot_become_a_key_without_dequalifying() { + // The only string-accepting constructors are `from_bare`, which the + // caller names explicitly, and the qualified constructors, which strip + // the qualifier. The qualified path therefore always lands on the bare + // name, never on the qualified string. + let qualified = QualifiedCollection::new(DB, "users"); + let key = CollectionKey::from_qualified_str(DB, qualified.as_str()).expect("qualified"); + assert_ne!(key.name(), qualified.as_str()); + assert_eq!(key.name(), "users"); + assert_eq!(key.qualified(), qualified); + } + + #[test] + fn default_database_names_are_already_bare() { + let key = CollectionKey::from_qualified_str(DatabaseId::DEFAULT, "users").expect("default"); + assert_eq!(key, CollectionKey::from_bare(DatabaseId::DEFAULT, "users")); + } + + #[test] + fn empty_name_is_the_no_collection_sentinel_in_every_database() { + let key = CollectionKey::from_qualified_str(DB, "").expect("empty"); + assert_eq!(key, CollectionKey::from_bare(DB, "")); + } + + #[test] + fn exactly_one_qualifier_is_stripped() { + let qualified = QualifiedCollection::new(DB, "1024/nested"); + let key = CollectionKey::from_qualified(DB, &qualified).expect("qualified"); + assert_eq!(key.name(), "1024/nested"); + } + + #[test] + fn unqualified_name_in_a_named_database_is_rejected() { + let err = CollectionKey::from_qualified_str(DB, "users").expect_err("unqualified"); + assert_eq!( + err, + CollectionKeyError::NotQualified { + database_id: 1024, + name: "users".to_owned(), + } + ); + } + + #[test] + fn foreign_database_qualifier_is_rejected() { + assert!(CollectionKey::from_qualified_str(DB, "7/users").is_err()); + assert!(CollectionKey::from_qualified_str(DB, "+1024/users").is_err()); + assert!(CollectionKey::from_qualified_str(DB, "01024/users").is_err()); + assert!(CollectionKey::from_qualified_str(DB, "/users").is_err()); + } + + #[test] + fn bare_and_qualified_hashes_differ_for_a_named_database() { + // The two strings name one collection. Only the key keeps them on one + // vShard, so hashing the qualified string directly is a routing error. + let mut differs = false; + for i in 0..64 { + let name = format!("coll_{i}"); + let qualified = QualifiedCollection::new(DB, &name); + let key = CollectionKey::from_qualified(DB, &qualified).expect("qualified"); + let raw_qualified = CollectionKey::from_bare(DB, qualified.as_str()); + assert_eq!(key.vshard(), CollectionKey::from_bare(DB, &name).vshard()); + differs |= key.vshard() != raw_qualified.vshard(); + } + assert!(differs); + } +} diff --git a/nodedb-types/src/id/mod.rs b/nodedb-types/src/id/mod.rs index 51c792d1b..138b82395 100644 --- a/nodedb-types/src/id/mod.rs +++ b/nodedb-types/src/id/mod.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: Apache-2.0 pub mod collection; +pub mod collection_key; pub mod database; pub mod document; pub mod edge; @@ -15,6 +16,7 @@ pub mod txn; pub mod vshard; pub use collection::CollectionId; +pub use collection_key::{CollectionKey, CollectionKeyError}; pub use database::DatabaseId; pub use document::DocumentId; pub use edge::{EdgeId, EdgeIdParseError}; diff --git a/nodedb-types/src/id/vshard.rs b/nodedb-types/src/id/vshard.rs index 2ac085db8..0230f2a07 100644 --- a/nodedb-types/src/id/vshard.rs +++ b/nodedb-types/src/id/vshard.rs @@ -39,18 +39,19 @@ impl VShardId { self.0 } - /// Compute vShard from a database + collection name pair. + /// Compute the vShard a collection homes to. /// - /// The database identity is mixed into the hash so that the same collection - /// name in two different databases routes to independent vShards. Uses a - /// DJB-like multiply-31 hash, seeded with the database id bytes, followed - /// by a zero separator byte, followed by the collection name bytes. - pub fn from_collection_in_database(db: crate::id::DatabaseId, collection: &str) -> Self { - let db_bytes = db.as_u64().to_le_bytes(); + /// Takes a [`CollectionKey`](crate::id::CollectionKey), so the hashed name + /// is always the bare catalog name. The database identity is mixed into + /// the hash, so the same name in two databases routes to independent + /// vShards. Uses a DJB-like multiply-31 hash, seeded with the database id + /// bytes, then a zero separator byte, then the bare name bytes. + pub fn from_collection(key: crate::id::CollectionKey<'_>) -> Self { + let db_bytes = key.database_id().as_u64().to_le_bytes(); let hash = db_bytes .iter() .chain(std::iter::once(&0u8)) - .chain(collection.as_bytes().iter()) + .chain(key.name().as_bytes().iter()) .fold(0u32, |h, &b| h.wrapping_mul(31).wrapping_add(b as u32)); Self::new(hash % Self::COUNT) } @@ -114,25 +115,25 @@ mod tests { } #[test] - fn from_collection_in_database_deterministic() { - use crate::id::DatabaseId; + fn from_collection_deterministic() { + use crate::id::{CollectionKey, DatabaseId}; let db = DatabaseId::new(1024); - let a = VShardId::from_collection_in_database(db, "users"); - let b = VShardId::from_collection_in_database(db, "users"); + let a = VShardId::from_collection(CollectionKey::from_bare(db, "users")); + let b = VShardId::from_collection(CollectionKey::from_bare(db, "users")); assert_eq!(a, b); assert!(a.as_u32() < VShardId::COUNT); } #[test] - fn from_collection_in_database_different_dbs_differ() { - use crate::id::DatabaseId; + fn from_collection_different_dbs_differ() { + use crate::id::{CollectionKey, DatabaseId}; let db0 = DatabaseId::DEFAULT; let db1 = DatabaseId::new(1024); // Same collection name in different databases should typically route // to different vShards (probabilistic; collection "users" is a // canonical example and the two hashes are known to differ). - let a = VShardId::from_collection_in_database(db0, "users"); - let b = VShardId::from_collection_in_database(db1, "users"); + let a = VShardId::from_collection(CollectionKey::from_bare(db0, "users")); + let b = VShardId::from_collection(CollectionKey::from_bare(db1, "users")); assert_ne!( a, b, "same collection name, different databases should route differently" @@ -140,9 +141,9 @@ mod tests { } #[test] - fn from_collection_in_database_default_in_range() { - use crate::id::DatabaseId; - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, "orders"); + fn from_collection_default_in_range() { + use crate::id::{CollectionKey, DatabaseId}; + let v = VShardId::from_collection(CollectionKey::from_bare(DatabaseId::DEFAULT, "orders")); assert!(v.as_u32() < VShardId::COUNT); } } diff --git a/nodedb-types/src/lib.rs b/nodedb-types/src/lib.rs index 68f14647e..452c538d3 100644 --- a/nodedb-types/src/lib.rs +++ b/nodedb-types/src/lib.rs @@ -100,8 +100,8 @@ pub use graph::{Direction, GraphStats}; pub use hlc::{ClockSkew, Hlc, HlcClock, MAX_CLOCK_SKEW_NS}; pub use hnsw::{HnswCheckpoint, HnswNodeSnapshot, HnswParams}; pub use id::{ - CollectionId, DatabaseId, DocumentId, EdgeId, EdgeIdParseError, IdError, IdType, NodeId, - QualifiedCollection, ShapeId, TenantId, + CollectionId, CollectionKey, CollectionKeyError, DatabaseId, DocumentId, EdgeId, + EdgeIdParseError, IdError, IdType, NodeId, QualifiedCollection, ShapeId, TenantId, }; pub use identity::KeyRepr; pub use json_msgpack::{ @@ -144,6 +144,8 @@ pub use value::{NotScalar, Value, scalar_to_raw_bytes}; pub use vector_ann::{VectorAnnOptions, VectorQuantization}; pub use vector_dtype::VectorStorageDtype; pub use vector_index_params::StoredVectorIndexParams; -pub use vector_index_stats::{VectorIndexQuantization, VectorIndexStats, VectorIndexType}; +pub use vector_index_stats::{ + VectorIndexQuantization, VectorIndexStats, VectorIndexType, VectorIvfStats, +}; pub use vector_model::{VectorModelEntry, VectorModelMetadata}; pub use volatility::Volatility; diff --git a/nodedb-types/src/namespace.rs b/nodedb-types/src/namespace.rs index 15f034c90..8ac6e727f 100644 --- a/nodedb-types/src/namespace.rs +++ b/nodedb-types/src/namespace.rs @@ -114,6 +114,10 @@ pub enum Namespace { /// transport to Origin. Keys are big-endian monotonic u64 IDs; values are /// zerompk-encoded `PendingSpatialDelete` payloads. SpatialDeletePending = 24, + /// Durable FIFO queue of outbound KV writes waiting for transport to + /// Origin. Keys are big-endian monotonic u64 IDs; values are + /// zerompk-encoded `PendingKvWrite` payloads. + KvPushPending = 25, } impl Namespace { @@ -145,6 +149,7 @@ impl Namespace { 22 => Some(Self::FtsDeletePending), 23 => Some(Self::SpatialInsertPending), 24 => Some(Self::SpatialDeletePending), + 25 => Some(Self::KvPushPending), _ => None, } } @@ -156,10 +161,10 @@ mod tests { #[test] fn namespace_roundtrip() { - for v in 0u8..=24 { + for v in 0u8..=25 { let ns = Namespace::from_u8(v).unwrap(); assert_eq!(ns as u8, v); } - assert!(Namespace::from_u8(25).is_none()); + assert!(Namespace::from_u8(26).is_none()); } } diff --git a/nodedb-types/src/protocol/error_cause.rs b/nodedb-types/src/protocol/error_cause.rs new file mode 100644 index 000000000..838c8b2c3 --- /dev/null +++ b/nodedb-types/src/protocol/error_cause.rs @@ -0,0 +1,71 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The typed cause an error frame carries alongside its own classification. + +use serde::{Deserialize, Serialize}; + +use crate::error::{ErrorCode, ErrorDetails, NodeDbError}; + +/// The typed error that caused the one an error frame reports. +/// +/// A phase error such as `MOVE_TENANT_SNAPSHOT_FAILED` keeps its own code, +/// and the Data-Plane refusal that failed the phase rides here with its own +/// code and details. A client rebuilds it as the typed error's +/// [`NodeDbError::cause`]. +#[derive( + Debug, + Clone, + PartialEq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +#[msgpack(map)] +pub struct ErrorCausePayload { + /// Stable numeric NodeDB code of the cause. + pub ndb_code: u16, + /// Human-readable message of the cause. + pub message: String, + /// Structured details of the cause, when it had any. + #[serde(default, skip_serializing_if = "Option::is_none")] + #[msgpack(default)] + pub details: Option, +} + +impl From<&NodeDbError> for ErrorCausePayload { + fn from(error: &NodeDbError) -> Self { + Self { + ndb_code: error.code().0, + message: error.message().to_owned(), + details: Some(error.details().clone()), + } + } +} + +impl ErrorCausePayload { + /// Rebuild the typed cause. + pub fn to_error(&self) -> NodeDbError { + NodeDbError::from_wire_with_details( + ErrorCode(self.ndb_code), + self.message.clone(), + self.details.clone(), + ) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn cause_round_trips_its_class() { + let cause = NodeDbError::division_by_zero(); + let payload = ErrorCausePayload::from(&cause); + let bytes = zerompk::to_msgpack_vec(&payload).expect("encode"); + let decoded: ErrorCausePayload = zerompk::from_msgpack(&bytes).expect("decode"); + let rebuilt = decoded.to_error(); + assert_eq!(rebuilt.code(), ErrorCode::DIVISION_BY_ZERO); + assert_eq!(rebuilt.details(), cause.details()); + } +} diff --git a/nodedb-types/src/protocol/frames.rs b/nodedb-types/src/protocol/frames.rs index 233cd1cae..827ee1db9 100644 --- a/nodedb-types/src/protocol/frames.rs +++ b/nodedb-types/src/protocol/frames.rs @@ -125,6 +125,17 @@ pub struct ErrorPayload { #[serde(default, skip_serializing_if = "is_zero")] #[msgpack(default)] pub ndb_code: u16, + /// The structured details the server held: the collection, gate, or + /// document the error names. `None` when the server had none to send. + #[serde(default, skip_serializing_if = "Option::is_none")] + #[msgpack(default)] + pub details: Option, + /// The typed error that caused this one, when the server held one: the + /// Data-Plane refusal behind a phase failure, say. `None` when there is + /// no cause to report. + #[serde(default, skip_serializing_if = "Option::is_none")] + #[msgpack(default)] + pub cause: Option, } /// `skip_serializing_if` predicate for [`ErrorPayload::ndb_code`]: zero is the @@ -196,12 +207,32 @@ impl NativeResponse { code: code.into(), message: message.into(), ndb_code, + details: None, + cause: None, }), auth: None, warnings: Vec::new(), } } + /// Attach the structured error details to an error response. A response + /// with no error payload is returned unchanged. + pub fn with_error_details(mut self, details: crate::error::ErrorDetails) -> Self { + if let Some(payload) = self.error.as_mut() { + payload.details = Some(details); + } + self + } + + /// Attach the typed cause to an error response. A response with no error + /// payload is returned unchanged. + pub fn with_error_cause(mut self, cause: super::error_cause::ErrorCausePayload) -> Self { + if let Some(payload) = self.error.as_mut() { + payload.cause = Some(cause); + } + self + } + /// Create an auth success response. pub fn auth_ok(seq: u64, username: String, tenant_id: u64) -> Self { Self { @@ -331,6 +362,33 @@ mod tests { assert_eq!(payload.ndb_code, 1000); } + #[test] + fn error_payload_round_trips_the_details() { + let details = crate::error::ErrorDetails::Overflow { + collection: "counters".into(), + }; + let frame = NativeResponse::error_with_code(7, "22003", "overflow on counters", 1021) + .with_error_details(details.clone()); + let bytes = zerompk::to_msgpack_vec(&frame).expect("encode"); + let decoded: NativeResponse = zerompk::from_msgpack(&bytes).expect("decode"); + let payload = decoded.error.expect("error payload survives the wire"); + assert_eq!(payload.details, Some(details)); + } + + #[test] + fn error_payload_round_trips_the_cause() { + let cause = super::super::error_cause::ErrorCausePayload::from( + &crate::error::NodeDbError::division_by_zero(), + ); + let frame = NativeResponse::error_with_code(7, "XX000", "snapshot failed", 1602) + .with_error_cause(cause.clone()); + let bytes = zerompk::to_msgpack_vec(&frame).expect("encode"); + let decoded: NativeResponse = zerompk::from_msgpack(&bytes).expect("decode"); + let payload = decoded.error.expect("error payload survives the wire"); + assert_eq!(payload.ndb_code, 1602); + assert_eq!(payload.cause, Some(cause)); + } + #[test] fn error_payload_without_numeric_code_still_decodes() { // Hand-rolled 2-key map: exactly what a peer built before the numeric diff --git a/nodedb-types/src/protocol/mod.rs b/nodedb-types/src/protocol/mod.rs index 1034d15d3..8b16fcb9a 100644 --- a/nodedb-types/src/protocol/mod.rs +++ b/nodedb-types/src/protocol/mod.rs @@ -2,6 +2,7 @@ pub mod auth; pub mod batch; +pub mod error_cause; pub mod frames; pub mod handshake; pub mod opcodes; @@ -10,6 +11,7 @@ pub mod text_fields; pub use auth::{AuthMethod, AuthResponse}; pub use batch::{BatchDocument, BatchVector}; +pub use error_cause::ErrorCausePayload; pub use frames::{ErrorPayload, NativeRequest, NativeResponse}; pub use handshake::{ CAP_COLUMNAR, CAP_CRDT, CAP_FTS, CAP_GRAPHRAG, CAP_MSGPACK, CAP_SPATIAL, CAP_STREAMING, diff --git a/nodedb-types/src/protocol/text_fields/decode.rs b/nodedb-types/src/protocol/text_fields/decode.rs index 3e56301bd..93d6a4b76 100644 --- a/nodedb-types/src/protocol/text_fields/decode.rs +++ b/nodedb-types/src/protocol/text_fields/decode.rs @@ -234,7 +234,7 @@ impl<'a> zerompk::FromMessagePack<'a> for TextFields { out.incr_delta = Some(reader.read_i64()?); } FID_INCR_FLOAT_DELTA => { - out.incr_float_delta = Some(reader.read_f64()?); + out.incr_float_delta = Some(reader.read_string()?.into_owned()); } FID_EXPECTED => { out.expected = Some(reader.read_binary()?.into_owned()); diff --git a/nodedb-types/src/protocol/text_fields/types/text_fields.rs b/nodedb-types/src/protocol/text_fields/types/text_fields.rs index e129e2550..c789e8c63 100644 --- a/nodedb-types/src/protocol/text_fields/types/text_fields.rs +++ b/nodedb-types/src/protocol/text_fields/types/text_fields.rs @@ -171,9 +171,10 @@ pub struct TextFields { /// Integer delta for KvIncr. #[serde(skip_serializing_if = "Option::is_none")] pub incr_delta: Option, - /// Float delta for KvIncrFloat. + /// Delta for KvIncrFloat, as the client's decimal text, so no digit is + /// lost to an `f64` on the wire. #[serde(skip_serializing_if = "Option::is_none")] - pub incr_float_delta: Option, + pub incr_float_delta: Option, /// Expected value for KvCas. #[serde(skip_serializing_if = "Option::is_none")] pub expected: Option>, diff --git a/nodedb-types/src/sync/wire/ack_status.rs b/nodedb-types/src/sync/wire/ack_status.rs index 74469ecc7..7720ecc82 100644 --- a/nodedb-types/src/sync/wire/ack_status.rs +++ b/nodedb-types/src/sync/wire/ack_status.rs @@ -13,6 +13,7 @@ use serde::{Deserialize, Serialize}; Clone, Default, PartialEq, + Eq, Serialize, Deserialize, zerompk::ToMessagePack, diff --git a/nodedb-types/src/sync/wire/frame.rs b/nodedb-types/src/sync/wire/frame.rs index 702a71752..dbb9617f8 100644 --- a/nodedb-types/src/sync/wire/frame.rs +++ b/nodedb-types/src/sync/wire/frame.rs @@ -15,6 +15,9 @@ //! - `0xAB` SpatialInsertAck (server → client) //! - `0xAC` SpatialDelete (client → server) //! - `0xAD` SpatialDeleteAck (server → client) +//! - `0xAE` KvPush (client → server) +//! - `0xAF` KvPushAck (server → client) +//! - `0x16` RowPushReject (client → server) /// Sync message type identifiers. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -44,6 +47,10 @@ pub enum SyncMessageType { /// originated on the server (SQL DML, DDL-managed system rows), where no /// client-authored CRDT operation exists to replicate. RowPush = 0x15, + /// Row push refusal (client → server, 0x16). + /// + /// Lite sends this for a [`Self::RowPush`] it could not apply. + RowPushReject = 0x16, ShapeSubscribe = 0x20, ShapeSnapshot = 0x21, ShapeDelta = 0x22, @@ -116,6 +123,10 @@ pub enum SyncMessageType { SpatialDelete = 0xAC, /// Spatial delete acknowledgment (server → client, 0xAD). SpatialDeleteAck = 0xAD, + /// KV write push (client → server, 0xAE). + KvPush = 0xAE, + /// KV push acknowledgment (server → client, 0xAF). + KvPushAck = 0xAF, PingPong = 0xFF, } @@ -130,6 +141,7 @@ impl SyncMessageType { 0x13 => Some(Self::CollectionSchema), 0x14 => Some(Self::CollectionPurged), 0x15 => Some(Self::RowPush), + 0x16 => Some(Self::RowPushReject), 0x20 => Some(Self::ShapeSubscribe), 0x21 => Some(Self::ShapeSnapshot), 0x22 => Some(Self::ShapeDelta), @@ -167,6 +179,8 @@ impl SyncMessageType { 0xAB => Some(Self::SpatialInsertAck), 0xAC => Some(Self::SpatialDelete), 0xAD => Some(Self::SpatialDeleteAck), + 0xAE => Some(Self::KvPush), + 0xAF => Some(Self::KvPushAck), 0xFF => Some(Self::PingPong), _ => None, } diff --git a/nodedb-types/src/sync/wire/kv.rs b/nodedb-types/src/sync/wire/kv.rs new file mode 100644 index 000000000..13f0e22f6 --- /dev/null +++ b/nodedb-types/src/sync/wire/kv.rs @@ -0,0 +1,177 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! KV row push messages, and the refusal of a row push. +//! +//! `KvPushMsg` carries one KV write from a Lite client to Origin. A put +//! carries the `{key, value…}` row every KV read returns, the same shape +//! Origin sends in a `RowPushMsg`. Origin answers each push with a +//! `KvPushAckMsg`. A terminal refusal travels in that ack as +//! `AckStatus::Rejected`, the way every engine push ack carries one. +//! +//! `RowPushRejectMsg` is Lite's refusal of an Origin `RowPushMsg` it could +//! not apply. +//! +//! Wire opcodes: +//! - `0x16` — `RowPushReject` (Lite → Origin) +//! - `0xAE` — `KvPush` (Lite → Origin) +//! - `0xAF` — `KvPushAck` (Origin → Lite) + +use serde::{Deserialize, Serialize}; + +use crate::sync::wire::ack_status::AckStatus; + +/// The write a `KvPushMsg` carries. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum KvPushOp { + /// Store `row` at the entry's key. + Put { + /// The `{key, value…}` row as standard MessagePack. + row: Vec, + /// Absolute expiry in milliseconds since the Unix epoch. `0` means + /// the entry never expires. + expire_at_ms: u64, + }, + /// Remove the entry's key. + Delete, +} + +/// KV write push (Lite → Origin, 0xAE). +#[derive( + Debug, Clone, Serialize, Deserialize, zerompk::ToMessagePack, zerompk::FromMessagePack, +)] +pub struct KvPushMsg { + /// Lite instance ID. + pub lite_id: String, + /// Target KV collection. + pub collection: String, + /// The entry's key bytes. + pub key: Vec, + /// The write. + pub op: KvPushOp, + /// Lite-assigned ID for ACK correlation. + pub batch_id: u64, + /// Stable identity of the originating producer. + pub producer_id: u64, + /// Producer epoch. + pub epoch: u64, + /// Per-stream monotonic sequence number within the epoch. + pub seq: u64, +} + +/// KV write push acknowledgment (Origin → Lite, 0xAF). +#[derive( + Debug, Clone, Serialize, Deserialize, zerompk::ToMessagePack, zerompk::FromMessagePack, +)] +pub struct KvPushAckMsg { + /// Collection acknowledged. + pub collection: String, + /// Key from the originating `KvPushMsg`. + pub key: Vec, + /// Batch ID from the originating `KvPushMsg`. + pub batch_id: u64, + /// `true` unless `status` is `AckStatus::Rejected`. + pub accepted: bool, + /// Refusal detail when `status` is `AckStatus::Rejected`. + pub reject_reason: Option, + /// Highest sequence from this producer's stream that Origin applied. + pub applied_seq: u64, + /// Idempotency outcome of the push. + pub status: AckStatus, +} + +/// Why Lite refused an Origin row push. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum RowPushRefusal { + /// The payload is not the row shape the collection's engine expects. + Malformed { detail: String }, + /// The payload decoded, and the local write failed. + ApplyFailed { detail: String }, +} + +impl std::fmt::Display for RowPushRefusal { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Malformed { detail } => write!(f, "malformed row push: {detail}"), + Self::ApplyFailed { detail } => write!(f, "row push apply failed: {detail}"), + } + } +} + +/// Lite's refusal of an Origin `RowPushMsg` (Lite → Origin, 0x16). +#[derive( + Debug, Clone, Serialize, Deserialize, zerompk::ToMessagePack, zerompk::FromMessagePack, +)] +pub struct RowPushRejectMsg { + /// Collection of the refused row. + pub collection: String, + /// Document ID of the refused row. + pub document_id: String, + /// `sequence` of the refused `RowPushMsg`. + pub sequence: u64, + /// `peer_id` of the refused `RowPushMsg`. + pub peer_id: u64, + /// Why Lite refused the row. + pub refusal: RowPushRefusal, +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_put_push_round_trips() { + let msg = KvPushMsg { + lite_id: "lite-1".into(), + collection: "cfg".into(), + key: b"k1".to_vec(), + op: KvPushOp::Put { + row: vec![0x81, 0xa1, b'k', 0x01], + expire_at_ms: 42, + }, + batch_id: 7, + producer_id: 3, + epoch: 1, + seq: 9, + }; + let bytes = zerompk::to_msgpack_vec(&msg).expect("encode"); + let back: KvPushMsg = zerompk::from_msgpack(&bytes).expect("decode"); + assert_eq!(back.op, msg.op); + assert_eq!(back.key, msg.key); + assert_eq!(back.seq, 9); + } + + #[test] + fn a_row_push_reject_round_trips() { + let msg = RowPushRejectMsg { + collection: "cfg".into(), + document_id: "k1".into(), + sequence: 4, + peer_id: 2, + refusal: RowPushRefusal::Malformed { + detail: "not a row map".into(), + }, + }; + let bytes = zerompk::to_msgpack_vec(&msg).expect("encode"); + let back: RowPushRejectMsg = zerompk::from_msgpack(&bytes).expect("decode"); + assert_eq!(back.refusal, msg.refusal); + assert_eq!(back.sequence, 4); + } +} diff --git a/nodedb-types/src/sync/wire/mod.rs b/nodedb-types/src/sync/wire/mod.rs index ffdd23dd6..a6f56e679 100644 --- a/nodedb-types/src/sync/wire/mod.rs +++ b/nodedb-types/src/sync/wire/mod.rs @@ -12,6 +12,8 @@ //! - `0x12` DeltaReject (server → client) //! - `0x13` CollectionSchema (bidirectional) //! - `0x14` CollectionPurged (server → client) +//! - `0x15` RowPush (server → client) +//! - `0x16` RowPushReject (client → server) //! - `0x20` ShapeSubscribe (client → server) //! - `0x21` ShapeSnapshot (server → client) //! - `0x22` ShapeDelta (server → client) @@ -49,6 +51,8 @@ //! - `0xAB` SpatialInsertAck (server → client) //! - `0xAC` SpatialDelete (client → server) //! - `0xAD` SpatialDeleteAck (server → client) +//! - `0xAE` KvPush (client → server) +//! - `0xAF` KvPushAck (server → client) //! - `0xFF` Ping/Pong (bidirectional) pub mod ack_result; @@ -59,6 +63,7 @@ pub mod columnar; pub mod delta; pub mod frame; pub mod fts; +pub mod kv; pub mod presence; pub mod provenance; pub mod resync; @@ -82,6 +87,7 @@ pub use delta::{ }; pub use frame::{SyncFrame, SyncMessageType}; pub use fts::{FtsDeleteAckMsg, FtsDeleteMsg, FtsIndexAckMsg, FtsIndexMsg}; +pub use kv::{KvPushAckMsg, KvPushMsg, KvPushOp, RowPushRefusal, RowPushRejectMsg}; pub use presence::{PeerPresence, PresenceBroadcastMsg, PresenceLeaveMsg, PresenceUpdateMsg}; pub use provenance::SyncProvenance; pub use resync::{ResyncReason, ResyncRequestMsg, ThrottleMsg}; diff --git a/nodedb-types/src/sync/wire/stream_id.rs b/nodedb-types/src/sync/wire/stream_id.rs index 9214c200f..c6fa8cb90 100644 --- a/nodedb-types/src/sync/wire/stream_id.rs +++ b/nodedb-types/src/sync/wire/stream_id.rs @@ -22,6 +22,7 @@ pub enum EngineKind { Fts, Spatial, Array, + Kv, } /// Derive a stable, deterministic `stream_id` for a `(engine, collection)` pair. diff --git a/nodedb-types/src/timeseries/series.rs b/nodedb-types/src/timeseries/series.rs index 40937353c..6c7c21303 100644 --- a/nodedb-types/src/timeseries/series.rs +++ b/nodedb-types/src/timeseries/series.rs @@ -52,7 +52,7 @@ impl SeriesKey { /// On insert, if the SeriesId already maps to a *different* SeriesKey, the /// catalog rehashes with an incrementing attempt counter until it finds a free /// slot. This is one lookup per new series (not per row). -#[derive(Debug, Default, Serialize, Deserialize)] +#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] pub struct SeriesCatalog { /// SeriesId → (SeriesKey, rehash attempt that produced this ID). entries: HashMap, diff --git a/nodedb-types/src/vector_index_stats.rs b/nodedb-types/src/vector_index_stats.rs index c1f8221b9..0bbec7e01 100644 --- a/nodedb-types/src/vector_index_stats.rs +++ b/nodedb-types/src/vector_index_stats.rs @@ -126,6 +126,46 @@ pub struct VectorIndexStats { /// bytes. `None` when the collection has no dedicated arena (e.g., it is /// not vector-primary, or the runtime does not support per-arena stats). pub arena_bytes: Option, + /// IVF-PQ training state. `None` unless the index type is `ivf_pq`. + pub ivf: Option, + /// HNSW builds of this index waiting for or running on the builder. + pub builds_queued: usize, + /// HNSW builds of this index installed since the core opened it. + pub builds_completed: u64, + /// HNSW builds of this index that failed since the core opened it. + pub builds_failed: u64, +} + +/// Training state of an IVF-PQ vector index. +/// +/// An IVF-PQ index buffers vectors, searched exactly, until it holds +/// `training_threshold` live vectors. It then trains its cells and PQ +/// codebooks on them and searches through IVF-PQ from then on. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct VectorIvfStats { + /// Live vectors the index needs before it trains: `max(ivf_cells, pq_k)`. + pub training_threshold: usize, + /// Whether training has run. + pub trained: bool, + /// Vectors the training read. `0` before training. + pub trained_on: usize, + /// Unix milliseconds of the training. `0` before training. + pub trained_at_ms: u64, + /// Vectors held by the trained index, live or soft-deleted. + pub indexed_vectors: usize, + /// Voronoi cells of the trained index. `0` before training. + pub cells: usize, + /// Cells probed per query. + pub nprobe: usize, } #[cfg(test)] @@ -155,6 +195,18 @@ mod tests { seal_threshold: 65_536, mmap_segment_count: 1, arena_bytes: Some(4 * 1024 * 1024), + ivf: Some(VectorIvfStats { + training_threshold: 256, + trained: true, + trained_on: 300, + trained_at_ms: 1_700_000_000_000, + indexed_vectors: 310, + cells: 16, + nprobe: 4, + }), + builds_queued: 1, + builds_completed: 2, + builds_failed: 0, }; let bytes = zerompk::to_msgpack_vec(&stats).unwrap(); let restored: VectorIndexStats = zerompk::from_msgpack(&bytes).unwrap(); @@ -162,5 +214,7 @@ mod tests { assert_eq!(restored.live_count, 183_000); assert_eq!(restored.quantization, VectorIndexQuantization::Sq8); assert_eq!(restored.index_type, VectorIndexType::Hnsw); + assert_eq!(restored.ivf, stats.ivf); + assert_eq!(restored.builds_completed, 2); } } diff --git a/nodedb-vector/src/adaptive_filter.rs b/nodedb-vector/src/adaptive_filter.rs index 9b3c59430..767367ee8 100644 --- a/nodedb-vector/src/adaptive_filter.rs +++ b/nodedb-vector/src/adaptive_filter.rs @@ -66,6 +66,10 @@ pub fn select_strategy(selectivity: f64, thresholds: &FilterThresholds) -> Filte } /// Execute adaptive filtered search on an HNSW index. +/// +/// A query without the index dimension fails with +/// [`VectorError::DimensionMismatch`](crate::error::VectorError::DimensionMismatch) +/// under every strategy. pub fn adaptive_search( index: &HnswIndex, query: &[f32], @@ -73,7 +77,8 @@ pub fn adaptive_search( ef: usize, bitmap: &RoaringBitmap, thresholds: &FilterThresholds, -) -> Vec { +) -> Result, crate::error::VectorError> { + crate::error::check_dim(index.dim(), query.len())?; let total = index.len(); let selectivity = estimate_selectivity(bitmap, total); let strategy = select_strategy(selectivity, thresholds); @@ -82,13 +87,13 @@ pub fn adaptive_search( FilterStrategy::PreFilter => index.search_filtered(query, top_k, ef, bitmap), FilterStrategy::PostFilter { over_fetch_factor } => { let fetch_k = top_k * over_fetch_factor; - let results = index.search(query, fetch_k, ef.max(fetch_k)); + let results = index.search(query, fetch_k, ef.max(fetch_k))?; let mut filtered: Vec = results .into_iter() .filter(|r| bitmap.contains(r.id)) .collect(); filtered.truncate(top_k); - filtered + Ok(filtered) } FilterStrategy::BruteForceMatching => { let metric = index.params().metric; @@ -119,7 +124,7 @@ pub fn adaptive_search( .partial_cmp(&b.distance) .unwrap_or(std::cmp::Ordering::Equal) }); - results + Ok(results) } } } @@ -179,7 +184,8 @@ mod tests { bitmap.insert(i); } - let results = adaptive_search(&idx, &[505.0, 0.0, 0.0], 3, 64, &bitmap, &thresholds); + let results = + adaptive_search(&idx, &[505.0, 0.0, 0.0], 3, 64, &bitmap, &thresholds).unwrap(); assert_eq!(results.len(), 3); for r in &results { assert!(bitmap.contains(r.id), "got filtered-out id {}", r.id); @@ -197,7 +203,8 @@ mod tests { bitmap.insert(i); } - let results = adaptive_search(&idx, &[100.0, 0.0, 0.0], 5, 64, &bitmap, &thresholds); + let results = + adaptive_search(&idx, &[100.0, 0.0, 0.0], 5, 64, &bitmap, &thresholds).unwrap(); assert_eq!(results.len(), 5); for r in &results { assert!(bitmap.contains(r.id)); @@ -213,4 +220,26 @@ mod tests { let sel = estimate_selectivity(&bitmap, 1000); assert!((sel - 0.9).abs() < 0.01); } + + /// Every strategy refuses a query of the wrong dimension, including the + /// brute-force path that never touches the graph. + #[test] + fn wrong_dimension_query_is_a_typed_error() { + let idx = build_test_index(); + let thresholds = FilterThresholds::default(); + for n in [5u32, 500, 1000] { + let bitmap: RoaringBitmap = (0..n).collect(); + let result = adaptive_search(&idx, &[1.0, 0.0], 3, 64, &bitmap, &thresholds); + assert!( + matches!( + result, + Err(crate::error::VectorError::DimensionMismatch { + expected: 3, + got: 2 + }) + ), + "{result:?}" + ); + } + } } diff --git a/nodedb-vector/src/builder.rs b/nodedb-vector/src/builder.rs index 372119f9b..99b5ba889 100644 --- a/nodedb-vector/src/builder.rs +++ b/nodedb-vector/src/builder.rs @@ -1,9 +1,20 @@ // SPDX-License-Identifier: Apache-2.0 -//! Background HNSW index builder thread. +//! Background HNSW builder thread. //! -//! Each Data Plane core has one builder thread that processes HNSW -//! construction requests sequentially (FIFO). +//! Each Data Plane core owns one builder thread. The core sends build +//! requests with `try_send` and drains finished builds with `try_recv` once +//! per tick, so a build never blocks the core's reactor. Both channels are +//! bounded: +//! +//! - The request queue holds `capacity` requests. When it is full the core +//! keeps the job in its own backlog and sends it on a later tick. The +//! segment stays searchable by brute force meanwhile. +//! - The completion queue holds `capacity` results. When it is full the +//! builder thread waits for the core to drain it. +//! +//! The thread builds requests in FIFO order and stops when the core drops +//! its request sender. use std::sync::mpsc; use std::thread::JoinHandle; @@ -11,18 +22,28 @@ use std::thread::JoinHandle; use tracing::{debug, info, warn}; use crate::collection::{BuildComplete, BuildRequest}; +use crate::error::VectorError; use crate::hnsw::HnswIndex; /// Sender half: TPC core sends build requests to the builder thread. -pub type BuildSender = mpsc::Sender; +pub type BuildSender = mpsc::SyncSender; /// Receiver half: TPC core receives completed builds. pub type CompleteReceiver = mpsc::Receiver; -/// Spawn a background HNSW builder thread for a Data Plane core. -pub fn spawn_builder(core_id: usize) -> (BuildSender, CompleteReceiver, JoinHandle<()>) { - let (request_tx, request_rx) = mpsc::channel::(); - let (complete_tx, complete_rx) = mpsc::channel::(); +/// Build requests a core's builder queue holds. Each request carries a whole +/// segment's vectors, so the bound caps the memory in flight. +pub const BUILD_QUEUE_CAPACITY: usize = 4; + +/// Spawn the HNSW builder thread for Data Plane core `core_id`, with request +/// and completion queues of `capacity` entries each. Fails when the OS +/// refuses the thread. +pub fn spawn_builder( + core_id: usize, + capacity: usize, +) -> std::io::Result<(BuildSender, CompleteReceiver, JoinHandle<()>)> { + let (request_tx, request_rx) = mpsc::sync_channel::(capacity); + let (complete_tx, complete_rx) = mpsc::sync_channel::(capacity); let handle = std::thread::Builder::new() .name(format!("hnsw-builder-{core_id}")) @@ -30,13 +51,16 @@ pub fn spawn_builder(core_id: usize) -> (BuildSender, CompleteReceiver, JoinHand info!(core_id, "HNSW builder thread started"); builder_loop(core_id, request_rx, complete_tx); info!(core_id, "HNSW builder thread stopped"); - }) - .expect("failed to spawn HNSW builder thread"); + })?; - (request_tx, complete_rx, handle) + Ok((request_tx, complete_rx, handle)) } -fn builder_loop(core_id: usize, rx: mpsc::Receiver, tx: mpsc::Sender) { +fn builder_loop( + core_id: usize, + rx: mpsc::Receiver, + tx: mpsc::SyncSender, +) { while let Ok(req) = rx.recv() { debug!( core_id, @@ -46,40 +70,96 @@ fn builder_loop(core_id: usize, rx: mpsc::Receiver, tx: mpsc::Send dim = req.dim, "building HNSW index" ); - let start = std::time::Instant::now(); - let mut index = HnswIndex::with_seed( - req.dim, - req.params, - (core_id as u64 + 1) * 1000 + req.segment_id as u64, - ); + let (key, segment_id, kind) = (req.key.clone(), req.segment_id, req.kind); + let result = build(core_id, req); + match &result { + Ok(index) => info!( + core_id, + key = %key, + segment_id, + vectors = index.len(), + elapsed_ms = start.elapsed().as_millis() as u64, + "HNSW index built" + ), + Err(e) => warn!(core_id, key = %key, segment_id, error = %e, "HNSW build failed"), + } + let complete = BuildComplete { + key, + segment_id, + kind, + result, + }; + if tx.send(complete).is_err() { + warn!(core_id, "builder: core channel closed, stopping"); + break; + } + } +} - for vector in req.vectors { - index - .insert(vector) - .unwrap_or_else(|e| tracing::error!(error = %e, "HNSW insert failed")); +/// Build one graph, inserting every vector in local-id order so node `i` of +/// the graph is vector `i` of the request. The first insert error fails the +/// build: a skipped vector would shift every later id. +fn build(core_id: usize, req: BuildRequest) -> Result { + let seed = (core_id as u64 + 1) * 1000 + u64::from(req.segment_id); + let mut index = HnswIndex::with_seed(req.dim, req.params, seed); + for vector in req.vectors { + index.insert(vector)?; + } + Ok(index) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::collection::BuildKind; + use crate::hnsw::HnswParams; + + fn request(segment_id: u32, vectors: Vec>) -> BuildRequest { + BuildRequest { + key: "k".into(), + segment_id, + kind: BuildKind::Seal, + vectors, + dim: 2, + params: HnswParams::default(), } + } - let elapsed = start.elapsed(); - info!( - core_id, - key = %req.key, - segment_id = req.segment_id, - vectors = index.len(), - elapsed_ms = elapsed.as_millis() as u64, - "HNSW index built" - ); + #[test] + fn builds_keep_one_node_per_vector_and_report_errors() { + let (tx, rx, handle) = spawn_builder(0, 1).unwrap(); + tx.send(request(1, vec![vec![1.0, 0.0], vec![0.0, 1.0]])) + .unwrap(); + tx.send(request(2, vec![vec![1.0, 0.0], vec![1.0]])) + .unwrap(); - if tx - .send(BuildComplete { - key: req.key, - segment_id: req.segment_id, - index, - }) - .is_err() - { - warn!(core_id, "builder: core channel closed, stopping"); - break; + let first = rx.recv().unwrap(); + assert_eq!(first.segment_id, 1); + assert_eq!(first.result.unwrap().len(), 2); + let second = rx.recv().unwrap(); + assert!(matches!( + second.result, + Err(VectorError::DimensionMismatch { .. }) + )); + + drop(tx); + handle.join().unwrap(); + } + + #[test] + fn a_full_request_queue_refuses_without_blocking() { + let (tx, _rx, _handle) = spawn_builder(0, 1).unwrap(); + // The thread holds at most one request in hand and one queued; with + // the completion side undrained, later sends find the queue full. + let mut refused = false; + for id in 0..8 { + if let Err(mpsc::TrySendError::Full(_)) = tx.try_send(request(id, vec![vec![1.0, 0.0]])) + { + refused = true; + break; + } } + assert!(refused, "a bounded queue must refuse once full"); } } diff --git a/nodedb-vector/src/codec_index/build.rs b/nodedb-vector/src/codec_index/build.rs index 8234f90a4..7f5c63825 100644 --- a/nodedb-vector/src/codec_index/build.rs +++ b/nodedb-vector/src/codec_index/build.rs @@ -11,6 +11,7 @@ use std::collections::{BinaryHeap, HashSet}; use nodedb_codec::vector_quant::codec::VectorCodec; use super::graph::{HnswCodecIndex, NodeC}; +use crate::error::{VectorError, check_dim}; /// Ordered pair for priority queues (dist, node_idx in `nodes` vec). #[derive(Clone, Copy, PartialEq)] @@ -41,8 +42,10 @@ impl HnswCodecIndex { /// Insert a vector with the given caller-supplied `id`. /// /// Encodes `v` via `codec.encode`, assigns a random layer, and runs the - /// standard HNSW neighbour-selection algorithm. - pub fn insert(&mut self, id: u32, v: &[f32]) { + /// standard HNSW neighbour-selection algorithm. A vector without the + /// index dimension fails with [`VectorError::DimensionMismatch`]. + pub fn insert(&mut self, id: u32, v: &[f32]) -> Result<(), VectorError> { + check_dim(self.dim, v.len())?; let quantized = self.codec.encode(v); let node_layer = self.random_layer(); @@ -64,7 +67,7 @@ impl HnswCodecIndex { // First node: it becomes the entry point. self.entry_point = Some(new_idx); self.max_layer = node_layer; - return; + return Ok(()); }; // Phase 1: greedy descent from max_layer down to node_layer + 1. @@ -124,6 +127,7 @@ impl HnswCodecIndex { self.entry_point = Some(new_idx); self.max_layer = node_layer; } + Ok(()) } /// Greedy descent: starting at `ep_idx`, find the single nearest node to @@ -262,14 +266,14 @@ mod tests { .map(|i| (0..dim).map(|d| (i * dim + d) as f32 * 0.1).collect()) .collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - Sq8Codec::calibrate(&refs, dim) + Sq8Codec::calibrate(&refs, dim).unwrap() } #[test] fn insert_sets_entry_point() { let codec = make_sq8(4, 10); let mut idx: HnswCodecIndex = HnswCodecIndex::new(4, 8, 50, codec, 1); - idx.insert(0, &[0.1, 0.2, 0.3, 0.4]); + idx.insert(0, &[0.1, 0.2, 0.3, 0.4]).unwrap(); assert!(idx.entry_point.is_some()); assert_eq!(idx.len(), 1); } @@ -280,7 +284,7 @@ mod tests { let mut idx: HnswCodecIndex = HnswCodecIndex::new(4, 8, 50, codec, 42); for i in 0..20u32 { let v: Vec = (0..4).map(|d| (i as usize * 4 + d) as f32).collect(); - idx.insert(i, &v); + idx.insert(i, &v).unwrap(); } assert_eq!(idx.len(), 20); assert!(idx.entry_point.is_some()); diff --git a/nodedb-vector/src/codec_index/graph.rs b/nodedb-vector/src/codec_index/graph.rs index df4baf961..3cd0edc61 100644 --- a/nodedb-vector/src/codec_index/graph.rs +++ b/nodedb-vector/src/codec_index/graph.rs @@ -145,7 +145,7 @@ mod tests { .map(|i| (0..dim).map(|d| (i * dim + d) as f32 * 0.1).collect()) .collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - Sq8Codec::calibrate(&refs, dim) + Sq8Codec::calibrate(&refs, dim).unwrap() } #[test] diff --git a/nodedb-vector/src/codec_index/search.rs b/nodedb-vector/src/codec_index/search.rs index befa7d46b..51b3a04d8 100644 --- a/nodedb-vector/src/codec_index/search.rs +++ b/nodedb-vector/src/codec_index/search.rs @@ -13,6 +13,7 @@ use std::collections::{BinaryHeap, HashSet}; use nodedb_codec::vector_quant::codec::VectorCodec; use super::graph::HnswCodecIndex; +use crate::error::{VectorError, check_dim}; /// A single result from a codec-index search. #[derive(Debug, Clone)] @@ -54,13 +55,21 @@ impl HnswCodecIndex { /// `ef_search` controls the beam width at layer 0 (must be >= k). /// /// The returned results are sorted ascending by `exact_asymmetric_distance`. - pub fn search(&self, query: &[f32], k: usize, ef_search: usize) -> Vec { + /// A query without the index dimension fails with + /// [`VectorError::DimensionMismatch`]. + pub fn search( + &self, + query: &[f32], + k: usize, + ef_search: usize, + ) -> Result, VectorError> { + check_dim(self.dim, query.len())?; if self.is_empty() { - return Vec::new(); + return Ok(Vec::new()); } let Some(ep) = self.entry_point else { - return Vec::new(); + return Ok(Vec::new()); }; let ef = ef_search.max(k); @@ -93,10 +102,10 @@ impl HnswCodecIndex { reranked.sort_unstable_by(|a, b| a.0.total_cmp(&b.0)); reranked.truncate(k); - reranked + Ok(reranked .into_iter() .map(|(distance, id)| CodecSearchResult { id, distance }) - .collect() + .collect()) } /// Greedy single-nearest descent at `layer` using the pre-encoded query. @@ -229,14 +238,14 @@ mod tests { let mut state = 0xDEAD_BEEF_u64; let vecs: Vec> = (0..n).map(|_| rand_vec(&mut state, dim)).collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = Sq8Codec::calibrate(&refs, dim); + let codec = Sq8Codec::calibrate(&refs, dim).unwrap(); let mut idx: HnswCodecIndex = HnswCodecIndex::new(dim, 8, 100, codec, 7); for (i, v) in vecs.iter().enumerate() { - idx.insert(i as u32, v); + idx.insert(i as u32, v).unwrap(); } // Query with vector 17 — top-1 should return id 17. let query = vecs[17].clone(); - let results = idx.search(&query, 1, 50); + let results = idx.search(&query, 1, 50).unwrap(); assert_eq!(results.len(), 1, "expected 1 result"); assert_eq!( results[0].id, 17, @@ -262,7 +271,7 @@ mod tests { let codec = RaBitQCodec::calibrate(&refs, dim, 0xABCD_1234); let mut idx: HnswCodecIndex = HnswCodecIndex::new(dim, 8, 150, codec, 99); for (i, v) in vecs.iter().enumerate() { - idx.insert(i as u32, v); + idx.insert(i as u32, v).unwrap(); } let n_queries = 10usize; @@ -272,7 +281,7 @@ mod tests { let query = rand_vec(&mut state, dim); let truth: std::collections::HashSet = ground_truth(&vecs, &query, k).into_iter().collect(); - let results = idx.search(&query, k, k * 4); + let results = idx.search(&query, k, k * 4).unwrap(); let found: std::collections::HashSet = results.iter().map(|r| r.id).collect(); total_hits += found.intersection(&truth).count(); total += k; @@ -300,7 +309,7 @@ mod tests { let codec = BbqCodec::calibrate(&refs, dim, 3); let mut idx: HnswCodecIndex = HnswCodecIndex::new(dim, 8, 150, codec, 42); for (i, v) in vecs.iter().enumerate() { - idx.insert(i as u32, v); + idx.insert(i as u32, v).unwrap(); } let n_queries = 10usize; @@ -310,7 +319,7 @@ mod tests { let query = rand_vec(&mut state, dim); let truth: std::collections::HashSet = ground_truth(&vecs, &query, k).into_iter().collect(); - let results = idx.search(&query, k, k * 4); + let results = idx.search(&query, k, k * 4).unwrap(); let found: std::collections::HashSet = results.iter().map(|r| r.id).collect(); total_hits += found.intersection(&truth).count(); total += k; @@ -332,10 +341,10 @@ mod tests { let codec = { let vecs: Vec> = (0..5).map(|i| vec![i as f32; 4]).collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - Sq8Codec::calibrate(&refs, 4) + Sq8Codec::calibrate(&refs, 4).unwrap() }; let idx: HnswCodecIndex = HnswCodecIndex::new(4, 8, 50, codec, 1); - let results = idx.search(&[0.0, 0.0, 0.0, 0.0], 5, 20); + let results = idx.search(&[0.0, 0.0, 0.0, 0.0], 5, 20).unwrap(); assert!(results.is_empty(), "empty index must return no results"); } @@ -344,12 +353,43 @@ mod tests { let dim = 4; let vecs = [vec![1.0f32, 2.0, 3.0, 4.0]]; let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = Sq8Codec::calibrate(&refs, dim); + let codec = Sq8Codec::calibrate(&refs, dim).unwrap(); let mut idx: HnswCodecIndex = HnswCodecIndex::new(dim, 8, 50, codec, 5); - idx.insert(0, &vecs[0]); + idx.insert(0, &vecs[0]).unwrap(); // Query with a completely different vector. - let results = idx.search(&[10.0, 20.0, 30.0, 40.0], 1, 10); + let results = idx.search(&[10.0, 20.0, 30.0, 40.0], 1, 10).unwrap(); assert_eq!(results.len(), 1, "single-node index must return 1 result"); assert_eq!(results[0].id, 0, "the only node must be returned"); } + + #[test] + fn wrong_dimension_is_a_typed_error() { + use crate::error::VectorError; + let vecs: Vec> = (0..5).map(|i| vec![i as f32; 4]).collect(); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let codec = Sq8Codec::calibrate(&refs, 4).unwrap(); + let mut idx: HnswCodecIndex = HnswCodecIndex::new(4, 8, 50, codec, 1); + assert!(matches!( + idx.search(&[0.0; 3], 5, 20), + Err(VectorError::DimensionMismatch { + expected: 4, + got: 3 + }) + )); + idx.insert(0, &vecs[0]).unwrap(); + assert!(matches!( + idx.search(&[0.0; 5], 5, 20), + Err(VectorError::DimensionMismatch { + expected: 4, + got: 5 + }) + )); + assert!(matches!( + idx.insert(1, &[0.0; 2]), + Err(VectorError::DimensionMismatch { + expected: 4, + got: 2 + }) + )); + } } diff --git a/nodedb-vector/src/collection/budget.rs b/nodedb-vector/src/collection/budget.rs index de1850c85..2dfcfa752 100644 --- a/nodedb-vector/src/collection/budget.rs +++ b/nodedb-vector/src/collection/budget.rs @@ -36,7 +36,8 @@ impl VectorCollection { .filter(|s| s.tier == StorageTier::L0Ram) .map(|s| s.index.len() * bytes_per_vector) .sum(); - growing + building + sealed_ram + let ivf = self.ivf.as_ref().map_or(0, |ivf| ivf.memory_bytes()); + growing + building + sealed_ram + ivf } /// Whether the RAM budget is exceeded. diff --git a/nodedb-vector/src/collection/build.rs b/nodedb-vector/src/collection/build.rs new file mode 100644 index 000000000..2fb5209b7 --- /dev/null +++ b/nodedb-vector/src/collection/build.rs @@ -0,0 +1,389 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! HNSW builds for a `VectorCollection`: the requests the owning core sends +//! to its builder thread, and installing the finished graphs. +//! +//! Every request carries one vector per local node id, soft-deleted nodes +//! included, so a built graph keeps the ids its segment had. Installing a +//! build applies the tombstones the segment holds at that moment, so a +//! delete that landed while the build ran is kept. +//! +//! - A seal build promotes a building segment to sealed. +//! - A rebuild replaces a sealed segment in place, from that segment's own +//! vectors, and quantizes it again under the collection's config. + +use nodedb_mem::ScopedMemory; +use nodedb_types::VectorQuantization; + +use crate::error::VectorError; +use crate::hnsw::HnswIndex; +use crate::index_config::IndexType; + +use super::lifecycle::VectorCollection; +use super::lifecycle_insert_ops::sealed_vector; +use super::segment::{BuildKind, BuildRequest, SealedSegment}; + +impl VectorCollection { + /// Segment ids of the building segments, oldest first. + pub fn building_segment_ids(&self) -> Vec { + self.building.iter().map(|b| b.segment_id).collect() + } + + /// A seal build request for building segment `segment_id`, read from its + /// flat vectors. `None` when no building segment has that id. + pub fn build_request_for(&self, key: &str, segment_id: u32) -> Option { + let seg = self.building.iter().find(|b| b.segment_id == segment_id)?; + let vectors = (0..seg.flat.len() as u32) + .filter_map(|i| seg.flat.get_vector_raw(i).map(<[f32]>::to_vec)) + .collect(); + Some(BuildRequest { + key: key.to_string(), + segment_id, + kind: BuildKind::Seal, + vectors, + dim: self.dim, + params: self.params.clone(), + }) + } + + /// Base ids of the non-empty sealed segments, in segment order. + pub fn sealed_base_ids(&self) -> Vec { + self.sealed + .iter() + .filter(|s| !s.index.is_empty()) + .map(|s| s.base_id) + .collect() + } + + /// A rebuild request for the sealed segment at `base_id`, under the + /// current HNSW params, read from that segment only: the growing and + /// building segments keep their own vectors. `None` when no sealed + /// segment starts at `base_id`. + /// + /// Fails with [`VectorError::VectorUnavailable`] when a node's vector + /// cannot be read. + pub fn rebuild_request_for( + &mut self, + key: &str, + base_id: u32, + ) -> Result, VectorError> { + let Some(seg) = self.sealed.iter().find(|s| s.base_id == base_id) else { + return Ok(None); + }; + let len = seg.index.len(); + let mut vectors = Vec::with_capacity(len); + for local in 0..len as u32 { + let v = sealed_vector(seg, local).ok_or(VectorError::VectorUnavailable { + id: base_id + local, + })?; + vectors.push(v); + } + let segment_id = self.next_segment_id; + self.next_segment_id += 1; + Ok(Some(BuildRequest { + key: key.to_string(), + segment_id, + kind: BuildKind::Rebuild { base_id, len }, + vectors, + dim: self.dim, + params: self.params.clone(), + })) + } + + /// Install a seal build: promote building segment `segment_id` to sealed + /// with its current tombstones. Returns `false`, changing nothing, when + /// no building segment has that id (a truncate dropped it) or the graph + /// does not hold one node per vector of the segment. + pub fn complete_build( + &mut self, + segment_id: u32, + index: HnswIndex, + memory: ScopedMemory, + ) -> bool { + let Some(pos) = self + .building + .iter() + .position(|b| b.segment_id == segment_id) + else { + return false; + }; + let mut index = index; + if index.len() != self.building[pos].flat.len() { + tracing::error!( + segment_id, + built = index.len(), + expected = self.building[pos].flat.len(), + "HNSW build does not match its segment; segment stays on brute force" + ); + return false; + } + let building = self.building.remove(pos); + for local in 0..building.flat.len() as u32 { + if building.flat.is_deleted(local) { + index.delete(local); + } + } + let seg = self.sealed_segment_from(segment_id, building.base_id, index, &memory); + self.sealed.push(seg); + self.builds_completed += 1; + self.refresh_codec_dispatch(); + true + } + + /// Install a rebuild of the sealed segment at `base_id`, read when it held + /// `len` nodes. The old segment's tombstones carry over and the segment is + /// quantized again under the collection's config. Returns `false`, + /// changing nothing, when no sealed segment at `base_id` still holds + /// `len` nodes, or the graph does not hold `len` nodes. + pub fn complete_rebuild( + &mut self, + segment_id: u32, + base_id: u32, + len: usize, + index: HnswIndex, + memory: ScopedMemory, + ) -> bool { + let Some(pos) = self + .sealed + .iter() + .position(|s| s.base_id == base_id && s.index.len() == len) + else { + return false; + }; + if index.len() != len { + tracing::error!( + base_id, + built = index.len(), + expected = len, + "HNSW rebuild does not match its segment; the old segment stays" + ); + return false; + } + let mut index = index; + for local in 0..len as u32 { + if self.sealed[pos].index.is_deleted(local) { + index.delete(local); + } + } + let seg = self.sealed_segment_from(segment_id, base_id, index, &memory); + let old = std::mem::replace(&mut self.sealed[pos], seg); + self.drop_sealed(old); + self.builds_completed += 1; + self.refresh_codec_dispatch(); + true + } + + /// Record a build that failed. The segment stays as it was. + pub fn note_build_failed(&mut self) { + self.builds_failed += 1; + } + + /// A sealed segment for `index`: quantized under the collection's config + /// and placed on the storage tier the memory budget allows. + fn sealed_segment_from( + &mut self, + segment_id: u32, + base_id: u32, + index: HnswIndex, + memory: &ScopedMemory, + ) -> SealedSegment { + let use_codec_dispatch = self.codec_dispatch_tag().is_some(); + let use_pq = !use_codec_dispatch && self.index_config.index_type == IndexType::HnswPq; + let (sq8, pq) = if use_codec_dispatch { + (None, None) + } else if use_pq { + ( + None, + Self::build_pq_for_index(&index, self.index_config.pq_m, memory.clone()), + ) + } else { + (Self::build_sq8_for_index(&index), None) + }; + let (tier, mmap_vectors) = self.resolve_tier_for_build(segment_id, base_id, &index, memory); + SealedSegment { + index, + base_id, + sq8, + pq, + tier, + mmap_vectors, + } + } + + /// Drop a replaced sealed segment, removing its mmap file. + fn drop_sealed(&mut self, seg: SealedSegment) { + let mmap_path = seg.mmap_vectors.as_ref().map(|m| m.path().to_path_buf()); + // Unmap before the file goes. + drop(seg); + if let Some(path) = mmap_path { + self.mmap_segment_count = self.mmap_segment_count.saturating_sub(1); + if let Err(e) = std::fs::remove_file(&path) { + tracing::warn!( + path = %path.display(), + error = %e, + "vector rebuild: replaced mmap segment file not removed" + ); + } + } + } + + /// The codec-dispatch tag the collection's quantization selects. + fn codec_dispatch_tag(&self) -> Option<&'static str> { + match self.quantization { + VectorQuantization::RaBitQ => Some("rabitq"), + VectorQuantization::Bbq => Some("bbq"), + _ => None, + } + } + + /// Rebuild the collection-level codec-dispatch index over the sealed + /// segments when the quantization selects one. + fn refresh_codec_dispatch(&mut self) { + let Some(tag) = self.codec_dispatch_tag() else { + return; + }; + if let Err(e) = self.build_codec_dispatch(tag).map(|_| ()) { + // Without the codec index the sealed segments are searched by + // their own HNSW graphs, which answer the same queries. + tracing::error!(error = %e, tag, "codec-dispatch build failed; searching sealed segments directly"); + self.codec_dispatch = None; + } + } +} + +#[cfg(test)] +mod tests { + use nodedb_types::Surrogate; + + use super::*; + use crate::hnsw::HnswParams; + use crate::test_support::test_memory; + + fn vector(i: usize) -> Vec { + vec![(i + 1) as f32, (i % 7 + 1) as f32, (i % 11 + 1) as f32] + } + + fn l2() -> HnswParams { + HnswParams { + metric: crate::distance::DistanceMetric::L2, + ..HnswParams::default() + } + } + + fn build(req: &BuildRequest) -> HnswIndex { + let mut index = HnswIndex::with_seed(req.dim, req.params.clone(), 7); + for v in &req.vectors { + index.insert(v.clone()).unwrap(); + } + index + } + + /// Ten L2 vectors bound to surrogates 1..=10, with a seal threshold of 10. + fn collection() -> VectorCollection { + let mut coll = VectorCollection::with_seal_threshold(3, l2(), 10); + for i in 0..10 { + coll.insert_with_surrogate(vector(i), Surrogate::new(i as u32 + 1)) + .unwrap(); + } + coll + } + + #[test] + fn a_seal_build_keeps_ids_and_the_deletes_made_meanwhile() { + let mut coll = collection(); + coll.delete(2); + let req = coll.seal("k").unwrap(); + assert_eq!(req.vectors.len(), 10, "deleted vectors keep their slot"); + coll.delete(4); + + assert!(coll.complete_build(req.segment_id, build(&req), test_memory())); + + assert!(coll.building.is_empty()); + assert!(!coll.is_live(2) && !coll.is_live(4) && coll.is_live(9)); + assert_eq!(coll.vector_for_id(7), Some(vector(7))); + assert_eq!(coll.search(&vector(7), 1, 64).unwrap()[0].id, 7); + assert_eq!(coll.get_surrogate(7), Some(Surrogate::new(8))); + } + + #[test] + fn a_build_of_the_wrong_size_is_refused() { + let mut coll = collection(); + let req = coll.seal("k").unwrap(); + let mut short = HnswIndex::new(3, l2()); + short.insert(vector(0)).unwrap(); + assert!(!coll.complete_build(req.segment_id, short, test_memory())); + assert_eq!(coll.building.len(), 1, "the segment stays on brute force"); + } + + #[test] + fn a_rebuild_keeps_ids_codes_and_tombstones() { + let mut coll = VectorCollection::with_pq_config(3, l2(), 3); + coll.set_seal_threshold(10); + for i in 0..10 { + coll.insert_with_surrogate(vector(i), Surrogate::new(i as u32 + 1)) + .unwrap(); + } + let req = coll.seal("k").unwrap(); + coll.complete_build(req.segment_id, build(&req), test_memory()); + coll.delete(1); + + let req = coll.rebuild_request_for("k", 0).unwrap().unwrap(); + assert_eq!( + req.kind, + BuildKind::Rebuild { + base_id: 0, + len: 10 + } + ); + coll.delete(6); + let BuildKind::Rebuild { base_id, len } = req.kind else { + panic!("a rebuild request"); + }; + assert!(coll.complete_rebuild(req.segment_id, base_id, len, build(&req), test_memory())); + + assert_eq!(coll.sealed.len(), 1); + assert!( + coll.sealed[0].pq.is_some(), + "the segment is quantized again" + ); + assert!(!coll.is_live(1) && !coll.is_live(6)); + for i in [0usize, 3, 9] { + assert_eq!(coll.search(&vector(i), 1, 64).unwrap()[0].id, i as u32); + assert_eq!( + coll.get_surrogate(i as u32), + Some(Surrogate::new(i as u32 + 1)) + ); + } + } + + #[test] + fn a_rebuild_of_a_compacted_segment_is_refused() { + let mut coll = collection(); + let req = coll.seal("k").unwrap(); + coll.complete_build(req.segment_id, build(&req), test_memory()); + coll.delete(3); + let req = coll.rebuild_request_for("k", 0).unwrap().unwrap(); + assert_eq!(coll.compact(), 1); + let BuildKind::Rebuild { base_id, len } = req.kind else { + panic!("a rebuild request"); + }; + assert!(!coll.complete_rebuild(req.segment_id, base_id, len, build(&req), test_memory())); + assert_eq!(coll.sealed[0].index.len(), 9, "the compacted segment stays"); + } + + #[test] + fn an_unbuilt_segment_survives_a_checkpoint_as_building() { + let mut coll = collection(); + coll.delete(5); + coll.seal("k").unwrap(); + let bytes = coll.checkpoint_to_bytes(None).unwrap(); + let restored = VectorCollection::from_checkpoint(&bytes, None, test_memory()).unwrap(); + + let ids = restored.building_segment_ids(); + assert_eq!(ids.len(), 1); + let req = restored.build_request_for("k", ids[0]).unwrap(); + assert_eq!(req.vectors.len(), 10); + assert!(!restored.is_live(5)); + assert_eq!(restored.search(&vector(8), 1, 64).unwrap()[0].id, 8); + } +} diff --git a/nodedb-vector/src/collection/checkpoint.rs b/nodedb-vector/src/collection/checkpoint.rs index 3ff84e40d..e39985717 100644 --- a/nodedb-vector/src/collection/checkpoint.rs +++ b/nodedb-vector/src/collection/checkpoint.rs @@ -24,12 +24,13 @@ use nodedb_types::{Surrogate, VectorQuantization}; use serde::{Deserialize, Serialize}; use crate::collection::payload_index::PayloadIndexSetSnapshot; -use crate::collection::segment::{DEFAULT_SEAL_THRESHOLD, SealedSegment}; +use crate::collection::segment::{BuildingSegment, DEFAULT_SEAL_THRESHOLD, SealedSegment}; use crate::collection::tier::StorageTier; use crate::distance::DistanceMetric; use crate::error::VectorError; use crate::flat::FlatIndex; use crate::hnsw::{HnswIndex, HnswParams}; +use crate::ivf::IvfPqIndex; use crate::quantize::pq::PqCodec; use crate::quantize::sq8::Sq8Codec; @@ -98,6 +99,12 @@ pub(crate) struct CollectionSnapshot { /// nothing and replay everything, exactly as before. #[serde(default)] pub checkpoint_wal_lsn: u64, + /// Index type and PQ/IVF parameters. Its HNSW parameters are the + /// `params_*` fields above. + pub index_config: crate::index_config::IndexConfig, + /// Encoded trained IVF-PQ index. `None` for a non-IVF collection or one + /// still buffering toward its training threshold. + pub ivf_bytes: Option>, } #[derive(Serialize, Deserialize, zerompk::ToMessagePack, zerompk::FromMessagePack)] @@ -225,6 +232,8 @@ impl VectorCollection { } }, checkpoint_wal_lsn: self.checkpoint_wal_lsn.max(self.applied_wal_lsn), + index_config: self.index_config.clone(), + ivf_bytes: self.ivf.as_ref().map(IvfPqIndex::to_bytes).transpose()?, }; let msgpack = match zerompk::to_msgpack_vec(&snapshot) { Ok(bytes) => bytes, @@ -309,11 +318,14 @@ impl VectorCollection { let mut growing = FlatIndex::new(snap.dim, metric); for (i, v) in snap.growing_vectors.iter().enumerate() { let deleted = snap.growing_deleted.get(i).copied().unwrap_or(false); - if deleted { - growing.insert_tombstoned(v.clone()); + let inserted = if deleted { + growing.insert_tombstoned(v.clone()) } else { - growing.insert(v.clone()); - } + growing.insert(v.clone()) + }; + inserted.map_err(|e| VectorError::CheckpointDeserializationError { + detail: format!("growing-segment replay insert: {e}"), + })?; } let mut sealed = Vec::with_capacity(snap.sealed_segments.len()); @@ -366,43 +378,44 @@ impl VectorCollection { }); } + // A segment sealed but not yet built comes back as a building + // segment, searched by brute force; the owning core queues its build. + let mut next_segment_id = (sealed.len() + 1) as u32; + let mut building = Vec::with_capacity(snap.building_segments.len()); for bs in &snap.building_segments { - let mut index = HnswIndex::new(snap.dim, params.clone()); - for v in &bs.vectors { - index.insert(v.clone()).map_err(|e| { - VectorError::CheckpointDeserializationError { - detail: format!("building-segment replay insert: {e}"), - } + let mut flat = FlatIndex::new(snap.dim, metric); + for (i, v) in bs.vectors.iter().enumerate() { + let inserted = if bs.deleted.get(i).copied().unwrap_or(false) { + flat.insert_tombstoned(v.clone()) + } else { + flat.insert(v.clone()) + }; + inserted.map_err(|e| VectorError::CheckpointDeserializationError { + detail: format!("building-segment replay insert: {e}"), })?; } - // Replay building-segment tombstones onto the HNSW index. - for (i, &dead) in bs.deleted.iter().enumerate() { - if dead { - index.delete(i as u32); - } - } - let sq8 = VectorCollection::build_sq8_for_index(&index); - sealed.push(SealedSegment { - index, + building.push(BuildingSegment { + flat, base_id: bs.base_id, - sq8, - pq: None, - tier: StorageTier::L0Ram, - mmap_vectors: None, + segment_id: next_segment_id, }); + next_segment_id += 1; } - let next_segment_id = (sealed.len() + 1) as u32; - let index_config = crate::index_config::IndexConfig { hnsw: params.clone(), - ..crate::index_config::IndexConfig::default() + ..snap.index_config }; + let ivf = snap + .ivf_bytes + .as_deref() + .map(|bytes| IvfPqIndex::from_bytes(bytes, memory.clone())) + .transpose()?; Ok(Self { growing, growing_base_id: snap.growing_base_id, sealed, - building: Vec::new(), + building, params, next_id: snap.next_id, next_segment_id, @@ -429,6 +442,7 @@ impl VectorCollection { seal_threshold: DEFAULT_SEAL_THRESHOLD, index_config, codec_dispatch: None, + ivf, quantization: quantization_from_tag(snap.quantization_tag), payload: if snap.payload_index_bytes.is_empty() { super::payload_index::PayloadIndexSet::default() @@ -440,6 +454,8 @@ impl VectorCollection { arena_index: None, checkpoint_wal_lsn: snap.checkpoint_wal_lsn, applied_wal_lsn: snap.checkpoint_wal_lsn, + builds_completed: 0, + builds_failed: 0, }) } } @@ -503,7 +519,7 @@ mod tests { for (d, slot) in v.iter_mut().enumerate() { *slot = ((i as f32) * 0.01 + (d as f32) * 0.1).sin(); } - coll.insert(v); + coll.insert(v).unwrap(); } let req = coll.seal("sq8_test").expect("seal produced request"); let mut idx = HnswIndex::new(req.dim, req.params.clone()); @@ -551,14 +567,14 @@ mod tests { }, ); for i in 0..50u32 { - coll.insert(vec![i as f32, 0.0, 0.0]); + coll.insert(vec![i as f32, 0.0, 0.0]).unwrap(); } let bytes = coll.checkpoint_to_bytes(None).unwrap(); let restored = VectorCollection::from_checkpoint(&bytes, None, test_memory()).unwrap(); assert_eq!(restored.len(), 50); assert_eq!(restored.dim(), 3); - let results = restored.search(&[25.0, 0.0, 0.0], 1, 64); + let results = restored.search(&[25.0, 0.0, 0.0], 1, 64).unwrap(); assert_eq!(results[0].id, 25); } @@ -579,7 +595,7 @@ mod tests { ..HnswParams::default() }, ); - coll.insert(vec![1.0, 0.0, 0.0]); + coll.insert(vec![1.0, 0.0, 0.0]).unwrap(); coll.note_checkpoint_lsn(42); assert_eq!(coll.applied_wal_lsn(), 42); assert_eq!( @@ -634,7 +650,7 @@ mod tests { coll.payload .add_index("category".to_string(), PayloadIndexKind::Equality); for i in 0u32..10 { - let node_id = coll.insert(vec![i as f32, 0.0, 0.0]); + let node_id = coll.insert(vec![i as f32, 0.0, 0.0]).unwrap(); let mut fields = HashMap::new(); let cat = if i % 2 == 0 { "A" } else { "B" }; fields.insert("category".to_string(), Value::String(cat.to_string())); diff --git a/nodedb-vector/src/collection/codec_build.rs b/nodedb-vector/src/collection/codec_build.rs index dde9c6841..e683c1c71 100644 --- a/nodedb-vector/src/collection/codec_build.rs +++ b/nodedb-vector/src/collection/codec_build.rs @@ -8,57 +8,46 @@ use super::codec_dispatch::{CollectionCodec, build_collection_codec}; use super::lifecycle::VectorCollection; -impl VectorCollection { - /// Collect all live FP32 vectors from every segment (growing, building, - /// and sealed) in insertion order. Used to train the collection-level - /// codec-dispatch index. - pub(crate) fn gather_all_vectors_fp32(&self) -> Vec> { - let total = self.len(); - let mut out = Vec::with_capacity(total); - - for i in 0..self.growing.len() as u32 { - if let Some(v) = self.growing.get_vector(i) { - out.push(v.to_vec()); - } - } - - for seg in &self.building { - for i in 0..seg.flat.len() as u32 { - if let Some(v) = seg.flat.get_vector(i) { - out.push(v.to_vec()); - } - } - } +use crate::error::VectorError; +impl VectorCollection { + /// Every live FP32 vector of the sealed segments, keyed by its global + /// vector id. The collection-level codec index covers the sealed + /// segments only: search reads the growing and building segments by + /// brute force beside it. + pub(crate) fn gather_sealed_vectors_fp32(&self) -> Vec<(u32, Vec)> { + let mut out = Vec::new(); for seg in &self.sealed { let n = seg.index.len(); for i in 0..n as u32 { if !seg.index.is_deleted(i) && let Some(v) = seg.index.get_vector(i) { - out.push(v.to_vec()); + out.push((seg.base_id + i, v.to_vec())); } } } - out } - /// Build a codec-dispatched index over all current vectors using the + /// Build a codec-dispatched index over the sealed vectors using the /// requested quantization. Replaces any existing dispatch index for /// this collection. Idempotent. /// - /// Returns a reference to the new index, or `None` if the quantization - /// tag is not supported (falls back to per-segment Sq8/PQ paths) or there - /// are no vectors to train on. - pub fn build_codec_dispatch(&mut self, quantization: &str) -> Option<&CollectionCodec> { - let vectors = self.gather_all_vectors_fp32(); + /// Returns a reference to the new index, or `Ok(None)` if the + /// quantization tag is not supported (falls back to per-segment Sq8/PQ + /// paths) or there are no vectors to train on. + pub fn build_codec_dispatch( + &mut self, + quantization: &str, + ) -> Result, VectorError> { + let vectors = self.gather_sealed_vectors_fp32(); let dim = self.dim; let m = self.params.m; let ef_construction = self.params.ef_construction; let seed = 42_u64; self.codec_dispatch = - build_collection_codec(quantization, &vectors, dim, m, ef_construction, seed); - self.codec_dispatch.as_ref() + build_collection_codec(quantization, &vectors, dim, m, ef_construction, seed)?; + Ok(self.codec_dispatch.as_ref()) } } diff --git a/nodedb-vector/src/collection/codec_dispatch.rs b/nodedb-vector/src/collection/codec_dispatch.rs index 6c574c2cc..0da1379ef 100644 --- a/nodedb-vector/src/collection/codec_dispatch.rs +++ b/nodedb-vector/src/collection/codec_dispatch.rs @@ -8,6 +8,7 @@ use nodedb_codec::vector_quant::bbq::BbqCodec; use nodedb_codec::vector_quant::rabitq::RaBitQCodec; use crate::codec_index::HnswCodecIndex; +use crate::error::VectorError; /// One built codec-index per collection (other than Sq8). Variants match /// the publicly-selectable quantization choices that route through @@ -26,7 +27,7 @@ impl CollectionCodec { query: &[f32], k: usize, ef_search: usize, - ) -> Vec { + ) -> Result, VectorError> { match self { Self::RaBitQ(idx) => idx.search(query, k, ef_search), Self::Bbq(idx) => idx.search(query, k, ef_search), @@ -34,7 +35,7 @@ impl CollectionCodec { } /// Forwarding `insert`. - pub fn insert(&mut self, id: u32, v: &[f32]) { + pub fn insert(&mut self, id: u32, v: &[f32]) -> Result<(), VectorError> { match self { Self::RaBitQ(idx) => idx.insert(id, v), Self::Bbq(idx) => idx.insert(id, v), @@ -64,38 +65,47 @@ impl CollectionCodec { /// Build a `CollectionCodec` from a quantization tag and training vectors. /// -/// Returns `None` for unsupported or unrecognised tags (e.g. "sq8", "pq", -/// "none") — those variants use separate per-segment code paths. +/// Returns `Ok(None)` for unsupported or unrecognised tags (e.g. "sq8", +/// "pq", "none") — those variants use separate per-segment code paths — and +/// for an empty vector set. A vector without `dim` components fails with +/// [`VectorError::DimensionMismatch`]. +/// +/// Each entry is `(id, vector)`; the index reports `id` in its results, so +/// it must be the collection's global vector id. pub fn build_collection_codec( quantization: &str, - vectors: &[Vec], + vectors: &[(u32, Vec)], dim: usize, m: usize, ef_construction: usize, seed: u64, -) -> Option { +) -> Result, VectorError> { if vectors.is_empty() { - return None; + return Ok(None); + } + // Calibration reads every vector as `dim` components; check before it. + for (_, v) in vectors { + crate::error::check_dim(dim, v.len())?; } - let refs: Vec<&[f32]> = vectors.iter().map(|v| v.as_slice()).collect(); + let refs: Vec<&[f32]> = vectors.iter().map(|(_, v)| v.as_slice()).collect(); match quantization { "rabitq" => { let codec = RaBitQCodec::calibrate(&refs, dim, seed); let mut idx = HnswCodecIndex::new(dim, m, ef_construction, codec, seed); - for (i, v) in vectors.iter().enumerate() { - idx.insert(i as u32, v); + for (id, v) in vectors { + idx.insert(*id, v)?; } - Some(CollectionCodec::RaBitQ(idx)) + Ok(Some(CollectionCodec::RaBitQ(idx))) } "bbq" => { let codec = BbqCodec::calibrate(&refs, dim, 3); let mut idx = HnswCodecIndex::new(dim, m, ef_construction, codec, seed); - for (i, v) in vectors.iter().enumerate() { - idx.insert(i as u32, v); + for (id, v) in vectors { + idx.insert(*id, v)?; } - Some(CollectionCodec::Bbq(idx)) + Ok(Some(CollectionCodec::Bbq(idx))) } - _ => None, + _ => Ok(None), } } @@ -103,16 +113,19 @@ pub fn build_collection_codec( mod tests { use super::*; - fn make_vectors(n: usize, dim: usize) -> Vec> { + fn make_vectors(n: usize, dim: usize) -> Vec<(u32, Vec)> { (0..n) - .map(|i| (0..dim).map(|d| (i * dim + d) as f32 * 0.01).collect()) + .map(|i| { + let v = (0..dim).map(|d| (i * dim + d) as f32 * 0.01).collect(); + (i as u32, v) + }) .collect() } #[test] fn build_rabitq_returns_some() { let vecs = make_vectors(50, 8); - let result = build_collection_codec("rabitq", &vecs, 8, 16, 100, 42); + let result = build_collection_codec("rabitq", &vecs, 8, 16, 100, 42).unwrap(); assert!( matches!(result, Some(CollectionCodec::RaBitQ(_))), "expected RaBitQ variant" @@ -122,7 +135,7 @@ mod tests { #[test] fn build_bbq_returns_some() { let vecs = make_vectors(50, 8); - let result = build_collection_codec("bbq", &vecs, 8, 16, 100, 42); + let result = build_collection_codec("bbq", &vecs, 8, 16, 100, 42).unwrap(); assert!( matches!(result, Some(CollectionCodec::Bbq(_))), "expected Bbq variant" @@ -132,14 +145,14 @@ mod tests { #[test] fn unknown_codec_returns_none() { let vecs = make_vectors(50, 8); - let result = build_collection_codec("unknown_codec", &vecs, 8, 16, 100, 42); + let result = build_collection_codec("unknown_codec", &vecs, 8, 16, 100, 42).unwrap(); assert!(result.is_none(), "unknown codec should return None"); } #[test] fn sq8_tag_returns_none() { let vecs = make_vectors(50, 8); - let result = build_collection_codec("sq8", &vecs, 8, 16, 100, 42); + let result = build_collection_codec("sq8", &vecs, 8, 16, 100, 42).unwrap(); assert!( result.is_none(), "sq8 tag should fall through to per-segment path" @@ -148,14 +161,16 @@ mod tests { #[test] fn empty_vectors_returns_none() { - let result = build_collection_codec("rabitq", &[], 8, 16, 100, 42); + let result = build_collection_codec("rabitq", &[], 8, 16, 100, 42).unwrap(); assert!(result.is_none(), "empty vectors should return None"); } #[test] fn len_and_is_empty() { let vecs = make_vectors(20, 4); - let codec = build_collection_codec("bbq", &vecs, 4, 8, 50, 1).unwrap(); + let codec = build_collection_codec("bbq", &vecs, 4, 8, 50, 1) + .unwrap() + .unwrap(); assert_eq!(codec.len(), 20); assert!(!codec.is_empty()); } @@ -163,9 +178,45 @@ mod tests { #[test] fn quantization_tag() { let vecs = make_vectors(10, 4); - let rabitq = build_collection_codec("rabitq", &vecs, 4, 8, 50, 1).unwrap(); + let rabitq = build_collection_codec("rabitq", &vecs, 4, 8, 50, 1) + .unwrap() + .unwrap(); assert_eq!(rabitq.quantization(), "rabitq"); - let bbq = build_collection_codec("bbq", &vecs, 4, 8, 50, 1).unwrap(); + let bbq = build_collection_codec("bbq", &vecs, 4, 8, 50, 1) + .unwrap() + .unwrap(); assert_eq!(bbq.quantization(), "bbq"); } + + #[test] + fn wrong_dimension_query_is_a_typed_error() { + let vecs = make_vectors(20, 4); + for tag in ["rabitq", "bbq"] { + let mut codec = build_collection_codec(tag, &vecs, 4, 8, 50, 1) + .unwrap() + .unwrap(); + assert!(matches!( + codec.search(&[0.0; 3], 5, 20), + Err(VectorError::DimensionMismatch { + expected: 4, + got: 3 + }) + )); + assert!(matches!( + codec.insert(99, &[0.0; 5]), + Err(VectorError::DimensionMismatch { + expected: 4, + got: 5 + }) + )); + } + let short = vec![(0_u32, vec![0.0_f32; 3])]; + assert!(matches!( + build_collection_codec("rabitq", &short, 4, 8, 50, 1), + Err(VectorError::DimensionMismatch { + expected: 4, + got: 3 + }) + )); + } } diff --git a/nodedb-vector/src/collection/ivf_mode.rs b/nodedb-vector/src/collection/ivf_mode.rs new file mode 100644 index 000000000..7fc1a467c --- /dev/null +++ b/nodedb-vector/src/collection/ivf_mode.rs @@ -0,0 +1,283 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The IVF-PQ mode of a `VectorCollection`. +//! +//! An `IvfPq` collection buffers its vectors in the growing segment until it +//! holds the training threshold, `max(ivf_cells, pq_k)`. The growing segment +//! is searched exactly, checkpointed, and rebuilt by WAL replay, so the +//! buffer is durable and searchable with no extra machinery. It never seals. +//! +//! [`VectorCollection::train_ivf`] trains the IVF centroids and PQ codebooks +//! on every live vector, moves each one into the trained index under its +//! global id, and empties the segments. The move happens in one call on the +//! owning core, so no search sees a vector twice or misses one. Later inserts +//! land in the trained index. + +use nodedb_mem::ScopedMemory; + +use crate::error::VectorError; +use crate::index_config::{IndexConfig, IndexType}; +use crate::ivf::IvfPqIndex; + +use super::lifecycle::VectorCollection; +use super::lifecycle_insert_ops::sealed_vector; + +impl VectorCollection { + /// Whether the collection is configured as an IVF-PQ index. + pub fn is_ivf(&self) -> bool { + self.index_config.index_type == IndexType::IvfPq + } + + /// The full index configuration. + pub fn index_config(&self) -> &IndexConfig { + &self.index_config + } + + /// Replace the index configuration. The HNSW parameters the collection + /// holds stay: its segments were built with them. + pub fn set_index_config(&mut self, config: IndexConfig) { + self.index_config = IndexConfig { + hnsw: self.params.clone(), + ..config + }; + } + + /// Live vectors an `IvfPq` collection needs before it trains. + pub fn ivf_training_threshold(&self) -> usize { + self.index_config.to_ivf_params().training_threshold() + } + + /// Whether an `IvfPq` collection is untrained and holds the training + /// threshold of live vectors. + pub fn needs_ivf_training(&self) -> bool { + self.is_ivf() && self.ivf.is_none() && self.live_count() >= self.ivf_training_threshold() + } + + /// The trained IVF-PQ index, if any. + pub fn ivf_index(&self) -> Option<&IvfPqIndex> { + self.ivf.as_ref() + } + + /// Train the IVF-PQ index on every live vector and move them all into it. + /// `trained_at_ms` stamps the training, in Unix milliseconds. + /// + /// Fails with [`VectorError::InvalidInput`] when the collection is not + /// `IvfPq`, is already trained, or holds fewer live vectors than the + /// training threshold. A training error propagates. Any failure leaves + /// the collection unchanged. + pub fn train_ivf( + &mut self, + memory: ScopedMemory, + trained_at_ms: u64, + ) -> Result<(), VectorError> { + if !self.is_ivf() || self.ivf.is_some() { + return Err(VectorError::InvalidInput { + detail: "IVF-PQ training needs an untrained ivf_pq collection".into(), + }); + } + let live = self.gather_live_vectors(); + let threshold = self.ivf_training_threshold(); + if live.len() < threshold { + return Err(VectorError::InvalidInput { + detail: format!( + "IVF-PQ training needs {threshold} live vectors; the collection holds {}", + live.len() + ), + }); + } + let mut ivf = IvfPqIndex::new(self.dim, self.index_config.to_ivf_params()); + { + let refs: Vec<&[f32]> = live.iter().map(|(_, v)| v.as_slice()).collect(); + ivf.train(&refs, memory)?; + } + for (id, vector) in live { + ivf.insert_with_id(id, vector)?; + } + ivf.set_trained_at_ms(trained_at_ms); + self.clear_segments(); + self.ivf = Some(ivf); + Ok(()) + } + + /// Every live FP32 vector outside the IVF index, keyed by global id. + fn gather_live_vectors(&self) -> Vec<(u32, Vec)> { + let mut out = Vec::with_capacity(self.live_count()); + for seg in &self.sealed { + for local in 0..seg.index.len() as u32 { + if !seg.index.is_deleted(local) + && let Some(v) = sealed_vector(seg, local) + { + out.push((seg.base_id + local, v)); + } + } + } + for seg in &self.building { + for local in 0..seg.flat.len() as u32 { + if let Some(v) = seg.flat.get_vector(local) { + out.push((seg.base_id + local, v.to_vec())); + } + } + } + for local in 0..self.growing.len() as u32 { + if let Some(v) = self.growing.get_vector(local) { + out.push((self.growing_base_id + local, v.to_vec())); + } + } + out + } +} + +#[cfg(test)] +mod tests { + use nodedb_types::Surrogate; + + use super::*; + use crate::test_support::test_memory; + + const DIM: usize = 8; + + /// An L2 `IvfPq` collection with 4 cells, all probed, and 256 PQ + /// centroids: the threshold is 256 vectors. + fn ivf_collection() -> VectorCollection { + let config = IndexConfig { + hnsw: crate::hnsw::HnswParams { + metric: crate::distance::DistanceMetric::L2, + ..crate::hnsw::HnswParams::default() + }, + index_type: IndexType::IvfPq, + pq_m: 4, + ivf_cells: 4, + ivf_nprobe: 4, + ..IndexConfig::default() + }; + VectorCollection::with_index_config(DIM, config) + } + + /// Distinct per `i`: the first component is `i + 1`. + fn vector(i: usize) -> Vec { + [1, 7, 11, 13, 17, 19, 23, 29] + .iter() + .map(|&m| (if m == 1 { i + 1 } else { i % m + 1 }) as f32) + .collect() + } + + fn all_ids(coll: &VectorCollection, query: &[f32]) -> Vec { + let mut ids: Vec = coll + .search(query, 10_000, 64) + .unwrap() + .iter() + .map(|r| r.id) + .collect(); + ids.sort_unstable(); + ids + } + + #[test] + fn below_the_threshold_vectors_wait_in_the_exact_buffer() { + let mut coll = ivf_collection(); + assert_eq!(coll.ivf_training_threshold(), 256); + for i in 0..10 { + coll.insert_with_surrogate(vector(i), Surrogate::new(i as u32 + 1)) + .unwrap(); + } + assert!(!coll.needs_ivf_training()); + assert!(!coll.needs_seal()); + let hit = &coll.search(&vector(3), 1, 64).unwrap()[0]; + assert_eq!(hit.id, 3); + assert_eq!(hit.distance, 0.0, "the buffer is searched exactly"); + } + + #[test] + fn training_moves_every_vector_once_and_later_inserts_follow() { + let mut coll = ivf_collection(); + for i in 0..256 { + coll.insert_with_surrogate(vector(i), Surrogate::new(i as u32 + 1)) + .unwrap(); + } + coll.delete(5); + coll.insert(vector(256)).unwrap(); + assert!(coll.needs_ivf_training()); + let before = all_ids(&coll, &vector(0)); + + coll.train_ivf(test_memory(), 42).unwrap(); + + assert!(!coll.needs_ivf_training()); + assert!(coll.growing_is_empty()); + let ivf = coll.ivf_index().unwrap(); + assert_eq!(ivf.trained_on(), 256); + assert_eq!(ivf.trained_at_ms(), 42); + assert_eq!( + all_ids(&coll, &vector(0)), + before, + "no vector lost or doubled" + ); + + let id = coll + .insert_with_surrogate(vector(300), Surrogate::new(9_999)) + .unwrap(); + assert_eq!(id, 257); + assert!( + coll.growing_is_empty(), + "a trained collection inserts into IVF" + ); + assert_eq!(coll.search(&vector(300), 1, 64).unwrap()[0].id, 257); + assert_eq!( + coll.vector_for_surrogate(Surrogate::new(9_999)), + Some(vector(300)) + ); + assert!(coll.delete_by_surrogate(Surrogate::new(9_999))); + assert!(coll.search(&vector(300), 1, 64).unwrap()[0].id != 257); + } + + #[test] + fn training_below_the_threshold_is_refused() { + let mut coll = ivf_collection(); + coll.insert(vector(0)).unwrap(); + assert!(matches!( + coll.train_ivf(test_memory(), 0), + Err(VectorError::InvalidInput { .. }) + )); + assert!(coll.ivf_index().is_none()); + assert_eq!(coll.live_count(), 1); + } + + #[test] + fn a_trained_collection_survives_a_checkpoint() { + let mut coll = ivf_collection(); + for i in 0..260 { + coll.insert(vector(i)).unwrap(); + } + coll.train_ivf(test_memory(), 7).unwrap(); + coll.insert(vector(400)).unwrap(); + coll.delete(9); + + let bytes = coll.checkpoint_to_bytes(None).unwrap(); + let restored = VectorCollection::from_checkpoint(&bytes, None, test_memory()).unwrap(); + + assert!(restored.is_ivf()); + assert_eq!(restored.ivf_index().unwrap().trained_at_ms(), 7); + assert_eq!(all_ids(&restored, &vector(0)), all_ids(&coll, &vector(0))); + assert_eq!(restored.search(&vector(400), 1, 64).unwrap()[0].id, 260); + } + + #[test] + fn an_untrained_buffer_survives_a_checkpoint_and_trains_after() { + let mut coll = ivf_collection(); + for i in 0..100 { + coll.insert(vector(i)).unwrap(); + } + let bytes = coll.checkpoint_to_bytes(None).unwrap(); + let mut restored = VectorCollection::from_checkpoint(&bytes, None, test_memory()).unwrap(); + assert!(restored.is_ivf()); + assert!(restored.ivf_index().is_none()); + for i in 100..256 { + restored.insert(vector(i)).unwrap(); + } + assert!(restored.needs_ivf_training()); + restored.train_ivf(test_memory(), 1).unwrap(); + assert_eq!( + all_ids(&restored, &vector(0)), + (0..256).collect::>() + ); + } +} diff --git a/nodedb-vector/src/collection/lifecycle.rs b/nodedb-vector/src/collection/lifecycle.rs index d27fe109b..32457e42b 100644 --- a/nodedb-vector/src/collection/lifecycle.rs +++ b/nodedb-vector/src/collection/lifecycle.rs @@ -1,6 +1,8 @@ // SPDX-License-Identifier: Apache-2.0 -//! VectorCollection lifecycle: insert, delete, seal, complete_build, compact. +//! VectorCollection lifecycle: construction, seal, counts and settings. +//! +//! Installing finished builds lives in `build`. //! //! Identity model: every vector inserted into the collection is bound to //! a global `Surrogate` allocated by the Control Plane before the engine @@ -16,16 +18,18 @@ use std::collections::HashMap; -use nodedb_mem::ScopedMemory; use nodedb_types::{Surrogate, VectorQuantization}; use crate::flat::FlatIndex; -use crate::hnsw::{HnswIndex, HnswParams}; -use crate::index_config::{IndexConfig, IndexType}; +use crate::hnsw::HnswParams; +use crate::index_config::IndexConfig; +use crate::ivf::IvfPqIndex; use super::codec_dispatch::CollectionCodec; use super::payload_index::PayloadIndexSet; -use super::segment::{BuildRequest, BuildingSegment, DEFAULT_SEAL_THRESHOLD, SealedSegment}; +use super::segment::{ + BuildKind, BuildRequest, BuildingSegment, DEFAULT_SEAL_THRESHOLD, SealedSegment, +}; /// Manages all vector segments for a single collection (one index key). /// @@ -71,6 +75,12 @@ pub struct VectorCollection { /// Coexists with sealed segments — for codec-dispatched collections the /// per-segment Sq8 builder is skipped and this index is used instead. pub codec_dispatch: Option, + /// Trained IVF-PQ index of an `IvfPq` collection. `None` until the + /// collection holds the training threshold of vectors: until then its + /// vectors wait in the growing segment, searched exactly. Training moves + /// every vector into this index under its global id, and later inserts + /// land here. + pub(crate) ivf: Option, /// Quantization mode requested at collection-creation time. /// /// When `!= None && != Sq8`, each call to `complete_build` additionally @@ -111,6 +121,10 @@ pub struct VectorCollection { /// gating its siblings). Folded into the persisted watermark at checkpoint /// save time via `max(checkpoint_wal_lsn, applied_wal_lsn)`. pub(crate) applied_wal_lsn: u64, + /// HNSW builds installed since the collection was opened. + pub(crate) builds_completed: u64, + /// HNSW builds that failed since the collection was opened. + pub(crate) builds_failed: u64, } impl VectorCollection { @@ -159,11 +173,14 @@ impl VectorCollection { seal_threshold, index_config: config, codec_dispatch: None, + ivf: None, quantization: VectorQuantization::default(), payload: PayloadIndexSet::default(), arena_index: None, checkpoint_wal_lsn: 0, applied_wal_lsn: 0, + builds_completed: 0, + builds_failed: 0, } } @@ -203,24 +220,34 @@ impl VectorCollection { Self::with_seal_threshold(dim, params, DEFAULT_SEAL_THRESHOLD) } - /// Check if the growing segment should be sealed. + /// Set the growing-segment size that triggers a seal. A zero threshold + /// is raised to one vector. + pub fn set_seal_threshold(&mut self, threshold: usize) { + self.seal_threshold = threshold.max(1); + } + + /// Check if the growing segment should be sealed. An `IvfPq` collection + /// never seals: its growing segment is the buffer IVF-PQ training reads. pub fn needs_seal(&self) -> bool { - self.growing.len() >= self.seal_threshold + !self.is_ivf() && self.growing.len() >= self.seal_threshold } - /// Seal the growing segment and return a build request. + /// Seal the growing segment and return a build request. `None` for an + /// empty growing segment or an `IvfPq` collection. pub fn seal(&mut self, key: &str) -> Option { - if self.growing.is_empty() { + if self.growing.is_empty() || self.is_ivf() { return None; } let segment_id = self.next_segment_id; self.next_segment_id += 1; + // Soft-deleted vectors go too: the built graph keeps every local id, + // and the tombstones are applied when the build is installed. let count = self.growing.len(); let mut vectors = Vec::with_capacity(count); for i in 0..count as u32 { - if let Some(v) = self.growing.get_vector(i) { + if let Some(v) = self.growing.get_vector_raw(i) { vectors.push(v.to_vec()); } } @@ -241,65 +268,13 @@ impl VectorCollection { Some(BuildRequest { key: key.to_string(), segment_id, + kind: BuildKind::Seal, vectors, dim: self.dim, params: self.params.clone(), }) } - /// Accept a completed HNSW build from the background thread. - /// - /// After promoting the segment to sealed, rebuilds the collection-level - /// codec-dispatch index when `self.quantization` is `RaBitQ` or `Bbq`. - /// The rebuild trains over all vectors so the codec index always covers - /// every sealed segment. - pub fn complete_build(&mut self, segment_id: u32, index: HnswIndex, memory: ScopedMemory) { - if let Some(pos) = self - .building - .iter() - .position(|b| b.segment_id == segment_id) - { - let building = self.building.remove(pos); - let use_codec_dispatch = matches!( - self.quantization, - VectorQuantization::RaBitQ | VectorQuantization::Bbq - ); - let use_pq = !use_codec_dispatch && self.index_config.index_type == IndexType::HnswPq; - let (sq8, pq) = if use_codec_dispatch { - (None, None) - } else if use_pq { - ( - None, - Self::build_pq_for_index(&index, self.index_config.pq_m, memory.clone()), - ) - } else { - (Self::build_sq8_for_index(&index), None) - }; - let (tier, mmap_vectors) = - self.resolve_tier_for_build(segment_id, building.base_id, &index, &memory); - - self.sealed.push(SealedSegment { - index, - base_id: building.base_id, - sq8, - pq, - tier, - mmap_vectors, - }); - - if use_codec_dispatch { - let tag = match self.quantization { - VectorQuantization::RaBitQ => "rabitq", - VectorQuantization::Bbq => "bbq", - _ => unreachable!( - "invariant: use_codec_dispatch is only true for RaBitQ and Bbq quantization variants" - ), - }; - self.build_codec_dispatch(tag); - } - } - } - /// Access sealed segments (read-only). pub fn sealed_segments(&self) -> &[SealedSegment] { &self.sealed @@ -316,7 +291,7 @@ impl VectorCollection { } pub fn len(&self) -> usize { - let mut total = self.growing.len(); + let mut total = self.growing.len() + self.ivf.as_ref().map_or(0, IvfPqIndex::len); for seg in &self.sealed { total += seg.index.len(); } @@ -327,7 +302,8 @@ impl VectorCollection { } pub fn live_count(&self) -> usize { - let mut total = self.growing.live_count(); + let mut total = + self.growing.live_count() + self.ivf.as_ref().map_or(0, IvfPqIndex::live_count); for seg in &self.sealed { total += seg.index.live_count(); } diff --git a/nodedb-vector/src/collection/lifecycle_compact.rs b/nodedb-vector/src/collection/lifecycle_compact.rs index 7d7dc13bc..55db42cb7 100644 --- a/nodedb-vector/src/collection/lifecycle_compact.rs +++ b/nodedb-vector/src/collection/lifecycle_compact.rs @@ -21,8 +21,24 @@ impl VectorCollection { /// no matching entry in `building` and is ignored, and an mmap file name /// is never reused. The mmap file of each dropped sealed segment is /// removed from disk. + /// + /// An `IvfPq` collection drops its trained index too and buffers again: + /// its next training reads the vectors inserted after the truncate. pub fn truncate(&mut self) -> usize { let dropped = self.live_count(); + self.clear_segments(); + self.ivf = None; + self.surrogate_map.clear(); + self.surrogate_to_local.clear(); + self.multi_doc_map.clear(); + self.payload.clear_rows(); + dropped + } + + /// Empty the growing segment and drop every sealed segment, in-flight + /// build and codec-dispatch index. The next insert keeps the id counter. + /// The mmap file of each dropped sealed segment is removed from disk. + pub(super) fn clear_segments(&mut self) { self.growing = FlatIndex::new(self.dim, self.params.metric); self.growing_base_id = self.next_id; for seg in self.sealed.drain(..) { @@ -41,21 +57,17 @@ impl VectorCollection { } self.building.clear(); self.mmap_segment_count = 0; - self.surrogate_map.clear(); - self.surrogate_to_local.clear(); - self.multi_doc_map.clear(); self.codec_dispatch = None; - self.payload.clear_rows(); - dropped } - /// Compact sealed segments by removing tombstoned nodes. + /// Compact sealed segments and the IVF-PQ index by removing tombstoned + /// nodes. /// /// Rewrites `surrogate_map` and `multi_doc_map` for every sealed /// segment so that global ids continue to resolve to the correct - /// surrogate after local-id renumbering. + /// surrogate after local-id renumbering. IVF-PQ entries keep their ids. pub fn compact(&mut self) -> usize { - let mut total_removed = 0; + let mut total_removed = self.ivf.as_mut().map_or(0, |ivf| ivf.compact()); for seg in &mut self.sealed { let base_id = seg.base_id; let (removed, id_map) = seg.index.compact_with_map(); @@ -125,6 +137,13 @@ impl VectorCollection { pub fn export_snapshot(&self) -> Result, crate::error::VectorError> { let mut result = Vec::new(); + if let Some(ivf) = &self.ivf { + for (vid, data) in ivf.live_vectors() { + let surrogate = self.surrogate_map.get(&vid).copied(); + result.push((vid, data, surrogate)); + } + } + for i in 0..self.growing.len() as u32 { let vid = self.growing_base_id + i; if let Some(data) = self.growing.get_vector(i) { @@ -171,7 +190,7 @@ mod tests { fields.insert("owner".to_string(), Value::String("a".into())); for (i, v) in [[1.0, 0.0], [0.0, 1.0], [1.0, 1.0]].into_iter().enumerate() { let s = Surrogate::new(i as u32 + 1); - let id = coll.insert_with_surrogate(v.to_vec(), s); + let id = coll.insert_with_surrogate(v.to_vec(), s).unwrap(); coll.payload.insert_row(id, &fields); } assert!( @@ -204,7 +223,7 @@ mod tests { assert!(hits.is_empty(), "payload rows cleared"); let s = Surrogate::new(42); - let id = coll.insert_with_surrogate(vec![0.5, 0.5], s); + let id = coll.insert_with_surrogate(vec![0.5, 0.5], s).unwrap(); assert_eq!(coll.local_for_surrogate(s), Some(id)); assert_eq!(coll.live_count(), 1); } diff --git a/nodedb-vector/src/collection/lifecycle_insert_ops.rs b/nodedb-vector/src/collection/lifecycle_insert_ops.rs index f962f1508..3acd15324 100644 --- a/nodedb-vector/src/collection/lifecycle_insert_ops.rs +++ b/nodedb-vector/src/collection/lifecycle_insert_ops.rs @@ -5,14 +5,26 @@ use nodedb_types::Surrogate; use super::lifecycle::VectorCollection; +use crate::error::{VectorError, check_dim}; impl VectorCollection { - /// Insert a vector. Returns the global vector ID. - pub fn insert(&mut self, vector: Vec) -> u32 { + /// Insert a vector. Returns the global vector ID. A trained IVF-PQ + /// collection inserts into its IVF index, every other collection into + /// the growing segment. + /// + /// A vector without the collection dimension fails with + /// [`VectorError::DimensionMismatch`] and changes nothing. + pub fn insert(&mut self, vector: Vec) -> Result { + check_dim(self.dim, vector.len())?; let id = self.next_id; - self.growing.insert(vector); + match &mut self.ivf { + Some(ivf) => ivf.insert_with_id(id, vector)?, + None => { + self.growing.insert(vector)?; + } + } self.next_id += 1; - id + Ok(id) } /// Insert a vector with an associated surrogate. The surrogate is @@ -24,31 +36,69 @@ impl VectorCollection { /// re-insert never leaves an unreachable node scoring in searches. /// The caller owns the payload bitmap entries of the old node and /// removes them with [`Self::local_for_surrogate`] before this call. - pub fn insert_with_surrogate(&mut self, vector: Vec, surrogate: Surrogate) -> u32 { + /// + /// A vector without the collection dimension fails with + /// [`VectorError::DimensionMismatch`] before the old binding is touched. + pub fn insert_with_surrogate( + &mut self, + vector: Vec, + surrogate: Surrogate, + ) -> Result { + check_dim(self.dim, vector.len())?; if surrogate != Surrogate::ZERO && let Some(old) = self.surrogate_to_local.get(&surrogate).copied() { self.delete_inner(old); self.surrogate_map.remove(&old); } - let id = self.insert(vector); + let id = self.insert(vector)?; if surrogate != Surrogate::ZERO { self.surrogate_map.insert(id, surrogate); self.surrogate_to_local.insert(surrogate, id); } - id + Ok(id) + } + + /// Insert a batch of vectors, the `i`-th bound to `surrogates[i]` + /// ([`Surrogate::ZERO`] when the slice is shorter). Returns the global ids + /// in batch order. + /// + /// Every vector is checked against the collection dimension before any is + /// inserted, so a mismatch fails with [`VectorError::DimensionMismatch`] + /// and inserts none. + pub fn insert_batch_with_surrogates( + &mut self, + vectors: &[Vec], + surrogates: &[Surrogate], + ) -> Result, VectorError> { + for v in vectors { + check_dim(self.dim, v.len())?; + } + let mut ids = Vec::with_capacity(vectors.len()); + for (i, v) in vectors.iter().enumerate() { + let surrogate = surrogates.get(i).copied().unwrap_or(Surrogate::ZERO); + ids.push(self.insert_with_surrogate(v.clone(), surrogate)?); + } + Ok(ids) } /// Insert multiple vectors for a single document (ColBERT-style). /// All N vectors are bound to the same `document_surrogate`. + /// + /// Every vector is checked against the collection dimension before any + /// is inserted, so a mismatch fails with + /// [`VectorError::DimensionMismatch`] and inserts none. pub fn insert_multi_vector( &mut self, vectors: &[&[f32]], document_surrogate: Surrogate, - ) -> Vec { + ) -> Result, VectorError> { + for v in vectors { + check_dim(self.dim, v.len())?; + } let mut ids = Vec::with_capacity(vectors.len()); for &v in vectors { - let id = self.insert(v.to_vec()); + let id = self.insert(v.to_vec())?; if document_surrogate != Surrogate::ZERO { self.surrogate_map.insert(id, document_surrogate); } @@ -57,7 +107,7 @@ impl VectorCollection { if document_surrogate != Surrogate::ZERO { self.multi_doc_map.insert(document_surrogate, ids.clone()); } - ids + Ok(ids) } /// Delete all vectors belonging to a multi-vector document. @@ -96,6 +146,11 @@ impl VectorCollection { } pub(super) fn delete_inner(&mut self, id: u32) -> bool { + if let Some(ivf) = &mut self.ivf + && ivf.contains(id) + { + return ivf.delete(id); + } if id >= self.growing_base_id { let local = id - self.growing_base_id; if (local as usize) < self.growing.len() { @@ -124,6 +179,11 @@ impl VectorCollection { /// The live FP32 vector stored under global `id`, whichever segment /// holds it. `None` for an unknown or soft-deleted id. pub fn vector_for_id(&self, id: u32) -> Option> { + if let Some(ivf) = &self.ivf + && ivf.contains(id) + { + return ivf.get_vector(id).map(<[f32]>::to_vec); + } if id >= self.growing_base_id { let local = id - self.growing_base_id; if (local as usize) < self.growing.len() { @@ -168,13 +228,16 @@ impl VectorCollection { /// Un-delete a previously soft-deleted vector (for transaction rollback). /// - /// Symmetric to [`Self::delete_inner`]: the vector may live in the growing - /// segment (the common case for a just-inserted vector), a sealed HNSW - /// segment, or an in-flight building segment — reverse the tombstone - /// wherever it landed. Only clearing sealed tombstones (the prior behavior) - /// silently failed to restore growing/building vectors, leaving a - /// rolled-back delete permanently unsearchable. + /// Symmetric to [`Self::delete_inner`]: the vector may live in the IVF + /// index, the growing segment (the common case for a just-inserted + /// vector), a sealed HNSW segment, or an in-flight building segment. The + /// tombstone is reversed wherever it landed. pub fn undelete(&mut self, id: u32) -> bool { + if let Some(ivf) = &mut self.ivf + && ivf.contains(id) + { + return ivf.undelete(id); + } if id >= self.growing_base_id { let local = id - self.growing_base_id; if (local as usize) < self.growing.len() { @@ -204,7 +267,7 @@ impl VectorCollection { /// The FP32 vector at `local` in a sealed segment: the mmap tier when the /// segment lives there, else the HNSW node (decoded from a narrow dtype or /// fetched from the segment backing when the node holds no local copy). -fn sealed_vector(seg: &super::segment::SealedSegment, local: u32) -> Option> { +pub(super) fn sealed_vector(seg: &super::segment::SealedSegment, local: u32) -> Option> { if let Some(mmap) = &seg.mmap_vectors { return mmap.get_vector(local).map(<[f32]>::to_vec); } @@ -233,8 +296,8 @@ mod tests { fn re_insert_under_the_same_surrogate_leaves_one_live_node() { let mut coll = collection(); let s = Surrogate::new(7); - let first = coll.insert_with_surrogate(vec![1.0, 0.0], s); - let second = coll.insert_with_surrogate(vec![0.0, 1.0], s); + let first = coll.insert_with_surrogate(vec![1.0, 0.0], s).unwrap(); + let second = coll.insert_with_surrogate(vec![0.0, 1.0], s).unwrap(); assert_ne!(first, second); assert_eq!(coll.live_count(), 1, "the old node must be tombstoned"); assert_eq!(coll.local_for_surrogate(s), Some(second)); @@ -246,10 +309,10 @@ mod tests { fn delete_then_insert_under_the_same_surrogate_leaves_one_live_node() { let mut coll = collection(); let s = Surrogate::new(9); - let first = coll.insert_with_surrogate(vec![1.0, 0.0], s); + let first = coll.insert_with_surrogate(vec![1.0, 0.0], s).unwrap(); assert!(coll.delete_by_surrogate(s)); assert_eq!(coll.local_for_surrogate(s), None); - let second = coll.insert_with_surrogate(vec![0.0, 1.0], s); + let second = coll.insert_with_surrogate(vec![0.0, 1.0], s).unwrap(); assert_ne!(first, second); assert_eq!(coll.live_count(), 1); assert_eq!(coll.local_for_surrogate(s), Some(second)); @@ -260,7 +323,7 @@ mod tests { fn delete_by_surrogate_is_idempotent() { let mut coll = collection(); let s = Surrogate::new(3); - coll.insert_with_surrogate(vec![1.0, 0.0], s); + coll.insert_with_surrogate(vec![1.0, 0.0], s).unwrap(); assert!(coll.delete_by_surrogate(s)); assert!(!coll.delete_by_surrogate(s)); assert_eq!(coll.live_count(), 0); @@ -270,7 +333,7 @@ mod tests { fn vector_for_surrogate_reads_the_growing_segment_and_hides_deletes() { let mut coll = collection(); let s = Surrogate::new(11); - coll.insert_with_surrogate(vec![0.5, 0.25], s); + coll.insert_with_surrogate(vec![0.5, 0.25], s).unwrap(); assert_eq!(coll.vector_for_surrogate(s), Some(vec![0.5, 0.25])); assert!(coll.delete_by_surrogate(s)); assert_eq!(coll.vector_for_surrogate(s), None); diff --git a/nodedb-vector/src/collection/mod.rs b/nodedb-vector/src/collection/mod.rs index 88da448d2..452daf7e1 100644 --- a/nodedb-vector/src/collection/mod.rs +++ b/nodedb-vector/src/collection/mod.rs @@ -1,15 +1,18 @@ // SPDX-License-Identifier: Apache-2.0 pub mod budget; +pub mod build; pub mod checkpoint; pub mod codec_build; pub mod codec_dispatch; +pub mod ivf_mode; pub mod lifecycle; pub mod lifecycle_compact; pub mod lifecycle_insert_ops; pub mod lifecycle_reindex; pub mod payload_index; pub mod quantize; +pub mod rollback; pub mod search; pub mod segment; pub mod stats; @@ -17,7 +20,8 @@ pub mod tier; pub use lifecycle::VectorCollection; pub use payload_index::{FilterPredicate, PayloadIndex, PayloadIndexKind, PayloadIndexSet}; +pub use rollback::VectorWriteMark; pub use segment::{ - BuildComplete, BuildRequest, BuildingSegment, DEFAULT_SEAL_THRESHOLD, SealedSegment, + BuildComplete, BuildKind, BuildRequest, BuildingSegment, DEFAULT_SEAL_THRESHOLD, SealedSegment, }; pub use tier::StorageTier; diff --git a/nodedb-vector/src/collection/payload_index.rs b/nodedb-vector/src/collection/payload_index.rs index c7daeaba1..581111c31 100644 --- a/nodedb-vector/src/collection/payload_index.rs +++ b/nodedb-vector/src/collection/payload_index.rs @@ -295,6 +295,22 @@ impl PayloadIndexSet { } } + /// A set with the same registered fields and kinds and no rows. + pub fn definitions_only(&self) -> Self { + Self { + indexes: self + .indexes + .iter() + .map(|(field, index)| { + ( + field.clone(), + PayloadIndex::new(index.field.clone(), index.kind), + ) + }) + .collect(), + } + } + /// Drop every node from every index. The registered fields and their /// kinds stay, so the next `insert_row` indexes the same columns. pub fn clear_rows(&mut self) { diff --git a/nodedb-vector/src/collection/quantize.rs b/nodedb-vector/src/collection/quantize.rs index e3a6c4eba..ea3e2881d 100644 --- a/nodedb-vector/src/collection/quantize.rs +++ b/nodedb-vector/src/collection/quantize.rs @@ -70,7 +70,9 @@ impl VectorCollection { return None; } - let codec = Sq8Codec::calibrate(&refs, dim); + // The refs are live vectors of this index, so calibration fails only + // for a zero dimension, which has nothing to quantize. + let codec = Sq8Codec::calibrate(&refs, dim).ok()?; let mut data = Vec::with_capacity(dim * n); for i in 0..n { @@ -85,7 +87,7 @@ impl VectorCollection { } /// Train a PQ codec from a built HNSW index's live vectors, tracking - /// codebook allocations against `memory`. + /// codebook allocations against `memory`, and encode every node. pub fn build_pq_for_index( index: &HnswIndex, pq_m: usize, @@ -109,8 +111,29 @@ impl VectorCollection { } let refs_slices: Vec<&[f32]> = refs.iter().map(|v| v.as_slice()).collect(); let k = 256usize.min(refs.len()); - let codec = PqCodec::train(&refs_slices, dim, pq_m, k, 20, memory); - let codes = codec.encode_batch(&refs_slices).ok()?; + // A PQ shape the codec cannot train (a codebook over its byte limit, + // a dimension over its decode limit) leaves the segment on plain + // HNSW, which answers the same queries exactly. + let codec = match PqCodec::train(&refs_slices, dim, pq_m, k, 20, memory) { + Ok(codec) => codec, + Err(e) => { + tracing::warn!(error = %e, dim, pq_m, "PQ training refused; segment stays unquantized"); + return None; + } + }; + // One code per local node id, soft-deleted nodes included: search + // reads the code of node `id` at `id * pq_m`. + let zero = vec![0.0f32; dim]; + let all: Vec<&[f32]> = (0..n as u32) + .map(|i| index.get_vector(i).unwrap_or(zero.as_slice())) + .collect(); + let codes = match codec.encode_batch(&all) { + Ok(codes) => codes, + Err(e) => { + tracing::warn!(error = %e, dim, pq_m, "PQ encoding refused; segment stays unquantized"); + return None; + } + }; Some((codec, codes)) } } diff --git a/nodedb-vector/src/collection/rollback.rs b/nodedb-vector/src/collection/rollback.rs new file mode 100644 index 000000000..7a574d019 --- /dev/null +++ b/nodedb-vector/src/collection/rollback.rs @@ -0,0 +1,262 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Roll a collection back to the state it held before a write. +//! +//! A [`VectorWriteMark`] records what a write can change: the id counter, +//! the binding of every surrogate the write names, which of the named nodes +//! were live, and the multi-vector documents it names. Rolling back drops +//! every node inserted since the mark from the growing segment, so the id +//! counter returns to where it was and the next insert takes the same id it +//! would have taken without the write. Then it puts every binding and every +//! tombstone back. +//! +//! The inserted nodes must still sit in the growing segment or the IVF-PQ +//! index: a seal or an IVF-PQ training between the mark and the rollback +//! moves them out, and the rollback then reports that it cannot restore the +//! mark. + +use nodedb_types::Surrogate; + +use super::lifecycle::VectorCollection; +use crate::flat::FlatIndex; + +/// A collection's state before a write. See the module docs. +#[derive(Debug, Clone)] +pub struct VectorWriteMark { + next_id: u32, + growing_base_id: u32, + /// Each named surrogate with the node bound to it, if any. + bindings: Vec<(Surrogate, Option)>, + /// Named nodes that were live. + live: Vec, + /// Each named multi-vector document with its node list, if any. + multi_docs: Vec<(Surrogate, Option>)>, +} + +impl VectorCollection { + /// Whether node `id` exists and is not soft-deleted, whichever segment + /// holds it. + pub fn is_live(&self, id: u32) -> bool { + if let Some(ivf) = &self.ivf + && ivf.contains(id) + { + return !ivf.is_deleted(id); + } + if id >= self.growing_base_id { + let local = id - self.growing_base_id; + if (local as usize) < self.growing.len() { + return !self.growing.is_deleted(local); + } + } + for seg in &self.sealed { + if id >= seg.base_id { + let local = id - seg.base_id; + if (local as usize) < seg.index.len() { + return !seg.index.is_deleted(local); + } + } + } + for seg in &self.building { + if id >= seg.base_id { + let local = id - seg.base_id; + if (local as usize) < seg.flat.len() { + return !seg.flat.is_deleted(local); + } + } + } + false + } + + /// Mark the state a write naming `surrogates` and node `ids` can change. + pub fn write_mark(&self, surrogates: &[Surrogate], ids: &[u32]) -> VectorWriteMark { + let mut bindings = Vec::with_capacity(surrogates.len() + ids.len()); + let mut live = Vec::new(); + let mut multi_docs = Vec::with_capacity(surrogates.len()); + for &surrogate in surrogates { + let bound = self.surrogate_to_local.get(&surrogate).copied(); + bindings.push((surrogate, bound)); + if let Some(id) = bound + && self.is_live(id) + { + live.push(id); + } + let doc = self.multi_doc_map.get(&surrogate).cloned(); + if let Some(doc_ids) = &doc { + live.extend(doc_ids.iter().copied().filter(|id| self.is_live(*id))); + } + multi_docs.push((surrogate, doc)); + } + for &id in ids { + if !self.is_live(id) { + continue; + } + live.push(id); + if let Some(&surrogate) = self.surrogate_map.get(&id) { + bindings.push((surrogate, Some(id))); + } + } + VectorWriteMark { + next_id: self.next_id, + growing_base_id: self.growing_base_id, + bindings, + live, + multi_docs, + } + } + + /// Put the collection back to `mark`. Returns `false`, changing nothing, + /// when a seal or an IVF-PQ training moved the nodes inserted since the + /// mark out of the growing segment. + pub fn roll_back_to(&mut self, mark: VectorWriteMark) -> bool { + if self.growing_base_id != mark.growing_base_id || mark.next_id < self.growing_base_id { + return false; + } + for id in mark.next_id..self.next_id { + if let Some(surrogate) = self.surrogate_map.remove(&id) + && self.surrogate_to_local.get(&surrogate) == Some(&id) + { + self.surrogate_to_local.remove(&surrogate); + } + } + self.growing + .truncate((mark.next_id - self.growing_base_id) as usize); + if let Some(ivf) = &mut self.ivf { + ivf.roll_back_to(mark.next_id); + } + self.next_id = mark.next_id; + + for (surrogate, prior) in mark.bindings { + if let Some(current) = self.surrogate_to_local.remove(&surrogate) + && Some(current) != prior + { + self.surrogate_map.remove(¤t); + } + if let Some(id) = prior { + self.surrogate_to_local.insert(surrogate, id); + self.surrogate_map.insert(id, surrogate); + } + } + for (surrogate, prior) in mark.multi_docs { + match prior { + Some(ids) => { + for id in &ids { + self.surrogate_map.insert(*id, surrogate); + } + self.multi_doc_map.insert(surrogate, ids); + } + None => { + self.multi_doc_map.remove(&surrogate); + } + } + } + for id in mark.live { + if !self.is_live(id) { + self.undelete(id); + } + } + true + } + + /// An empty collection with this one's configuration and id counters: + /// same dimension, parameters, index config, quantization, seal + /// threshold, storage settings, registered payload fields and + /// watermarks. The next insert takes the id this collection's next insert + /// would have taken. + pub fn detached_empty(&self) -> Self { + let mut fresh = Self::with_seal_threshold_and_config( + self.dim, + self.index_config.clone(), + self.seal_threshold, + ); + fresh.params = self.params.clone(); + fresh.growing = FlatIndex::new(self.dim, self.params.metric); + fresh.next_id = self.next_id; + fresh.growing_base_id = self.next_id; + fresh.next_segment_id = self.next_segment_id; + fresh.data_dir = self.data_dir.clone(); + fresh.ram_budget_bytes = self.ram_budget_bytes; + fresh.mmap_fallback_count = self.mmap_fallback_count; + fresh.quantization = self.quantization; + fresh.payload = self.payload.definitions_only(); + fresh.arena_index = self.arena_index; + fresh.checkpoint_wal_lsn = self.checkpoint_wal_lsn; + fresh.applied_wal_lsn = self.applied_wal_lsn; + fresh + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::hnsw::HnswParams; + + fn collection() -> VectorCollection { + VectorCollection::new(2, HnswParams::default()) + } + + #[test] + fn a_rolled_back_insert_returns_the_id_counter_and_the_binding() { + let mut coll = collection(); + coll.insert_with_surrogate(vec![0.1, 0.2], Surrogate::new(1)) + .unwrap(); + let mark = coll.write_mark(&[Surrogate::new(2)], &[]); + coll.insert_with_surrogate(vec![0.3, 0.4], Surrogate::new(2)) + .unwrap(); + assert!(coll.roll_back_to(mark)); + assert_eq!(coll.local_for_surrogate(Surrogate::new(2)), None); + assert_eq!( + coll.insert_with_surrogate(vec![0.5, 0.6], Surrogate::new(3)) + .unwrap(), + 1, + "the next insert takes the id the rolled-back insert took" + ); + } + + #[test] + fn a_rolled_back_rebind_restores_the_replaced_node() { + let mut coll = collection(); + let first = coll + .insert_with_surrogate(vec![0.1, 0.2], Surrogate::new(7)) + .unwrap(); + let mark = coll.write_mark(&[Surrogate::new(7)], &[]); + coll.insert_with_surrogate(vec![0.3, 0.4], Surrogate::new(7)) + .unwrap(); + assert!(!coll.is_live(first)); + assert!(coll.roll_back_to(mark)); + assert!(coll.is_live(first)); + assert_eq!(coll.local_for_surrogate(Surrogate::new(7)), Some(first)); + } + + #[test] + fn a_rolled_back_delete_restores_the_node_and_its_binding() { + let mut coll = collection(); + let id = coll + .insert_with_surrogate(vec![0.1, 0.2], Surrogate::new(4)) + .unwrap(); + let mark = coll.write_mark(&[], &[id]); + coll.delete(id); + assert!(coll.roll_back_to(mark)); + assert!(coll.is_live(id)); + assert_eq!(coll.local_for_surrogate(Surrogate::new(4)), Some(id)); + } + + #[test] + fn a_seal_after_the_mark_refuses_the_rollback() { + let mut coll = VectorCollection::with_seal_threshold(2, HnswParams::default(), 1); + let mark = coll.write_mark(&[Surrogate::new(1)], &[]); + coll.insert_with_surrogate(vec![0.1, 0.2], Surrogate::new(1)) + .unwrap(); + assert!(coll.seal("k").is_some()); + assert!(!coll.roll_back_to(mark)); + } + + #[test] + fn a_detached_empty_collection_continues_the_id_counter() { + let mut coll = collection(); + coll.insert(vec![0.1, 0.2]).unwrap(); + coll.insert(vec![0.3, 0.4]).unwrap(); + let mut fresh = coll.detached_empty(); + assert_eq!(fresh.live_count(), 0); + assert_eq!(fresh.insert(vec![0.5, 0.6]).unwrap(), 2); + } +} diff --git a/nodedb-vector/src/collection/search.rs b/nodedb-vector/src/collection/search.rs index 32e4a0087..ecff5578f 100644 --- a/nodedb-vector/src/collection/search.rs +++ b/nodedb-vector/src/collection/search.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: Apache-2.0 -//! VectorCollection search: multi-segment merging with SQ8 reranking. +//! VectorCollection search: multi-segment merging with SQ8 reranking. A +//! trained IVF-PQ index answers beside the segments under global ids. //! //! The `search_with_payload_filter` method wires payload bitmap pre-filtering //! into the search path. When all referenced fields in the predicate are @@ -10,8 +11,9 @@ //! never silently dropped. use crate::distance::{DistanceMetric, distance}; -use crate::error::VectorError; +use crate::error::{VectorError, check_dim}; use crate::hnsw::SearchResult; +use crate::hnsw::search::decode_filter_bitmap; use super::lifecycle::VectorCollection; use super::payload_index::FilterPredicate; @@ -50,7 +52,7 @@ fn quantized_search( metric: DistanceMetric, ) -> Result, VectorError> { let rerank_k = top_k.saturating_mul(3).max(20); - let hnsw_candidates = seg.index.search(query, rerank_k, ef); + let hnsw_candidates = seg.index.search(query, rerank_k, ef)?; // Phase 1: rank candidates by quantized distance. let mut scored: Vec<(u32, f32)> = if let Some((codec, codes)) = &seg.pq { @@ -118,92 +120,95 @@ fn quantized_search( Ok(reranked) } -impl VectorCollection { - /// Search across all segments, merging results by distance. - pub fn search(&self, query: &[f32], top_k: usize, ef: usize) -> Vec { - // Codec-dispatch fast path: if a collection-level HnswCodecIndex has - // been built (RaBitQ or BBQ), use it exclusively for sealed-segment - // results and fall back to the growing/building flat segments only. - if let Some(ref dispatch) = self.codec_dispatch { - let mut all: Vec = Vec::new(); - - let codec_results = dispatch.search(query, top_k, ef); - for r in codec_results { - all.push(SearchResult { - id: r.id, - distance: r.distance, - }); - } - - // Growing segment (brute-force, not yet in codec index). - let growing_results = self.growing.search(query, top_k); - for mut r in growing_results { - r.id += self.growing_base_id; - all.push(r); - } +/// Search one sealed segment: through its quantized codec when it has one, +/// else its HNSW graph. A codec pass that exceeds the memory budget falls +/// back to the HNSW graph, which answers the same query from FP32 vectors. +/// Every other error fails the search. +fn search_sealed( + seg: &SealedSegment, + query: &[f32], + top_k: usize, + ef: usize, + metric: DistanceMetric, +) -> Result, VectorError> { + if seg.pq.is_none() && seg.sq8.is_none() { + return seg.index.search(query, top_k, ef); + } + match quantized_search(seg, query, top_k, ef, metric) { + Ok(results) => Ok(results), + Err(VectorError::BudgetExhausted(e)) => { + tracing::warn!(error = %e, "quantized search over budget; searching the HNSW graph"); + seg.index.search(query, top_k, ef) + } + Err(e) => Err(e), + } +} - // Building segments (brute-force while codec index rebuilds). - for seg in &self.building { - let results = seg.flat.search(query, top_k); - for mut r in results { - r.id += seg.base_id; - all.push(r); - } - } +/// Shift segment-local result ids to global ids and append them. +fn push_shifted(all: &mut Vec, results: Vec, base_id: u32) { + all.extend(results.into_iter().map(|mut r| { + r.id += base_id; + r + })); +} - all.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - all.truncate(top_k); - return all; - } +/// Order merged results by distance and keep the `top_k` nearest. +fn finish(mut all: Vec, top_k: usize) -> Vec { + all.sort_by(|a, b| { + a.distance + .partial_cmp(&b.distance) + .unwrap_or(std::cmp::Ordering::Equal) + }); + all.truncate(top_k); + all +} +impl VectorCollection { + /// Search across all segments, merging results by distance. + /// + /// A query without the collection dimension fails with + /// [`VectorError::DimensionMismatch`] before any segment is read. + pub fn search( + &self, + query: &[f32], + top_k: usize, + ef: usize, + ) -> Result, VectorError> { + check_dim(self.dim, query.len())?; let mut all: Vec = Vec::new(); - // Search growing segment (brute-force). - let growing_results = self.growing.search(query, top_k); - for mut r in growing_results { - r.id += self.growing_base_id; - all.push(r); - } - - // Search sealed segments. - for seg in &self.sealed { - let results = if seg.pq.is_some() || seg.sq8.is_some() { - match quantized_search(seg, query, top_k, ef, self.params.metric) { - Ok(r) => r, - Err(e) => { - tracing::warn!(error = %e, "quantized_search budget exhausted; skipping segment"); - seg.index.search(query, top_k, ef) - } - } - } else { - seg.index.search(query, top_k, ef) - }; - for mut r in results { - r.id += seg.base_id; - all.push(r); + // Codec-dispatch fast path: a collection-level HnswCodecIndex (RaBitQ + // or BBQ) answers for the sealed segments; the growing and building + // segments are read by brute force beside it. + if let Some(ref dispatch) = self.codec_dispatch { + all.extend( + dispatch + .search(query, top_k, ef)? + .into_iter() + .map(|r| SearchResult { + id: r.id, + distance: r.distance, + }), + ); + } else { + for seg in &self.sealed { + let results = search_sealed(seg, query, top_k, ef, self.params.metric)?; + push_shifted(&mut all, results, seg.base_id); } } + if let Some(ivf) = &self.ivf { + all.extend(ivf.search(query, top_k)?); + } - // Search building segments (brute-force while HNSW builds). + push_shifted( + &mut all, + self.growing.search(query, top_k)?, + self.growing_base_id, + ); for seg in &self.building { - let results = seg.flat.search(query, top_k); - for mut r in results { - r.id += seg.base_id; - all.push(r); - } + push_shifted(&mut all, seg.flat.search(query, top_k)?, seg.base_id); } - - all.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - all.truncate(top_k); - all + Ok(finish(all, top_k)) } /// Search across all segments using an explicit metric override. @@ -212,88 +217,56 @@ impl VectorCollection { /// during candidate reranking. Growing and building segments apply it exactly /// via brute-force. The HNSW graph structure was built with the collection /// metric; using a different metric affects the scoring but not graph traversal. + /// A codec-dispatch index scores with the collection metric. pub fn search_with_metric( &self, query: &[f32], top_k: usize, ef: usize, metric: DistanceMetric, - ) -> Vec { - // Codec-dispatch fast path: codec dispatch does not yet support per-query - // metric override — fall through to the non-codec path which does. - // When a codec index is active, we search only the growing/building - // segments with the override and add codec results with collection metric - // (approximate cross-metric search for the codec-indexed segments). - if let Some(ref dispatch) = self.codec_dispatch { - let mut all: Vec = Vec::new(); - let codec_results = dispatch.search(query, top_k, ef); - for r in codec_results { - all.push(SearchResult { - id: r.id, - distance: r.distance, - }); - } - for mut r in self.growing.search_with_metric(query, top_k, metric) { - r.id += self.growing_base_id; - all.push(r); - } - for seg in &self.building { - for mut r in seg.flat.search_with_metric(query, top_k, metric) { - r.id += seg.base_id; - all.push(r); - } - } - all.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - all.truncate(top_k); - return all; - } - + ) -> Result, VectorError> { + check_dim(self.dim, query.len())?; let mut all: Vec = Vec::new(); - for mut r in self.growing.search_with_metric(query, top_k, metric) { - r.id += self.growing_base_id; - all.push(r); - } - - for seg in &self.sealed { - let results = if seg.pq.is_some() || seg.sq8.is_some() { - match quantized_search(seg, query, top_k, ef, metric) { - Ok(r) => r, - Err(e) => { - tracing::warn!(error = %e, "quantized_search budget exhausted; skipping segment"); - seg.index.search(query, top_k, ef) - } - } - } else { - seg.index.search(query, top_k, ef) - }; - for mut r in results { - r.id += seg.base_id; - all.push(r); + if let Some(ref dispatch) = self.codec_dispatch { + all.extend( + dispatch + .search(query, top_k, ef)? + .into_iter() + .map(|r| SearchResult { + id: r.id, + distance: r.distance, + }), + ); + } else { + for seg in &self.sealed { + let results = search_sealed(seg, query, top_k, ef, metric)?; + push_shifted(&mut all, results, seg.base_id); } } + if let Some(ivf) = &self.ivf { + all.extend(ivf.search_with(query, top_k, metric, None)?); + } + push_shifted( + &mut all, + self.growing.search_with_metric(query, top_k, metric)?, + self.growing_base_id, + ); for seg in &self.building { - for mut r in seg.flat.search_with_metric(query, top_k, metric) { - r.id += seg.base_id; - all.push(r); - } + push_shifted( + &mut all, + seg.flat.search_with_metric(query, top_k, metric)?, + seg.base_id, + ); } - - all.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - all.truncate(top_k); - all + Ok(finish(all, top_k)) } /// Search with a pre-filter bitmap (byte-array format) and explicit metric override. + /// + /// Bitmap bytes that do not decode fail with + /// [`VectorError::InvalidFilterBitmap`]. pub fn search_with_bitmap_bytes_and_metric( &self, query: &[f32], @@ -301,103 +274,96 @@ impl VectorCollection { ef: usize, bitmap: &[u8], metric: DistanceMetric, - ) -> Vec { + ) -> Result, VectorError> { + check_dim(self.dim, query.len())?; let mut all: Vec = Vec::new(); - let growing_results = self.growing.search_filtered_offset_with_metric( - query, - top_k, - bitmap, + if let Some(ivf) = &self.ivf { + let filter = decode_filter_bitmap(bitmap)?; + all.extend(ivf.search_with(query, top_k, metric, Some(&filter))?); + } + push_shifted( + &mut all, + self.growing.search_filtered_offset_with_metric( + query, + top_k, + bitmap, + self.growing_base_id, + metric, + )?, self.growing_base_id, - metric, ); - for mut r in growing_results { - r.id += self.growing_base_id; - all.push(r); - } for seg in &self.sealed { - let results = + let mut results = seg.index - .search_with_bitmap_bytes_offset(query, top_k, ef, bitmap, seg.base_id); - for mut r in results { - // Rerank with the requested metric using the stored FP32 vector. - if let Some(v) = seg.index.get_vector(r.id.wrapping_sub(seg.base_id)) { - r.distance = crate::distance::distance(query, v, metric); + .search_with_bitmap_bytes_offset(query, top_k, ef, bitmap, seg.base_id)?; + // Rerank with the requested metric using the stored FP32 vector. + for r in &mut results { + if let Some(v) = seg.index.get_vector(r.id) { + r.distance = distance(query, v, metric); } - r.id += seg.base_id; - all.push(r); } + push_shifted(&mut all, results, seg.base_id); } for seg in &self.building { - let results = seg.flat.search_filtered_offset_with_metric( - query, - top_k, - bitmap, + push_shifted( + &mut all, + seg.flat.search_filtered_offset_with_metric( + query, + top_k, + bitmap, + seg.base_id, + metric, + )?, seg.base_id, - metric, ); - for mut r in results { - r.id += seg.base_id; - all.push(r); - } } - - all.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - all.truncate(top_k); - all + Ok(finish(all, top_k)) } /// Search with a pre-filter bitmap (byte-array format). + /// + /// Bitmap bytes that do not decode fail with + /// [`VectorError::InvalidFilterBitmap`]. pub fn search_with_bitmap_bytes( &self, query: &[f32], top_k: usize, ef: usize, bitmap: &[u8], - ) -> Vec { + ) -> Result, VectorError> { + check_dim(self.dim, query.len())?; let mut all: Vec = Vec::new(); - let growing_results = - self.growing - .search_filtered_offset(query, top_k, bitmap, self.growing_base_id); - for mut r in growing_results { - r.id += self.growing_base_id; - all.push(r); + if let Some(ivf) = &self.ivf { + let filter = decode_filter_bitmap(bitmap)?; + all.extend(ivf.search_with(query, top_k, self.params.metric, Some(&filter))?); } - + push_shifted( + &mut all, + self.growing + .search_filtered_offset(query, top_k, bitmap, self.growing_base_id)?, + self.growing_base_id, + ); for seg in &self.sealed { - let results = + push_shifted( + &mut all, seg.index - .search_with_bitmap_bytes_offset(query, top_k, ef, bitmap, seg.base_id); - for mut r in results { - r.id += seg.base_id; - all.push(r); - } + .search_with_bitmap_bytes_offset(query, top_k, ef, bitmap, seg.base_id)?, + seg.base_id, + ); } - for seg in &self.building { - let results = seg - .flat - .search_filtered_offset(query, top_k, bitmap, seg.base_id); - for mut r in results { - r.id += seg.base_id; - all.push(r); - } + push_shifted( + &mut all, + seg.flat + .search_filtered_offset(query, top_k, bitmap, seg.base_id)?, + seg.base_id, + ); } - - all.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - all.truncate(top_k); - all + Ok(finish(all, top_k)) } /// Search with a structured payload predicate. @@ -418,23 +384,24 @@ impl VectorCollection { top_k: usize, ef: usize, predicate: &FilterPredicate, - ) -> (Vec, bool) { + ) -> Result<(Vec, bool), VectorError> { match self.payload.pre_filter(predicate) { Some(bm) => { // Serialize the bitmap to the byte format expected by // `search_with_bitmap_bytes`. let mut bm_bytes = Vec::new(); if bm.serialize_into(&mut bm_bytes).is_ok() { - let results = self.search_with_bitmap_bytes(query, top_k, ef, &bm_bytes); - (results, true) + let results = self.search_with_bitmap_bytes(query, top_k, ef, &bm_bytes)?; + Ok((results, true)) } else { - // Serialization failure: fall back to unfiltered search. - (self.search(query, top_k, ef), false) + // Serialization failure: unfiltered search, and the + // caller applies the predicate as a post-filter. + Ok((self.search(query, top_k, ef)?, false)) } } None => { // Un-indexed field present: full scan, caller must post-filter. - (self.search(query, top_k, ef), false) + Ok((self.search(query, top_k, ef)?, false)) } } } @@ -462,10 +429,10 @@ mod tests { fn insert_and_search() { let mut coll = make_collection(); for i in 0..100u32 { - coll.insert(vec![i as f32, 0.0, 0.0]); + coll.insert(vec![i as f32, 0.0, 0.0]).unwrap(); } assert_eq!(coll.len(), 100); - let results = coll.search(&[50.0, 0.0, 0.0], 3, 64); + let results = coll.search(&[50.0, 0.0, 0.0], 3, 64).unwrap(); assert_eq!(results.len(), 3); assert_eq!(results[0].id, 50); } @@ -474,7 +441,7 @@ mod tests { fn seal_moves_to_building() { let mut coll = VectorCollection::new(2, HnswParams::default()); for i in 0..DEFAULT_SEAL_THRESHOLD { - coll.insert(vec![i as f32, 0.0]); + coll.insert(vec![i as f32, 0.0]).unwrap(); } assert!(coll.needs_seal()); @@ -483,7 +450,7 @@ mod tests { assert_eq!(coll.building.len(), 1); assert_eq!(coll.growing.len(), 0); - let results = coll.search(&[100.0, 0.0], 1, 64); + let results = coll.search(&[100.0, 0.0], 1, 64).unwrap(); assert!(!results.is_empty()); } @@ -491,7 +458,7 @@ mod tests { fn complete_build_promotes_to_sealed() { let mut coll = VectorCollection::new(2, HnswParams::default()); for i in 0..100 { - coll.insert(vec![i as f32, 0.0]); + coll.insert(vec![i as f32, 0.0]).unwrap(); } let req = coll.seal("test").unwrap(); @@ -504,7 +471,7 @@ mod tests { assert_eq!(coll.building.len(), 0); assert_eq!(coll.sealed.len(), 1); - let results = coll.search(&[50.0, 0.0], 3, 64); + let results = coll.search(&[50.0, 0.0], 3, 64).unwrap(); assert!(!results.is_empty()); } @@ -519,7 +486,7 @@ mod tests { ); for i in 0..100 { - coll.insert(vec![i as f32, 0.0]); + coll.insert(vec![i as f32, 0.0]).unwrap(); } let req = coll.seal("test").unwrap(); let mut idx = HnswIndex::new(2, req.params); @@ -529,10 +496,10 @@ mod tests { coll.complete_build(req.segment_id, idx, test_memory()); for i in 100..200 { - coll.insert(vec![i as f32, 0.0]); + coll.insert(vec![i as f32, 0.0]).unwrap(); } - let results = coll.search(&[150.0, 0.0], 3, 64); + let results = coll.search(&[150.0, 0.0], 3, 64).unwrap(); assert_eq!(results.len(), 3); assert_eq!(results[0].id, 150); } @@ -541,12 +508,12 @@ mod tests { fn delete_across_segments() { let mut coll = VectorCollection::new(2, HnswParams::default()); for i in 0..10 { - coll.insert(vec![i as f32, 0.0]); + coll.insert(vec![i as f32, 0.0]).unwrap(); } assert!(coll.delete(5)); assert_eq!(coll.live_count(), 9); - let results = coll.search(&[5.0, 0.0], 10, 64); + let results = coll.search(&[5.0, 0.0], 10, 64).unwrap(); assert!(results.iter().all(|r| r.id != 5)); } @@ -561,7 +528,7 @@ mod tests { }, ); for i in 0..n { - coll.insert(vec![i as f32, 0.0]); + coll.insert(vec![i as f32, 0.0]).unwrap(); } let req = coll.seal("seg").unwrap(); let mut idx = HnswIndex::new(req.dim, req.params); @@ -583,7 +550,7 @@ mod tests { .filter_map(|i| sealed.index.get_vector(i as u32).map(|v| v.to_vec())) .collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = Sq8Codec::calibrate(&refs, dim); + let codec = Sq8Codec::calibrate(&refs, dim).unwrap(); let sq8_data: Vec = vecs.iter().flat_map(|v| codec.quantize(v)).collect(); sealed.sq8 = Some((codec, sq8_data)); } @@ -593,7 +560,7 @@ mod tests { let mut coll = make_sealed_collection(200); attach_sq8(&mut coll); - let results = coll.search(&[100.0, 0.0], 5, 64); + let results = coll.search(&[100.0, 0.0], 5, 64).unwrap(); assert!(!results.is_empty(), "expected non-empty results"); assert_eq!( results[0].id, 100, @@ -612,8 +579,8 @@ mod tests { let query = [250.0f32, 0.0]; let top_k = 5; - let plain_results = coll_plain.search(&query, top_k, 64); - let sq8_results = coll_sq8.search(&query, top_k, 64); + let plain_results = coll_plain.search(&query, top_k, 64).unwrap(); + let sq8_results = coll_sq8.search(&query, top_k, 64).unwrap(); let plain_ids: std::collections::HashSet = plain_results.iter().map(|r| r.id).collect(); @@ -626,11 +593,11 @@ mod tests { ); } - #[test] - fn codec_dispatch_bbq_search_returns_results_and_stats_report_bbq() { - let dim = 4; + /// A collection whose first 50 vectors (`[i, 0, 0, 0]`) sit in one + /// sealed segment, with an empty growing segment. + fn sealed_dim4_collection() -> VectorCollection { let mut coll = VectorCollection::new( - dim, + 4, HnswParams { metric: DistanceMetric::L2, m: 8, @@ -638,14 +605,24 @@ mod tests { ..HnswParams::default() }, ); - - // Insert 50 vectors: vector i = [i as f32, 0, 0, 0]. for i in 0u32..50 { - coll.insert(vec![i as f32, 0.0, 0.0, 0.0]); + coll.insert(vec![i as f32, 0.0, 0.0, 0.0]).unwrap(); + } + let req = coll.seal("codec").unwrap(); + let mut idx = HnswIndex::new(req.dim, req.params); + for v in &req.vectors { + idx.insert(v.clone()).unwrap(); } + coll.complete_build(req.segment_id, idx, test_memory()); + coll + } - // Build the collection-level BBQ dispatch index over current vectors. - let dispatch = coll.build_codec_dispatch("bbq"); + #[test] + fn codec_dispatch_bbq_search_returns_results_and_stats_report_bbq() { + let mut coll = sealed_dim4_collection(); + + // Build the collection-level BBQ dispatch index over the sealed vectors. + let dispatch = coll.build_codec_dispatch("bbq").unwrap(); assert!( dispatch.is_some(), "build_codec_dispatch(bbq) should return Some" @@ -653,7 +630,7 @@ mod tests { // Query near id=25. let query = [25.0f32, 0.0, 0.0, 0.0]; - let results = coll.search(&query, 5, 32); + let results = coll.search(&query, 5, 32).unwrap(); assert!( !results.is_empty(), "BBQ codec-dispatch search should return results" @@ -670,22 +647,10 @@ mod tests { #[test] fn codec_dispatch_rabitq_search_non_empty() { - let dim = 4; - let mut coll = VectorCollection::new( - dim, - HnswParams { - metric: DistanceMetric::L2, - m: 8, - ef_construction: 50, - ..HnswParams::default() - }, - ); - for i in 0u32..50 { - coll.insert(vec![i as f32, 0.0, 0.0, 0.0]); - } - coll.build_codec_dispatch("rabitq").unwrap(); + let mut coll = sealed_dim4_collection(); + coll.build_codec_dispatch("rabitq").unwrap().unwrap(); - let results = coll.search(&[10.0, 0.0, 0.0, 0.0], 3, 32); + let results = coll.search(&[10.0, 0.0, 0.0, 0.0], 3, 32).unwrap(); assert!( !results.is_empty(), "RaBitQ dispatch search should return results" @@ -698,6 +663,71 @@ mod tests { ); } + /// The codec index covers the sealed segments under their global ids; + /// the growing segment is read beside it. A growing vector is found under + /// its own id, once. + #[test] + fn codec_dispatch_keeps_global_ids_and_reads_growing_once() { + let mut coll = sealed_dim4_collection(); + coll.build_codec_dispatch("bbq").unwrap().unwrap(); + let far = coll.insert(vec![1000.0, 0.0, 0.0, 0.0]).unwrap(); + assert_eq!(far, 50, "the first growing vector takes the next global id"); + + let results = coll.search(&[1000.0, 0.0, 0.0, 0.0], 3, 32).unwrap(); + assert_eq!(results[0].id, far); + assert_eq!( + results.iter().filter(|r| r.id == far).count(), + 1, + "{results:?}" + ); + assert!(results.iter().all(|r| r.id <= far), "{results:?}"); + } + + /// Every collection search entry point refuses a query of the wrong + /// dimension, on each segment kind and on the codec-dispatch path. + #[test] + fn wrong_dimension_query_is_a_typed_error() { + use crate::error::VectorError; + let mut coll = sealed_dim4_collection(); + coll.insert(vec![1.0, 0.0, 0.0, 0.0]).unwrap(); + let bytes = { + let bm: roaring::RoaringBitmap = (0..51u32).collect(); + let mut out = Vec::new(); + bm.serialize_into(&mut out).unwrap(); + out + }; + let short = [1.0_f32, 0.0]; + let check = |coll: &VectorCollection| { + for result in [ + coll.search(&short, 3, 32), + coll.search_with_metric(&short, 3, 32, DistanceMetric::Cosine), + coll.search_with_bitmap_bytes(&short, 3, 32, &bytes), + coll.search_with_bitmap_bytes_and_metric(&short, 3, 32, &bytes, DistanceMetric::L2), + ] { + assert!( + matches!( + result, + Err(VectorError::DimensionMismatch { + expected: 4, + got: 2 + }) + ), + "{result:?}" + ); + } + }; + check(&coll); + coll.build_codec_dispatch("rabitq").unwrap().unwrap(); + check(&coll); + assert!(matches!( + coll.insert(vec![1.0; 3]), + Err(VectorError::DimensionMismatch { + expected: 4, + got: 3 + }) + )); + } + #[test] fn sq8_search_does_not_scan_all_vectors() { // This test validates correctness of the SQ8 search path for a large @@ -708,7 +738,7 @@ mod tests { let mut coll = make_sealed_collection(2000); attach_sq8(&mut coll); - let results = coll.search(&[1000.0, 0.0], 5, 64); + let results = coll.search(&[1000.0, 0.0], 5, 64).unwrap(); assert!(!results.is_empty(), "expected non-empty results"); assert_eq!( results[0].id, 1000, diff --git a/nodedb-vector/src/collection/segment.rs b/nodedb-vector/src/collection/segment.rs index 048d0b13d..c30feb4eb 100644 --- a/nodedb-vector/src/collection/segment.rs +++ b/nodedb-vector/src/collection/segment.rs @@ -13,10 +13,26 @@ use crate::quantize::sq8::Sq8Codec; /// 64K vectors × 768 dims × 4 bytes = ~192 MiB per segment. pub const DEFAULT_SEAL_THRESHOLD: usize = 65_536; -/// Request to build an HNSW index from sealed vectors (sent to builder thread). +/// What a finished build replaces. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum BuildKind { + /// Promote building segment `segment_id` to sealed. + Seal, + /// Replace the sealed segment at `base_id`, which held `len` nodes when + /// its vectors were read. A segment that no longer holds `len` nodes + /// (compaction renumbered it) refuses the result. + Rebuild { base_id: u32, len: usize }, +} + +/// Request to build an HNSW index (sent to the builder thread). +/// +/// `vectors` holds one vector per local node id, soft-deleted nodes +/// included, so the built graph keeps every id. The owning core applies +/// the tombstones when it installs the result. pub struct BuildRequest { pub key: String, pub segment_id: u32, + pub kind: BuildKind, pub vectors: Vec>, pub dim: usize, pub params: HnswParams, @@ -26,7 +42,10 @@ pub struct BuildRequest { pub struct BuildComplete { pub key: String, pub segment_id: u32, - pub index: HnswIndex, + pub kind: BuildKind, + /// The built index, or the error that stopped the build. A failed build + /// leaves the segment as it was. + pub result: Result, } /// A sealed segment whose HNSW index is being built in background. diff --git a/nodedb-vector/src/collection/stats.rs b/nodedb-vector/src/collection/stats.rs index 226b51c9e..5f88d7693 100644 --- a/nodedb-vector/src/collection/stats.rs +++ b/nodedb-vector/src/collection/stats.rs @@ -13,11 +13,13 @@ impl VectorCollection { let sealed_vectors: usize = self.sealed.iter().map(|s| s.index.len()).sum(); let building_vectors: usize = self.building.iter().map(|s| s.flat.len()).sum(); - let tombstone_count: usize = self - .sealed - .iter() - .map(|s| s.index.tombstone_count()) - .sum::() + let ivf_vectors = self.ivf.as_ref().map_or(0, |ivf| ivf.len()); + let tombstone_count: usize = self.ivf.as_ref().map_or(0, |ivf| ivf.tombstone_count()) + + self + .sealed + .iter() + .map(|s| s.index.tombstone_count()) + .sum::() + self.growing.tombstone_count() + self .building @@ -25,7 +27,7 @@ impl VectorCollection { .map(|s| s.flat.tombstone_count()) .sum::(); - let total = growing_vectors + sealed_vectors + building_vectors; + let total = growing_vectors + sealed_vectors + building_vectors + ivf_vectors; let tombstone_ratio = if total > 0 { tombstone_count as f64 / total as f64 } else { @@ -38,7 +40,7 @@ impl VectorCollection { "bbq" => nodedb_types::VectorIndexQuantization::Bbq, _ => nodedb_types::VectorIndexQuantization::None, } - } else if self.sealed.iter().any(|s| s.pq.is_some()) { + } else if self.ivf.is_some() || self.sealed.iter().any(|s| s.pq.is_some()) { nodedb_types::VectorIndexQuantization::Pq } else if self.sealed.iter().any(|s| s.sq8.is_some()) { nodedb_types::VectorIndexQuantization::Sq8 @@ -64,7 +66,8 @@ impl VectorCollection { .sum(); let growing_mem = growing_vectors * self.dim * std::mem::size_of::(); let building_mem = building_vectors * self.dim * std::mem::size_of::(); - let memory_bytes = hnsw_mem + sq8_mem + growing_mem + building_mem; + let ivf_mem = self.ivf.as_ref().map_or(0, |ivf| ivf.memory_bytes()); + let memory_bytes = hnsw_mem + sq8_mem + growing_mem + building_mem + ivf_mem; let disk_bytes: usize = self .sealed @@ -99,6 +102,19 @@ impl VectorCollection { // is always `None` here; callers overwrite it after calling // `stats()` when a dedicated arena handle is available. arena_bytes: None, + ivf: self.is_ivf().then(|| nodedb_types::VectorIvfStats { + training_threshold: self.ivf_training_threshold(), + trained: self.ivf.is_some(), + trained_on: self.ivf.as_ref().map_or(0, |ivf| ivf.trained_on()), + trained_at_ms: self.ivf.as_ref().map_or(0, |ivf| ivf.trained_at_ms()), + indexed_vectors: ivf_vectors, + cells: self.ivf.as_ref().map_or(0, |ivf| ivf.n_cells()), + nprobe: self.index_config.ivf_nprobe, + }), + // The owning core fills this from its build queue. + builds_queued: 0, + builds_completed: self.builds_completed, + builds_failed: self.builds_failed, } } } diff --git a/nodedb-vector/src/delta/compaction.rs b/nodedb-vector/src/delta/compaction.rs index 7bcd01973..5676e8281 100644 --- a/nodedb-vector/src/delta/compaction.rs +++ b/nodedb-vector/src/delta/compaction.rs @@ -199,7 +199,7 @@ mod tests { let mut delta = DeltaIndex::new(3, 32); for i in 10u32..15 { let v = vec![i as f32, 1.0, 0.0]; - delta.insert(i, v); + delta.insert(i, v).unwrap(); } assert_eq!(delta.fresh_len(), 5); diff --git a/nodedb-vector/src/delta/index.rs b/nodedb-vector/src/delta/index.rs index 2b0832dee..c37a7c0ce 100644 --- a/nodedb-vector/src/delta/index.rs +++ b/nodedb-vector/src/delta/index.rs @@ -9,6 +9,7 @@ use std::collections::HashSet; use crate::distance::distance; +use crate::error::{VectorError, check_dim}; use nodedb_types::vector_distance::DistanceMetric; /// Secondary in-memory index that absorbs fresh inserts before they are @@ -38,9 +39,12 @@ impl DeltaIndex { } /// Stage a fresh insert. Does not deduplicate — callers must ensure IDs - /// are unique across the delta and the main HNSW. - pub fn insert(&mut self, id: u32, vector: Vec) { + /// are unique across the delta and the main HNSW. A vector without the + /// index dimension fails with [`VectorError::DimensionMismatch`]. + pub fn insert(&mut self, id: u32, vector: Vec) -> Result<(), VectorError> { + check_dim(self.dim, vector.len())?; self.fresh.push((id, vector)); + Ok(()) } /// Mark `id` as tombstoned. It will be excluded from `search` results @@ -60,10 +64,17 @@ impl DeltaIndex { } /// Brute-force scan over fresh vectors (excluding tombstones), returning - /// the top-`k` results sorted ascending by distance. - pub fn search(&self, query: &[f32], k: usize, metric: DistanceMetric) -> Vec<(u32, f32)> { + /// the top-`k` results sorted ascending by distance. A query without + /// the index dimension fails with [`VectorError::DimensionMismatch`]. + pub fn search( + &self, + query: &[f32], + k: usize, + metric: DistanceMetric, + ) -> Result, VectorError> { + check_dim(self.dim, query.len())?; if k == 0 { - return Vec::new(); + return Ok(Vec::new()); } let mut scored: Vec<(u32, f32)> = self @@ -83,7 +94,7 @@ impl DeltaIndex { scored.sort_unstable_by(|a, b| a.1.partial_cmp(&b.1).unwrap_or(std::cmp::Ordering::Equal)); - scored + Ok(scored) } /// Drain all staged fresh vectors for patching into the main HNSW. @@ -113,7 +124,7 @@ mod tests { let mut d = DeltaIndex::new(3, 16); for i in 0u32..10 { let v = vec![i as f32, 0.0, 0.0]; - d.insert(i, v); + d.insert(i, v).unwrap(); } d } @@ -122,7 +133,7 @@ mod tests { fn top_k_returns_nearest() { let d = make_delta(); let query = [0.0f32, 0.0, 0.0]; - let results = d.search(&query, 3, DistanceMetric::L2); + let results = d.search(&query, 3, DistanceMetric::L2).unwrap(); assert_eq!(results.len(), 3); // Nearest to [0,0,0] with L2^2 are ids 0,1,2 assert_eq!(results[0].0, 0); @@ -135,7 +146,7 @@ mod tests { let mut d = make_delta(); d.tombstone(0); let query = [0.0f32, 0.0, 0.0]; - let results = d.search(&query, 3, DistanceMetric::L2); + let results = d.search(&query, 3, DistanceMetric::L2).unwrap(); assert!(results.iter().all(|(id, _)| *id != 0)); } @@ -143,10 +154,10 @@ mod tests { fn is_full_triggers_at_threshold() { let mut d = DeltaIndex::new(3, 3); assert!(!d.is_full()); - d.insert(0, vec![0.0, 0.0, 0.0]); - d.insert(1, vec![1.0, 0.0, 0.0]); + d.insert(0, vec![0.0, 0.0, 0.0]).unwrap(); + d.insert(1, vec![1.0, 0.0, 0.0]).unwrap(); assert!(!d.is_full()); - d.insert(2, vec![2.0, 0.0, 0.0]); + d.insert(2, vec![2.0, 0.0, 0.0]).unwrap(); assert!(d.is_full()); } @@ -168,4 +179,23 @@ mod tests { assert!(!d.is_tombstoned(3)); assert!(!d.is_tombstoned(7)); } + + #[test] + fn wrong_dimension_is_a_typed_error() { + let mut d = DeltaIndex::new(3, 8); + assert!(matches!( + d.insert(0, vec![0.0; 2]), + Err(VectorError::DimensionMismatch { + expected: 3, + got: 2 + }) + )); + assert!(matches!( + d.search(&[0.0; 4], 1, DistanceMetric::L2), + Err(VectorError::DimensionMismatch { + expected: 3, + got: 4 + }) + )); + } } diff --git a/nodedb-vector/src/dtype/cast.rs b/nodedb-vector/src/dtype/cast.rs index 22455052f..b6b76797c 100644 --- a/nodedb-vector/src/dtype/cast.rs +++ b/nodedb-vector/src/dtype/cast.rs @@ -26,6 +26,9 @@ pub enum DtypeError { expected: usize, actual: usize, }, + /// A storage dtype this build has no decoder for. + #[error("no f32 decoder for vector storage dtype {dtype}")] + Unsupported { dtype: VectorStorageDtype }, } /// Verify that `bytes.len() == dtype.bytes_for_dim(dim)`. @@ -100,9 +103,9 @@ pub fn cast_to_f32( .collect(); Ok(out) } - // `VectorStorageDtype` is #[non_exhaustive]; this arm is required by - // the compiler but unreachable with any currently-defined variant. - _ => unreachable!("unrecognised VectorStorageDtype variant in cast_to_f32"), + // `VectorStorageDtype` is #[non_exhaustive]: a dtype added upstream + // without a decoder here fails the read instead of the process. + _ => Err(DtypeError::Unsupported { dtype }), } } @@ -241,55 +244,55 @@ mod tests { #[test] fn bad_byte_len_f32_mismatch() { let err = cast_to_f32(&[0u8; 7], VectorStorageDtype::F32, 2).unwrap_err(); - match err { - DtypeError::BadByteLen { - dtype, - dim, - expected, - actual, - } => { - assert_eq!(dtype, VectorStorageDtype::F32); - assert_eq!(dim, 2); - assert_eq!(expected, 8); - assert_eq!(actual, 7); - } - } + let DtypeError::BadByteLen { + dtype, + dim, + expected, + actual, + } = err + else { + panic!("expected BadByteLen, got {err:?}"); + }; + assert_eq!(dtype, VectorStorageDtype::F32); + assert_eq!(dim, 2); + assert_eq!(expected, 8); + assert_eq!(actual, 7); } #[test] fn bad_byte_len_f16_odd_byte_count() { let err = cast_to_f32(&[0u8; 3], VectorStorageDtype::F16, 2).unwrap_err(); - match err { - DtypeError::BadByteLen { - dtype, - dim, - expected, - actual, - } => { - assert_eq!(dtype, VectorStorageDtype::F16); - assert_eq!(dim, 2); - assert_eq!(expected, 4); - assert_eq!(actual, 3); - } - } + let DtypeError::BadByteLen { + dtype, + dim, + expected, + actual, + } = err + else { + panic!("expected BadByteLen, got {err:?}"); + }; + assert_eq!(dtype, VectorStorageDtype::F16); + assert_eq!(dim, 2); + assert_eq!(expected, 4); + assert_eq!(actual, 3); } #[test] fn bad_byte_len_bf16_mismatch() { let err = cast_to_f32(&[0u8; 5], VectorStorageDtype::BF16, 3).unwrap_err(); - match err { - DtypeError::BadByteLen { - dtype, - dim, - expected, - actual, - } => { - assert_eq!(dtype, VectorStorageDtype::BF16); - assert_eq!(dim, 3); - assert_eq!(expected, 6); - assert_eq!(actual, 5); - } - } + let DtypeError::BadByteLen { + dtype, + dim, + expected, + actual, + } = err + else { + panic!("expected BadByteLen, got {err:?}"); + }; + assert_eq!(dtype, VectorStorageDtype::BF16); + assert_eq!(dim, 3); + assert_eq!(expected, 6); + assert_eq!(actual, 5); } // ── validate_byte_len independently ────────────────────────────────────── @@ -304,13 +307,13 @@ mod tests { fn validate_byte_len_off_by_one_fails() { let bytes = [0u8; 11]; // should be 12 let err = validate_byte_len(&bytes, VectorStorageDtype::F32, 3).unwrap_err(); - match err { - DtypeError::BadByteLen { - expected, actual, .. - } => { - assert_eq!(expected, 12); - assert_eq!(actual, 11); - } - } + let DtypeError::BadByteLen { + expected, actual, .. + } = err + else { + panic!("expected BadByteLen, got {err:?}"); + }; + assert_eq!(expected, 12); + assert_eq!(actual, 11); } } diff --git a/nodedb-vector/src/error.rs b/nodedb-vector/src/error.rs index 41c233bfe..2bc53d2b9 100644 --- a/nodedb-vector/src/error.rs +++ b/nodedb-vector/src/error.rs @@ -10,8 +10,24 @@ use nodedb_mem::MemError; pub enum VectorError { #[error("memory budget exhausted: {0}")] BudgetExhausted(#[from] MemError), + /// An input vector — a search query or an inserted vector — has a + /// different dimension from the index. The caller's input is wrong; the + /// index is intact. #[error("vector dimension mismatch: expected {expected}, got {got}")] DimensionMismatch { expected: usize, got: usize }, + /// Stored data — a PQ code, a segment backing, a materialized node + /// vector — disagrees with the index dimension. The stored data is + /// corrupt or belongs to another index. + #[error("stored vector data has dimension {got}, index expects {expected}")] + StoredDimensionMismatch { expected: usize, got: usize }, + /// A serialized pre-filter bitmap does not decode. Searching without it + /// would return rows the filter excludes, so the search fails instead. + #[error("vector search filter bitmap does not decode: {detail}")] + InvalidFilterBitmap { detail: String }, + /// An index build or training call received input it cannot use: an + /// empty training set, a zero dimension, or parameters that do not fit. + #[error("invalid vector index input: {detail}")] + InvalidInput { detail: String }, /// A node's vector could not be materialized: the node is out of range, or /// its local storage is empty and no segment backing supplies the data. /// @@ -58,3 +74,13 @@ pub enum VectorError { #[error("vector segment I/O error: {0}")] SegmentIo(#[from] std::io::Error), } + +/// Check that an input vector of length `got` fits an index of dimension +/// `expected`. +pub fn check_dim(expected: usize, got: usize) -> Result<(), VectorError> { + if expected == got { + Ok(()) + } else { + Err(VectorError::DimensionMismatch { expected, got }) + } +} diff --git a/nodedb-vector/src/flat.rs b/nodedb-vector/src/flat.rs index 5ed0880f9..16dbfc265 100644 --- a/nodedb-vector/src/flat.rs +++ b/nodedb-vector/src/flat.rs @@ -12,7 +12,9 @@ use roaring::RoaringBitmap; use crate::distance::{DistanceMetric, distance}; +use crate::error::{VectorError, check_dim}; use crate::hnsw::SearchResult; +use crate::hnsw::search::decode_filter_bitmap; /// Default threshold below which collections use flat index instead of HNSW. pub const DEFAULT_FLAT_INDEX_THRESHOLD: usize = 10_000; @@ -41,20 +43,16 @@ impl FlatIndex { } } - /// Insert a vector. Returns the assigned vector ID. - pub fn insert(&mut self, vector: Vec) -> u32 { - assert_eq!( - vector.len(), - self.dim, - "dimension mismatch: expected {}, got {}", - self.dim, - vector.len() - ); + /// Insert a vector. Returns the assigned vector ID, or + /// [`VectorError::DimensionMismatch`] when `vector` does not have the + /// index dimension. + pub fn insert(&mut self, vector: Vec) -> Result { + check_dim(self.dim, vector.len())?; let id = self.len() as u32; self.data.extend_from_slice(&vector); self.deleted.push(false); self.live_count += 1; - id + Ok(id) } /// Soft-delete a vector by ID. @@ -92,83 +90,22 @@ impl FlatIndex { query: &[f32], top_k: usize, metric: DistanceMetric, - ) -> Vec { - assert_eq!(query.len(), self.dim); - let n = self.len(); - if n == 0 || top_k == 0 { - return Vec::new(); - } - - let mut candidates: Vec = Vec::with_capacity(n.min(top_k * 2)); - for i in 0..n { - if self.deleted[i] { - continue; - } - let start = i * self.dim; - let vec_slice = &self.data[start..start + self.dim]; - let dist = distance(query, vec_slice, metric); - candidates.push(SearchResult { - id: i as u32, - distance: dist, - }); - } - - if candidates.len() > top_k { - candidates.select_nth_unstable_by(top_k, |a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - candidates.truncate(top_k); - } - candidates.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - candidates + ) -> Result, VectorError> { + self.scan(query, top_k, metric, None) } /// Brute-force k-NN search. Exact results — no approximation. - pub fn search(&self, query: &[f32], top_k: usize) -> Vec { - assert_eq!(query.len(), self.dim); - let n = self.len(); - if n == 0 || top_k == 0 { - return Vec::new(); - } - - let mut candidates: Vec = Vec::with_capacity(n.min(top_k * 2)); - for i in 0..n { - if self.deleted[i] { - continue; - } - let start = i * self.dim; - let vec_slice = &self.data[start..start + self.dim]; - let dist = distance(query, vec_slice, self.metric); - candidates.push(SearchResult { - id: i as u32, - distance: dist, - }); - } - - if candidates.len() > top_k { - candidates.select_nth_unstable_by(top_k, |a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - candidates.truncate(top_k); - } - candidates.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - candidates + pub fn search(&self, query: &[f32], top_k: usize) -> Result, VectorError> { + self.scan(query, top_k, self.metric, None) } /// Search with a pre-filter bitmap (byte-array format). - pub fn search_filtered(&self, query: &[f32], top_k: usize, bitmap: &[u8]) -> Vec { + pub fn search_filtered( + &self, + query: &[f32], + top_k: usize, + bitmap: &[u8], + ) -> Result, VectorError> { self.search_filtered_offset(query, top_k, bitmap, 0) } @@ -180,89 +117,57 @@ impl FlatIndex { bitmap: &[u8], id_offset: u32, metric: DistanceMetric, - ) -> Vec { - assert_eq!(query.len(), self.dim); - let n = self.len(); - if n == 0 || top_k == 0 { - return Vec::new(); - } - - let parsed = RoaringBitmap::deserialize_from(bitmap).ok(); - - let mut candidates: Vec = Vec::with_capacity(top_k * 2); - for i in 0..n { - if self.deleted[i] { - continue; - } - if let Some(ref bm) = parsed { - let global = (i as u32).saturating_add(id_offset); - if !bm.contains(global) { - continue; - } - } - let start = i * self.dim; - let vec_slice = &self.data[start..start + self.dim]; - let dist = distance(query, vec_slice, metric); - candidates.push(SearchResult { - id: i as u32, - distance: dist, - }); - } - - if candidates.len() > top_k { - candidates.select_nth_unstable_by(top_k, |a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - candidates.truncate(top_k); - } - candidates.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - candidates + ) -> Result, VectorError> { + let filter = decode_filter_bitmap(bitmap)?; + self.scan(query, top_k, metric, Some((&filter, id_offset))) } /// Search with a pre-filter bitmap applying a global id offset. /// /// `bitmap` is a serialized `RoaringBitmap` (matching the HNSW filter /// format). Bit `i + id_offset` tests local id `i`. Used by multi-segment - /// collections where the bitmap holds GLOBAL vector ids. If the bytes - /// fail to deserialize, the search degrades to unfiltered. + /// collections where the bitmap holds GLOBAL vector ids. Bytes that do + /// not decode fail with [`VectorError::InvalidFilterBitmap`]. pub fn search_filtered_offset( &self, query: &[f32], top_k: usize, bitmap: &[u8], id_offset: u32, - ) -> Vec { - assert_eq!(query.len(), self.dim); + ) -> Result, VectorError> { + self.search_filtered_offset_with_metric(query, top_k, bitmap, id_offset, self.metric) + } + + /// Exact scan of every live vector under `metric`, restricted to ids + /// whose `local + offset` is in the filter when one is given. + fn scan( + &self, + query: &[f32], + top_k: usize, + metric: DistanceMetric, + filter: Option<(&RoaringBitmap, u32)>, + ) -> Result, VectorError> { + check_dim(self.dim, query.len())?; let n = self.len(); if n == 0 || top_k == 0 { - return Vec::new(); + return Ok(Vec::new()); } - let parsed = RoaringBitmap::deserialize_from(bitmap).ok(); - - let mut candidates: Vec = Vec::with_capacity(top_k * 2); + let mut candidates: Vec = Vec::with_capacity(n.min(top_k * 2)); for i in 0..n { if self.deleted[i] { continue; } - if let Some(ref bm) = parsed { - let global = (i as u32).saturating_add(id_offset); - if !bm.contains(global) { - continue; - } + if let Some((bitmap, id_offset)) = filter + && !bitmap.contains((i as u32).saturating_add(id_offset)) + { + continue; } let start = i * self.dim; let vec_slice = &self.data[start..start + self.dim]; - let dist = distance(query, vec_slice, self.metric); candidates.push(SearchResult { id: i as u32, - distance: dist, + distance: distance(query, vec_slice, metric), }); } @@ -279,13 +184,28 @@ impl FlatIndex { .partial_cmp(&b.distance) .unwrap_or(std::cmp::Ordering::Equal) }); - candidates + Ok(candidates) } pub fn len(&self) -> usize { self.deleted.len() } + /// Drop every vector at position `len` or later, as if it was never + /// inserted. A rollback uses it to withdraw the newest inserts. + pub fn truncate(&mut self, len: usize) { + if len >= self.deleted.len() { + return; + } + let dropped_live = self.deleted[len..] + .iter() + .filter(|deleted| !**deleted) + .count(); + self.live_count -= dropped_live; + self.deleted.truncate(len); + self.data.truncate(len * self.dim); + } + pub fn live_count(&self) -> usize { self.live_count } @@ -322,19 +242,13 @@ impl FlatIndex { } /// Insert a vector that is already tombstoned (for checkpoint restore). - pub fn insert_tombstoned(&mut self, vector: Vec) -> u32 { - assert_eq!( - vector.len(), - self.dim, - "dimension mismatch: expected {}, got {}", - self.dim, - vector.len() - ); + pub fn insert_tombstoned(&mut self, vector: Vec) -> Result { + check_dim(self.dim, vector.len())?; let id = self.len() as u32; self.data.extend_from_slice(&vector); self.deleted.push(true); // No live_count increment — it's dead on arrival. - id + Ok(id) } pub fn dim(&self) -> usize { @@ -358,12 +272,12 @@ mod tests { fn insert_and_search() { let mut idx = FlatIndex::new(3, DistanceMetric::L2); for i in 0..100u32 { - idx.insert(vec![i as f32, 0.0, 0.0]); + idx.insert(vec![i as f32, 0.0, 0.0]).unwrap(); } assert_eq!(idx.len(), 100); assert_eq!(idx.live_count(), 100); - let results = idx.search(&[50.0, 0.0, 0.0], 3); + let results = idx.search(&[50.0, 0.0, 0.0], 3).unwrap(); assert_eq!(results.len(), 3); assert_eq!(results[0].id, 50); assert!(results[0].distance < 0.01); @@ -372,14 +286,14 @@ mod tests { #[test] fn delete_excludes_from_search() { let mut idx = FlatIndex::new(2, DistanceMetric::L2); - idx.insert(vec![0.0, 0.0]); - idx.insert(vec![1.0, 0.0]); - idx.insert(vec![2.0, 0.0]); + idx.insert(vec![0.0, 0.0]).unwrap(); + idx.insert(vec![1.0, 0.0]).unwrap(); + idx.insert(vec![2.0, 0.0]).unwrap(); assert!(idx.delete(1)); assert_eq!(idx.live_count(), 2); - let results = idx.search(&[1.0, 0.0], 3); + let results = idx.search(&[1.0, 0.0], 3).unwrap(); assert_eq!(results.len(), 2); assert!(results.iter().all(|r| r.id != 1)); } @@ -387,11 +301,11 @@ mod tests { #[test] fn exact_results() { let mut idx = FlatIndex::new(2, DistanceMetric::Cosine); - idx.insert(vec![1.0, 0.0]); - idx.insert(vec![0.0, 1.0]); - idx.insert(vec![1.0, 1.0]); + idx.insert(vec![1.0, 0.0]).unwrap(); + idx.insert(vec![0.0, 1.0]).unwrap(); + idx.insert(vec![1.0, 1.0]).unwrap(); - let results = idx.search(&[1.0, 0.0], 1); + let results = idx.search(&[1.0, 0.0], 1).unwrap(); assert_eq!(results.len(), 1); assert_eq!(results[0].id, 0); } @@ -399,7 +313,7 @@ mod tests { #[test] fn empty_search() { let idx = FlatIndex::new(3, DistanceMetric::L2); - let results = idx.search(&[1.0, 0.0, 0.0], 5); + let results = idx.search(&[1.0, 0.0, 0.0], 5).unwrap(); assert!(results.is_empty()); } @@ -407,12 +321,67 @@ mod tests { fn filtered_search() { let mut idx = FlatIndex::new(2, DistanceMetric::L2); for i in 0..8u32 { - idx.insert(vec![i as f32, 0.0]); + idx.insert(vec![i as f32, 0.0]).unwrap(); } - let bitmap = vec![0b11001100u8]; - let results = idx.search_filtered(&[3.0, 0.0], 2, &bitmap); + let filter: RoaringBitmap = [2u32, 3, 6, 7].into_iter().collect(); + let mut bitmap = Vec::new(); + filter.serialize_into(&mut bitmap).unwrap(); + let results = idx.search_filtered(&[4.0, 0.0], 2, &bitmap).unwrap(); assert_eq!(results.len(), 2); assert_eq!(results[0].id, 3); - assert_eq!(results[1].id, 2); + assert!(results.iter().all(|r| filter.contains(r.id)), "{results:?}"); + } + + #[test] + fn wrong_dimension_is_a_typed_error() { + let mut idx = FlatIndex::new(2, DistanceMetric::L2); + assert!(matches!( + idx.insert(vec![1.0, 2.0, 3.0]), + Err(VectorError::DimensionMismatch { + expected: 2, + got: 3 + }) + )); + assert!(matches!( + idx.insert_tombstoned(vec![1.0]), + Err(VectorError::DimensionMismatch { + expected: 2, + got: 1 + }) + )); + idx.insert(vec![1.0, 0.0]).unwrap(); + let filter: RoaringBitmap = [0u32].into_iter().collect(); + let mut bitmap = Vec::new(); + filter.serialize_into(&mut bitmap).unwrap(); + let short = [1.0_f32]; + for result in [ + idx.search(&short, 1), + idx.search_with_metric(&short, 1, DistanceMetric::Cosine), + idx.search_filtered(&short, 1, &bitmap), + idx.search_filtered_offset(&short, 1, &bitmap, 0), + idx.search_filtered_offset_with_metric(&short, 1, &bitmap, 0, DistanceMetric::L2), + ] { + assert!( + matches!( + result, + Err(VectorError::DimensionMismatch { + expected: 2, + got: 1 + }) + ), + "{result:?}" + ); + } + } + + #[test] + fn undecodable_filter_bitmap_is_a_typed_error() { + let mut idx = FlatIndex::new(2, DistanceMetric::L2); + idx.insert(vec![1.0, 0.0]).unwrap(); + let result = idx.search_filtered(&[1.0, 0.0], 1, &[0b1100_1100]); + assert!( + matches!(result, Err(VectorError::InvalidFilterBitmap { .. })), + "{result:?}" + ); } } diff --git a/nodedb-vector/src/hnsw/build.rs b/nodedb-vector/src/hnsw/build.rs index 5c5862bc9..a93ee761a 100644 --- a/nodedb-vector/src/hnsw/build.rs +++ b/nodedb-vector/src/hnsw/build.rs @@ -227,7 +227,7 @@ mod tests { for target in 0..20u32 { let query = idx.get_vector(target).unwrap().to_vec(); - let results = idx.search(&query, 1, 32); + let results = idx.search(&query, 1, 32).unwrap(); assert_eq!(results[0].id, target, "node {target} not reachable"); } } @@ -247,7 +247,7 @@ mod tests { for target_old_id in (1..20u32).step_by(2) { let query = vec![target_old_id as f32, 0.0, 0.0]; - let results = idx.search(&query, 1, 32); + let results = idx.search(&query, 1, 32).unwrap(); assert!(!results.is_empty()); let found_vec = idx.get_vector(results[0].id).unwrap(); assert_eq!(found_vec[0], target_old_id as f32); diff --git a/nodedb-vector/src/hnsw/checkpoint.rs b/nodedb-vector/src/hnsw/checkpoint.rs index e90d2dd1f..ec6766eb3 100644 --- a/nodedb-vector/src/hnsw/checkpoint.rs +++ b/nodedb-vector/src/hnsw/checkpoint.rs @@ -235,8 +235,8 @@ mod tests { assert_eq!(restored.max_layer(), idx.max_layer()); let query = vec![1.0, 2.0, 3.0]; - let orig_results = idx.search(&query, 5, 32); - let rest_results = restored.search(&query, 5, 32); + let orig_results = idx.search(&query, 5, 32).unwrap(); + let rest_results = restored.search(&query, 5, 32).unwrap(); assert_eq!(orig_results.len(), rest_results.len()); for (a, b) in orig_results.iter().zip(rest_results.iter()) { assert_eq!(a.id, b.id); diff --git a/nodedb-vector/src/hnsw/graph/index/backing.rs b/nodedb-vector/src/hnsw/graph/index/backing.rs index 4ecaadb9f..fb5d35cdd 100644 --- a/nodedb-vector/src/hnsw/graph/index/backing.rs +++ b/nodedb-vector/src/hnsw/graph/index/backing.rs @@ -38,7 +38,7 @@ impl HnswIndex { /// segment is unusable — rebuild the index from the authoritative vectors, /// or leave the collection unloaded", never as something to ignore. /// - /// - [`VectorError::DimensionMismatch`] if the backing's `dim()` is not this + /// - [`VectorError::StoredDimensionMismatch`] if the backing's `dim()` is not this /// index's `dim`. /// - [`VectorError::VectorUnavailable`] if the backing holds fewer vectors /// than the index has nodes, or if a node that needs the backing has no @@ -49,7 +49,7 @@ impl HnswIndex { b: Arc, ) -> Result<&mut Self, VectorError> { if b.dim() != self.dim { - return Err(VectorError::DimensionMismatch { + return Err(VectorError::StoredDimensionMismatch { expected: self.dim, got: b.dim(), }); diff --git a/nodedb-vector/src/hnsw/graph/index/state.rs b/nodedb-vector/src/hnsw/graph/index/state.rs index 0743dd12b..ae2c62440 100644 --- a/nodedb-vector/src/hnsw/graph/index/state.rs +++ b/nodedb-vector/src/hnsw/graph/index/state.rs @@ -191,7 +191,7 @@ mod tests { for i in 0..10u32 { idx.insert(vec![i as f32, 0.0, 0.0]).unwrap(); } - let results = idx.search(&[5.0, 0.0, 0.0], 3, 32); + let results = idx.search(&[5.0, 0.0, 0.0], 3, 32).unwrap(); assert_eq!(results.len(), 3); // Results must be in monotonically non-decreasing distance order. for w in results.windows(2) { @@ -210,7 +210,7 @@ mod tests { for i in 0..10u32 { idx.insert(vec![i as f32, 0.0, 0.0]).unwrap(); } - let results = idx.search(&[5.0, 0.0, 0.0], 3, 32); + let results = idx.search(&[5.0, 0.0, 0.0], 3, 32).unwrap(); assert_eq!(results.len(), 3); for w in results.windows(2) { assert!( @@ -376,7 +376,7 @@ mod tests { dim: 4, served: vec![vec![0.0; 4], vec![0.0; 4]], })), - Err(VectorError::DimensionMismatch { .. }) + Err(VectorError::StoredDimensionMismatch { .. }) ), "a backing with the wrong dim must be refused" ); @@ -405,7 +405,7 @@ mod tests { .unwrap() .expect("graph checkpoint must be recognized"); - let results = idx.search(&[1.0, 2.0, 3.0], 5, 16); + let results = idx.search(&[1.0, 2.0, 3.0], 5, 16).unwrap(); for r in &results { assert!( r.distance.is_infinite(), diff --git a/nodedb-vector/src/hnsw/graph/index/vectors.rs b/nodedb-vector/src/hnsw/graph/index/vectors.rs index d41faf70c..bb3e64cfa 100644 --- a/nodedb-vector/src/hnsw/graph/index/vectors.rs +++ b/nodedb-vector/src/hnsw/graph/index/vectors.rs @@ -96,7 +96,7 @@ impl HnswIndex { /// storage is empty and no backing provides the vector. /// - [`VectorError::VectorDecodeFailed`] if dtype-encoded bytes cannot be /// decoded to f32. - /// - [`VectorError::DimensionMismatch`] if the materialized vector's length + /// - [`VectorError::StoredDimensionMismatch`] if the materialized vector's length /// is not `self.dim`. pub fn materialize_vector(&self, id: u32) -> Result, VectorError> { let node = self @@ -121,7 +121,7 @@ impl HnswIndex { None => self.backing_vector(id)?, }; if vector.len() != self.dim { - return Err(VectorError::DimensionMismatch { + return Err(VectorError::StoredDimensionMismatch { expected: self.dim, got: vector.len(), }); diff --git a/nodedb-vector/src/hnsw/search.rs b/nodedb-vector/src/hnsw/search.rs index 4cd4bfa40..3b82e9fbd 100644 --- a/nodedb-vector/src/hnsw/search.rs +++ b/nodedb-vector/src/hnsw/search.rs @@ -32,48 +32,27 @@ fn prefetch_t0(ptr: *const u8) { use roaring::RoaringBitmap; use crate::dtype::cast_from_f32; +use crate::error::{VectorError, check_dim}; use crate::hnsw::graph::{Candidate, HnswIndex, SearchResult}; +/// Maximum beam width to prevent runaway search cost. +const MAX_EF: usize = 8192; + impl HnswIndex { /// K-NN search: find the `k` closest vectors to `query`. /// /// `ef` controls the search beam width (higher = better recall, slower). /// Must be >= k. Typical values: ef = 2*k to 10*k. - pub fn search(&self, query: &[f32], k: usize, ef: usize) -> Vec { - assert_eq!(query.len(), self.dim, "query dimension mismatch"); - if self.is_empty() { - return Vec::new(); - } - - /// Maximum beam width to prevent runaway search cost. - const MAX_EF: usize = 8192; - let ef = ef.max(k).min(MAX_EF); - let Some(ep) = self.entry_point else { - return Vec::new(); - }; - - let query_bytes = cast_from_f32(query, self.params.dtype); - - // Phase 1: Greedy descent from top layer to layer 1. - let mut current_ep = ep; - for layer in (1..=self.max_layer).rev() { - let results = search_layer(self, &query_bytes, current_ep, 1, layer, None, 0); - if let Some(nearest) = results.first() { - current_ep = nearest.id; - } - } - - // Phase 2: Beam search at layer 0. - let results = search_layer(self, &query_bytes, current_ep, ef, 0, None, 0); - - results - .into_iter() - .take(k) - .map(|c| SearchResult { - id: c.id, - distance: c.dist, - }) - .collect() + /// + /// Returns [`VectorError::DimensionMismatch`] when `query` does not have + /// the index dimension. + pub fn search( + &self, + query: &[f32], + k: usize, + ef: usize, + ) -> Result, VectorError> { + self.search_inner(query, k, ef.max(k).min(MAX_EF), None, 0) } /// Filtered K-NN search with Roaring bitmap pre-filtering. @@ -83,7 +62,7 @@ impl HnswIndex { k: usize, ef: usize, filter: &RoaringBitmap, - ) -> Vec { + ) -> Result, VectorError> { self.search_filtered_offset(query, k, ef, filter, 0) } @@ -99,19 +78,60 @@ impl HnswIndex { ef: usize, filter: &RoaringBitmap, id_offset: u32, - ) -> Vec { - assert_eq!(query.len(), self.dim, "query dimension mismatch"); + ) -> Result, VectorError> { + self.search_inner(query, k, ef.max(k), Some(filter), id_offset) + } + + /// Deserialize a Roaring bitmap from bytes and perform filtered search. + pub fn search_with_bitmap_bytes( + &self, + query: &[f32], + k: usize, + ef: usize, + bitmap_bytes: &[u8], + ) -> Result, VectorError> { + self.search_with_bitmap_bytes_offset(query, k, ef, bitmap_bytes, 0) + } + + /// Deserialize a Roaring bitmap and search with an ID offset applied + /// before testing membership. See `search_filtered_offset` for rationale. + /// + /// Bytes that do not decode fail with + /// [`VectorError::InvalidFilterBitmap`]: an unfiltered search would + /// return rows the filter excludes. + pub fn search_with_bitmap_bytes_offset( + &self, + query: &[f32], + k: usize, + ef: usize, + bitmap_bytes: &[u8], + id_offset: u32, + ) -> Result, VectorError> { + let bitmap = decode_filter_bitmap(bitmap_bytes)?; + self.search_filtered_offset(query, k, ef, &bitmap, id_offset) + } + + /// Greedy descent to layer 1, then a beam search of width `ef` at layer + /// 0, restricted to `filter` when one is given. + fn search_inner( + &self, + query: &[f32], + k: usize, + ef: usize, + filter: Option<&RoaringBitmap>, + id_offset: u32, + ) -> Result, VectorError> { + check_dim(self.dim, query.len())?; if self.is_empty() { - return Vec::new(); + return Ok(Vec::new()); } - - let ef = ef.max(k); let Some(ep) = self.entry_point else { - return Vec::new(); + return Ok(Vec::new()); }; let query_bytes = cast_from_f32(query, self.params.dtype); + // Phase 1: Greedy descent from top layer to layer 1. let mut current_ep = ep; for layer in (1..=self.max_layer).rev() { let results = search_layer(self, &query_bytes, current_ep, 1, layer, None, 0); @@ -120,52 +140,25 @@ impl HnswIndex { } } - let results = search_layer( - self, - &query_bytes, - current_ep, - ef, - 0, - Some(filter), - id_offset, - ); + // Phase 2: Beam search at layer 0. + let results = search_layer(self, &query_bytes, current_ep, ef, 0, filter, id_offset); - results + Ok(results .into_iter() .take(k) .map(|c| SearchResult { id: c.id, distance: c.dist, }) - .collect() - } - - /// Deserialize a Roaring bitmap from bytes and perform filtered search. - pub fn search_with_bitmap_bytes( - &self, - query: &[f32], - k: usize, - ef: usize, - bitmap_bytes: &[u8], - ) -> Vec { - self.search_with_bitmap_bytes_offset(query, k, ef, bitmap_bytes, 0) + .collect()) } +} - /// Deserialize a Roaring bitmap and search with an ID offset applied - /// before testing membership. See `search_filtered_offset` for rationale. - pub fn search_with_bitmap_bytes_offset( - &self, - query: &[f32], - k: usize, - ef: usize, - bitmap_bytes: &[u8], - id_offset: u32, - ) -> Vec { - match RoaringBitmap::deserialize_from(bitmap_bytes) { - Ok(bitmap) => self.search_filtered_offset(query, k, ef, &bitmap, id_offset), - Err(_) => self.search(query, k, ef), - } - } +/// Decode a serialized Roaring pre-filter bitmap. +pub fn decode_filter_bitmap(bytes: &[u8]) -> Result { + RoaringBitmap::deserialize_from(bytes).map_err(|e| VectorError::InvalidFilterBitmap { + detail: e.to_string(), + }) } /// Unified HNSW beam search on a single layer with optional pre-filter. @@ -313,7 +306,7 @@ mod tests { #[test] fn search_empty_index() { let idx = HnswIndex::new(3, HnswParams::default()); - let results = idx.search(&[1.0, 2.0, 3.0], 5, 50); + let results = idx.search(&[1.0, 2.0, 3.0], 5, 50).unwrap(); assert!(results.is_empty()); } @@ -331,7 +324,7 @@ mod tests { 1, ); idx.insert(vec![1.0, 0.0]).unwrap(); - let results = idx.search(&[1.0, 0.0], 1, 10); + let results = idx.search(&[1.0, 0.0], 1, 10).unwrap(); assert_eq!(results.len(), 1); assert_eq!(results[0].id, 0); assert!(results[0].distance < 1e-6); @@ -341,7 +334,7 @@ mod tests { fn search_finds_exact_match() { let idx = build_index(50, 3); let query = idx.get_vector(25).unwrap().to_vec(); - let results = idx.search(&query, 1, 50); + let results = idx.search(&query, 1, 50).unwrap(); assert_eq!(results.len(), 1); assert_eq!(results[0].id, 25); assert!(results[0].distance < 1e-6); @@ -351,7 +344,7 @@ mod tests { fn search_returns_sorted_by_distance() { let idx = build_index(100, 4); let query = vec![50.0, 50.0, 50.0, 50.0]; - let results = idx.search(&query, 10, 64); + let results = idx.search(&query, 10, 64).unwrap(); assert_eq!(results.len(), 10); for w in results.windows(2) { assert!(w[0].distance <= w[1].distance); @@ -361,7 +354,7 @@ mod tests { #[test] fn search_k_larger_than_index() { let idx = build_index(5, 2); - let results = idx.search(&[0.0, 0.0], 20, 50); + let results = idx.search(&[0.0, 0.0], 20, 50).unwrap(); assert_eq!(results.len(), 5); } @@ -369,7 +362,7 @@ mod tests { fn search_recall_at_10() { let idx = build_index(500, 3); let query = vec![100.0, 100.0, 100.0]; - let results = idx.search(&query, 10, 128); + let results = idx.search(&query, 10, 128).unwrap(); let mut truth: Vec<(u32, f32)> = (0..500) .map(|i| { @@ -390,7 +383,7 @@ mod tests { fn search_excludes_tombstoned() { let mut idx = build_index(20, 3); idx.delete(0); - let results = idx.search(&[0.0, 0.0, 0.0], 5, 32); + let results = idx.search(&[0.0, 0.0, 0.0], 5, 32).unwrap(); for r in &results { assert_ne!(r.id, 0, "tombstoned node appeared in results"); } @@ -403,7 +396,9 @@ mod tests { for i in (0..50u32).step_by(2) { filter.insert(i); } - let results = idx.search_filtered(&[0.0, 0.0, 0.0], 5, 64, &filter); + let results = idx + .search_filtered(&[0.0, 0.0, 0.0], 5, 64, &filter) + .unwrap(); assert_eq!(results.len(), 5); for r in &results { assert!(r.id % 2 == 0, "got odd id {}", r.id); @@ -414,7 +409,9 @@ mod tests { fn search_filtered_empty_returns_empty() { let idx = build_index(20, 3); let filter = RoaringBitmap::new(); - let results = idx.search_filtered(&[0.0, 0.0, 0.0], 5, 64, &filter); + let results = idx + .search_filtered(&[0.0, 0.0, 0.0], 5, 64, &filter) + .unwrap(); assert!(results.is_empty()); } @@ -427,9 +424,56 @@ mod tests { } let mut bytes = Vec::new(); filter.serialize_into(&mut bytes).unwrap(); - let results = idx.search_with_bitmap_bytes(&[0.0, 0.0, 0.0], 5, 32, &bytes); + let results = idx + .search_with_bitmap_bytes(&[0.0, 0.0, 0.0], 5, 32, &bytes) + .unwrap(); for r in &results { assert!(r.id < 25, "got filtered-out node {}", r.id); } } + + /// A query of the wrong dimension is a typed error on every entry point, + /// on an empty index and a populated one alike. + #[test] + fn wrong_dimension_query_is_a_typed_error() { + use crate::error::VectorError; + let empty = HnswIndex::new(3, HnswParams::default()); + let idx = build_index(20, 3); + let filter: RoaringBitmap = (0..20u32).collect(); + let mut bytes = Vec::new(); + filter.serialize_into(&mut bytes).unwrap(); + let short = [0.0_f32, 0.0]; + for result in [ + empty.search(&short, 5, 32), + idx.search(&short, 5, 32), + idx.search_filtered(&short, 5, 32, &filter), + idx.search_filtered_offset(&short, 5, 32, &filter, 0), + idx.search_with_bitmap_bytes(&short, 5, 32, &bytes), + idx.search_with_bitmap_bytes_offset(&short, 5, 32, &bytes, 0), + ] { + assert!( + matches!( + result, + Err(VectorError::DimensionMismatch { + expected: 3, + got: 2 + }) + ), + "{result:?}" + ); + } + } + + /// Filter bytes that do not decode fail the search: an unfiltered + /// search would return rows the filter excludes. + #[test] + fn undecodable_filter_bitmap_is_a_typed_error() { + use crate::error::VectorError; + let idx = build_index(20, 3); + let result = idx.search_with_bitmap_bytes(&[0.0, 0.0, 0.0], 5, 32, &[0xff, 0x01]); + assert!( + matches!(result, Err(VectorError::InvalidFilterBitmap { .. })), + "{result:?}" + ); + } } diff --git a/nodedb-vector/src/index_config.rs b/nodedb-vector/src/index_config.rs index 90f4ab6b3..483b268c1 100644 --- a/nodedb-vector/src/index_config.rs +++ b/nodedb-vector/src/index_config.rs @@ -23,7 +23,9 @@ pub enum IndexType { Hnsw, /// HNSW graph with PQ-compressed storage for traversal. HnswPq, - /// IVF-PQ flat index. Lowest memory (~16 bytes/vector), best for >10M vectors. + /// IVF-PQ: vectors buffer, searched exactly, until `max(ivf_cells, pq_k)` + /// are held, then train k-means cells and PQ codebooks. A search probes + /// `ivf_nprobe` cells by PQ distance and reranks by exact distance. IvfPq, } diff --git a/nodedb-vector/src/ivf.rs b/nodedb-vector/src/ivf.rs deleted file mode 100644 index bcc371de4..000000000 --- a/nodedb-vector/src/ivf.rs +++ /dev/null @@ -1,357 +0,0 @@ -// SPDX-License-Identifier: Apache-2.0 - -//! IVF-PQ index for billion-scale datasets. -//! -//! Inverted File with Product Quantization: partition vectors into Voronoi -//! cells using k-means centroids, PQ-compress within cells. - -use nodedb_mem::ScopedMemory; - -use crate::distance::{DistanceMetric, distance}; -use crate::hnsw::SearchResult; -use crate::quantize::pq::PqCodec; - -/// IVF-PQ index configuration. -#[derive(Clone)] -pub struct IvfPqParams { - /// Number of Voronoi cells (partitions). Typical: sqrt(N). - pub n_cells: usize, - /// Number of PQ subvectors. Must divide dimension evenly. - pub pq_m: usize, - /// Centroids per PQ subvector (fixed at 256 for u8 encoding). - pub pq_k: usize, - /// Number of cells to probe at query time. Higher = better recall. - pub nprobe: usize, - /// Distance metric. - pub metric: DistanceMetric, -} - -impl Default for IvfPqParams { - fn default() -> Self { - Self { - n_cells: 256, - pq_m: 8, - pq_k: 256, - nprobe: 16, - metric: DistanceMetric::L2, - } - } -} - -/// IVF-PQ index: inverted file with product quantization. -pub struct IvfPqIndex { - dim: usize, - params: IvfPqParams, - /// Coarse centroids: `n_cells` × `dim` FP32 vectors. - centroids: Vec>, - /// PQ codec trained on the dataset. - pq: Option, - /// Per-cell inverted lists: `cells[cell_id]` = list of (vector_id, pq_code). - cells: Vec)>>, - /// Total vectors indexed. - count: u32, -} - -impl IvfPqIndex { - /// Create an empty IVF-PQ index. - pub fn new(dim: usize, params: IvfPqParams) -> Self { - Self { - dim, - params, - centroids: Vec::new(), - pq: None, - cells: Vec::new(), - count: 0, - } - } - - /// Train the index from a set of vectors, tracking PQ codebook - /// allocations against `memory`. - pub fn train(&mut self, vectors: &[&[f32]], memory: ScopedMemory) { - assert!(!vectors.is_empty()); - assert!(self.dim > 0); - assert!( - self.dim.is_multiple_of(self.params.pq_m), - "dim {} must be divisible by pq_m {}", - self.dim, - self.params.pq_m - ); - - let n_cells = self.params.n_cells.min(vectors.len()); - self.centroids = kmeans_centroids(vectors, self.dim, n_cells, 20); - self.cells = vec![Vec::new(); self.centroids.len()]; - - let mut residuals: Vec> = Vec::with_capacity(vectors.len()); - for v in vectors { - let cell = self.nearest_centroid(v); - let res: Vec = v - .iter() - .zip(&self.centroids[cell]) - .map(|(a, b)| a - b) - .collect(); - residuals.push(res); - } - let res_refs: Vec<&[f32]> = residuals.iter().map(|r| r.as_slice()).collect(); - self.pq = Some(PqCodec::train( - &res_refs, - self.dim, - self.params.pq_m, - self.params.pq_k, - 20, - memory, - )); - } - - /// Add a vector to the index. Returns the assigned ID. - pub fn add(&mut self, vector: &[f32]) -> u32 { - assert_eq!(vector.len(), self.dim); - let pq = self - .pq - .as_ref() - .expect("index must be trained before add()"); - - let cell = self.nearest_centroid(vector); - let residual: Vec = vector - .iter() - .zip(&self.centroids[cell]) - .map(|(a, b)| a - b) - .collect(); - let code = pq.encode(&residual); - let id = self.count; - self.cells[cell].push((id, code)); - self.count += 1; - id - } - - /// Batch add vectors. - pub fn add_batch(&mut self, vectors: &[&[f32]]) { - for v in vectors { - self.add(v); - } - } - - /// Search: find top-k nearest neighbors. - pub fn search(&self, query: &[f32], top_k: usize) -> Vec { - assert_eq!(query.len(), self.dim); - if self.centroids.is_empty() || self.count == 0 { - return Vec::new(); - } - - let pq = match &self.pq { - Some(p) => p, - None => return Vec::new(), - }; - - let nprobe = self.params.nprobe.min(self.centroids.len()); - let mut centroid_dists: Vec<(usize, f32)> = self - .centroids - .iter() - .enumerate() - .map(|(i, c)| (i, distance(query, c, self.params.metric))) - .collect(); - centroid_dists.sort_by(|a, b| a.1.partial_cmp(&b.1).unwrap_or(std::cmp::Ordering::Equal)); - - let mut candidates: Vec = Vec::new(); - - for &(cell_idx, _) in centroid_dists.iter().take(nprobe) { - let residual_query: Vec = query - .iter() - .zip(&self.centroids[cell_idx]) - .map(|(q, c)| q - c) - .collect(); - let table = match pq.build_distance_table(&residual_query) { - Ok(t) => t, - Err(e) => { - tracing::warn!(error = %e, "IVF PQ build_distance_table budget exhausted; skipping cell"); - continue; - } - }; - - for (id, code) in &self.cells[cell_idx] { - let dist = pq.asymmetric_distance(&table, code); - candidates.push(SearchResult { - id: *id, - distance: dist, - }); - } - } - - if candidates.len() > top_k { - candidates.select_nth_unstable_by(top_k, |a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - candidates.truncate(top_k); - } - candidates.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - candidates - } - - fn nearest_centroid(&self, vector: &[f32]) -> usize { - let mut best = 0; - let mut best_dist = f32::MAX; - for (i, c) in self.centroids.iter().enumerate() { - let d = distance(vector, c, self.params.metric); - if d < best_dist { - best_dist = d; - best = i; - } - } - best - } - - pub fn len(&self) -> usize { - self.count as usize - } - - pub fn is_empty(&self) -> bool { - self.count == 0 - } - - pub fn dim(&self) -> usize { - self.dim - } - - pub fn n_cells(&self) -> usize { - self.centroids.len() - } -} - -fn kmeans_centroids(data: &[&[f32]], dim: usize, k: usize, max_iter: usize) -> Vec> { - let n = data.len(); - let k = k.min(n); - if k == 0 { - return Vec::new(); - } - - let mut centroids: Vec> = vec![data[0].to_vec()]; - let mut min_dists = vec![f32::MAX; n]; - - // Initialize min_dists against the first centroid. - for (i, point) in data.iter().enumerate() { - let d = distance(point, ¢roids[0], DistanceMetric::L2); - if d < min_dists[i] { - min_dists[i] = d; - } - } - - let mut rng = crate::hnsw::Xorshift64::new(0xC0FF_EEDE_ADBE_EF42); - for _ in 1..k { - let total: f64 = min_dists.iter().map(|&d| d as f64).sum(); - let next_idx = if total < f64::EPSILON { - 0 - } else { - let target = rng.next_f64() * total; - let mut acc = 0.0f64; - let mut chosen = n - 1; - for (i, &d) in min_dists.iter().enumerate() { - acc += d as f64; - if acc >= target { - chosen = i; - break; - } - } - chosen - }; - centroids.push(data[next_idx].to_vec()); - let last = centroids.last().expect("just pushed"); - for (i, point) in data.iter().enumerate() { - let d = distance(point, last, DistanceMetric::L2); - if d < min_dists[i] { - min_dists[i] = d; - } - } - } - - let mut assignments = vec![0usize; n]; - for _ in 0..max_iter { - let mut changed = false; - for (i, point) in data.iter().enumerate() { - let mut best = 0; - let mut best_d = f32::MAX; - for (c, centroid) in centroids.iter().enumerate() { - let d = distance(point, centroid, DistanceMetric::L2); - if d < best_d { - best_d = d; - best = c; - } - } - if assignments[i] != best { - assignments[i] = best; - changed = true; - } - } - if !changed { - break; - } - let mut sums = vec![vec![0.0f32; dim]; k]; - let mut counts = vec![0usize; k]; - for (i, point) in data.iter().enumerate() { - let c = assignments[i]; - counts[c] += 1; - for d in 0..dim { - sums[c][d] += point[d]; - } - } - for c in 0..k { - if counts[c] > 0 { - for d in 0..dim { - centroids[c][d] = sums[c][d] / counts[c] as f32; - } - } - } - } - centroids -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::test_support::test_memory; - - fn make_vectors(n: usize, dim: usize) -> Vec> { - (0..n) - .map(|i| (0..dim).map(|d| ((i * dim + d) as f32) * 0.01).collect()) - .collect() - } - - #[test] - fn train_and_search() { - let vecs = make_vectors(1000, 16); - let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - - let mut idx = IvfPqIndex::new( - 16, - IvfPqParams { - n_cells: 32, - pq_m: 4, - pq_k: 32, - nprobe: 8, - metric: DistanceMetric::L2, - }, - ); - idx.train(&refs, test_memory()); - idx.add_batch(&refs); - - assert_eq!(idx.len(), 1000); - - let query = &vecs[500]; - let results = idx.search(query, 5); - assert_eq!(results.len(), 5); - assert!( - results.iter().any(|r| r.id == 500), - "exact match not found in top-5" - ); - } - - #[test] - fn empty_index() { - let idx = IvfPqIndex::new(8, IvfPqParams::default()); - assert!(idx.search(&[0.0; 8], 5).is_empty()); - } -} diff --git a/nodedb-vector/src/ivf/checkpoint.rs b/nodedb-vector/src/ivf/checkpoint.rs new file mode 100644 index 000000000..296fb4965 --- /dev/null +++ b/nodedb-vector/src/ivf/checkpoint.rs @@ -0,0 +1,165 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Checkpoint encoding for `IvfPqIndex`: centroids, the PQ codec, every +//! cell's ids, codes and FP32 vectors, the tombstones, and the training stamp. +//! Codes are stored, never recomputed on load. + +use std::collections::HashMap; + +use nodedb_mem::ScopedMemory; +use roaring::RoaringBitmap; + +use crate::error::VectorError; +use crate::quantize::pq::PqCodec; + +use super::index::{IvfCell, IvfPqIndex}; +use super::params::IvfPqParams; + +#[derive(zerompk::ToMessagePack, zerompk::FromMessagePack)] +struct IvfSnapshot { + dim: usize, + params: IvfPqParams, + centroids: Vec>, + pq_bytes: Option>, + cells: Vec, + deleted: Vec, + next_id: u32, + trained_on: usize, + trained_at_ms: u64, +} + +fn corrupt(detail: String) -> VectorError { + VectorError::CheckpointDeserializationError { detail } +} + +impl IvfPqIndex { + /// Encode the whole index as MessagePack. + pub fn to_bytes(&self) -> Result, VectorError> { + let snapshot = IvfSnapshot { + dim: self.dim, + params: self.params.clone(), + centroids: self.centroids.clone(), + pq_bytes: self.pq.as_ref().map(PqCodec::to_bytes).transpose()?, + cells: self.cells.clone(), + deleted: self.deleted.iter().collect(), + next_id: self.next_id, + trained_on: self.trained_on, + trained_at_ms: self.trained_at_ms, + }; + zerompk::to_msgpack_vec(&snapshot).map_err(|e| VectorError::CheckpointSerializationError { + detail: format!("IVF-PQ index encode: {e}"), + }) + } + + /// Decode an index written by [`Self::to_bytes`], charging the PQ codec + /// to `memory`. + /// + /// Fails with [`VectorError::CheckpointDeserializationError`] when the + /// bytes do not decode or the decoded cells disagree with the dimension, + /// the PQ code width, or the centroid count. + pub fn from_bytes(bytes: &[u8], memory: ScopedMemory) -> Result { + let snap: IvfSnapshot = zerompk::from_msgpack(bytes) + .map_err(|e| corrupt(format!("IVF-PQ index decode: {e}")))?; + let pq = snap + .pq_bytes + .as_deref() + .map(|b| PqCodec::from_bytes(b, memory)) + .transpose() + .map_err(|e| corrupt(format!("IVF-PQ codec decode: {e}")))?; + let m = pq.as_ref().map_or(0, |pq| pq.m); + if snap.cells.len() != snap.centroids.len() { + return Err(corrupt(format!( + "IVF-PQ index has {} cells for {} centroids", + snap.cells.len(), + snap.centroids.len() + ))); + } + let mut slots = HashMap::new(); + for (cell_idx, cell) in snap.cells.iter().enumerate() { + let n = cell.ids.len(); + if cell.codes.len() != n * m || cell.vectors.len() != n * snap.dim { + return Err(corrupt(format!( + "IVF-PQ cell {cell_idx} holds {n} ids, {} code bytes and {} vector \ + components; expected {} and {}", + cell.codes.len(), + cell.vectors.len(), + n * m, + n * snap.dim + ))); + } + for (pos, &id) in cell.ids.iter().enumerate() { + if slots.insert(id, (cell_idx as u32, pos as u32)).is_some() { + return Err(corrupt(format!("IVF-PQ index holds vector id {id} twice"))); + } + } + } + let deleted: RoaringBitmap = snap.deleted.into_iter().collect(); + if let Some(id) = deleted.iter().find(|id| !slots.contains_key(id)) { + return Err(corrupt(format!( + "IVF-PQ index tombstones vector id {id} it does not hold" + ))); + } + Ok(Self { + dim: snap.dim, + params: snap.params, + centroids: snap.centroids, + pq, + cells: snap.cells, + slots, + deleted, + next_id: snap.next_id, + trained_on: snap.trained_on, + trained_at_ms: snap.trained_at_ms, + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::distance::DistanceMetric; + use crate::test_support::test_memory; + + #[test] + fn round_trip_keeps_entries_tombstones_and_training() { + let vecs: Vec> = (0..32) + .map(|i| (0..8).map(|d| ((i * 8 + d) as f32) * 0.01).collect()) + .collect(); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let mut idx = IvfPqIndex::new( + 8, + IvfPqParams { + n_cells: 4, + pq_m: 4, + pq_k: 8, + nprobe: 4, + metric: DistanceMetric::L2, + }, + ); + idx.train(&refs, test_memory()).unwrap(); + for (i, v) in vecs.iter().enumerate() { + idx.insert_with_id(10 + i as u32, v.clone()).unwrap(); + } + idx.delete(12); + idx.set_trained_at_ms(1_700_000_000_000); + + let restored = IvfPqIndex::from_bytes(&idx.to_bytes().unwrap(), test_memory()).unwrap(); + assert_eq!(restored.len(), 32); + assert!(restored.is_deleted(12)); + assert_eq!(restored.trained_on(), 32); + assert_eq!(restored.trained_at_ms(), 1_700_000_000_000); + assert_eq!(restored.get_vector(20), Some(vecs[10].as_slice())); + assert_eq!( + restored.search(&vecs[5], 3).unwrap()[0].id, + idx.search(&vecs[5], 3).unwrap()[0].id + ); + } + + #[test] + fn garbage_is_a_typed_error() { + assert!(matches!( + IvfPqIndex::from_bytes(b"not an index", test_memory()), + Err(VectorError::CheckpointDeserializationError { .. }) + )); + } +} diff --git a/nodedb-vector/src/ivf/index.rs b/nodedb-vector/src/ivf/index.rs new file mode 100644 index 000000000..e52c57ac8 --- /dev/null +++ b/nodedb-vector/src/ivf/index.rs @@ -0,0 +1,494 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! IVF-PQ index: inverted file with product quantization. +//! +//! k-means centroids partition the vectors into Voronoi cells. Each entry +//! holds a PQ code of its residual against the cell centroid, plus the FP32 +//! vector. A search probes the nearest cells, ranks their entries by PQ +//! distance, and reranks the best of them by exact distance. +//! +//! Entries carry caller-assigned ids, so a collection can move its vectors +//! into the index under the ids they already have. + +use std::collections::HashMap; + +use nodedb_mem::ScopedMemory; +use roaring::RoaringBitmap; + +use crate::distance::distance; +use crate::error::{VectorError, check_dim}; +use crate::quantize::pq::PqCodec; + +use super::kmeans::kmeans_centroids; +use super::params::IvfPqParams; + +/// k-means iterations for the coarse centroids and the PQ codebooks. +pub(super) const TRAIN_ITERATIONS: usize = 20; + +/// The entries of one Voronoi cell, stored column-wise. Entry `i` has id +/// `ids[i]`, PQ code `codes[i * m .. (i + 1) * m]`, and FP32 vector +/// `vectors[i * dim .. (i + 1) * dim]`. +#[derive(Debug, Clone, Default, zerompk::ToMessagePack, zerompk::FromMessagePack)] +pub struct IvfCell { + pub(super) ids: Vec, + pub(super) codes: Vec, + pub(super) vectors: Vec, +} + +/// IVF-PQ index: inverted file with product quantization. +pub struct IvfPqIndex { + pub(super) dim: usize, + pub(super) params: IvfPqParams, + /// Coarse centroids: `n_cells` × `dim` FP32 vectors. + pub(super) centroids: Vec>, + /// PQ codec trained on the residuals. `None` until [`Self::train`]. + pub(super) pq: Option, + pub(super) cells: Vec, + /// Entry id → (cell, position in the cell). + pub(super) slots: HashMap, + /// Soft-deleted entry ids. + pub(super) deleted: RoaringBitmap, + /// Id [`Self::add`] assigns next: one past the highest id ever held. + pub(super) next_id: u32, + /// Vectors the codebooks were trained on. + pub(super) trained_on: usize, + /// Unix milliseconds of the training, as the caller stamped it. `0` when + /// never stamped. + pub(super) trained_at_ms: u64, +} + +impl IvfPqIndex { + /// Create an empty, untrained IVF-PQ index. + pub fn new(dim: usize, params: IvfPqParams) -> Self { + Self { + dim, + params, + centroids: Vec::new(), + pq: None, + cells: Vec::new(), + slots: HashMap::new(), + deleted: RoaringBitmap::new(), + next_id: 0, + trained_on: 0, + trained_at_ms: 0, + } + } + + /// Train the coarse centroids and the PQ codebooks on `vectors`, tracking + /// PQ codebook allocations against `memory`. + /// + /// Fails with [`VectorError::InvalidInput`] when the set is empty, the + /// dimension is zero or not divisible by `pq_m`, the set is smaller than + /// `pq_k`, or the index already holds vectors (their codes would not + /// match new codebooks). A vector without the index dimension fails with + /// [`VectorError::DimensionMismatch`]. A failed training changes nothing. + pub fn train(&mut self, vectors: &[&[f32]], memory: ScopedMemory) -> Result<(), VectorError> { + if vectors.is_empty() { + return Err(VectorError::InvalidInput { + detail: "IVF-PQ training needs at least one vector".into(), + }); + } + if !self.slots.is_empty() { + return Err(VectorError::InvalidInput { + detail: format!( + "IVF-PQ index already holds {} vectors; train an empty index", + self.slots.len() + ), + }); + } + if self.dim == 0 || self.params.pq_m == 0 || !self.dim.is_multiple_of(self.params.pq_m) { + return Err(VectorError::InvalidInput { + detail: format!( + "IVF-PQ dimension {} must be non-zero and divisible by pq_m {}", + self.dim, self.params.pq_m + ), + }); + } + for v in vectors { + check_dim(self.dim, v.len())?; + } + + let n_cells = self.params.n_cells.min(vectors.len()); + let centroids = kmeans_centroids(vectors, self.dim, n_cells, TRAIN_ITERATIONS); + let residuals: Vec> = vectors + .iter() + .map(|v| residual(v, ¢roids[nearest(¢roids, v, &self.params)])) + .collect(); + let res_refs: Vec<&[f32]> = residuals.iter().map(|r| r.as_slice()).collect(); + let pq = PqCodec::train( + &res_refs, + self.dim, + self.params.pq_m, + self.params.pq_k, + TRAIN_ITERATIONS, + memory, + )?; + + self.cells = vec![IvfCell::default(); centroids.len()]; + self.centroids = centroids; + self.pq = Some(pq); + self.trained_on = vectors.len(); + Ok(()) + } + + /// Add `vector` under the caller-assigned `id`. + /// + /// Fails with [`VectorError::DimensionMismatch`] for a vector without the + /// index dimension, and with [`VectorError::InvalidInput`] when the index + /// is untrained or already holds `id`. A failed add changes nothing. + pub fn insert_with_id(&mut self, id: u32, vector: Vec) -> Result<(), VectorError> { + check_dim(self.dim, vector.len())?; + let Some(pq) = self.pq.as_ref() else { + return Err(VectorError::InvalidInput { + detail: "IVF-PQ index must be trained before add".into(), + }); + }; + if self.slots.contains_key(&id) { + return Err(VectorError::InvalidInput { + detail: format!("IVF-PQ index already holds vector id {id}"), + }); + } + let cell_idx = nearest(&self.centroids, &vector, &self.params); + let code = pq.encode(&residual(&vector, &self.centroids[cell_idx])); + let cell = &mut self.cells[cell_idx]; + let pos = cell.ids.len() as u32; + cell.ids.push(id); + cell.codes.extend_from_slice(&code); + cell.vectors.extend_from_slice(&vector); + self.slots.insert(id, (cell_idx as u32, pos)); + self.next_id = self.next_id.max(id.saturating_add(1)); + Ok(()) + } + + /// Add a vector under the next free id. Returns the id. + /// + /// Fails as [`Self::insert_with_id`] does. + pub fn add(&mut self, vector: &[f32]) -> Result { + let id = self.next_id; + self.insert_with_id(id, vector.to_vec())?; + Ok(id) + } + + /// Add vectors under consecutive free ids. Stops at the first vector + /// that fails to add. + pub fn add_batch(&mut self, vectors: &[&[f32]]) -> Result<(), VectorError> { + for v in vectors { + self.add(v)?; + } + Ok(()) + } + + /// Whether the index holds a trained codebook. + pub fn is_trained(&self) -> bool { + self.pq.is_some() + } + + /// Whether the index holds `id`, live or soft-deleted. + pub fn contains(&self, id: u32) -> bool { + self.slots.contains_key(&id) + } + + /// Whether `id` is held and soft-deleted. + pub fn is_deleted(&self, id: u32) -> bool { + self.deleted.contains(id) + } + + /// Soft-delete `id`. `false` when the index does not hold it live. + pub fn delete(&mut self, id: u32) -> bool { + self.contains(id) && self.deleted.insert(id) + } + + /// Reverse a soft delete of `id`. `false` when `id` was not deleted. + pub fn undelete(&mut self, id: u32) -> bool { + self.deleted.remove(id) + } + + /// The FP32 vector of a live `id`. + pub fn get_vector(&self, id: u32) -> Option<&[f32]> { + if self.deleted.contains(id) { + return None; + } + let &(cell, pos) = self.slots.get(&id)?; + let start = pos as usize * self.dim; + self.cells + .get(cell as usize)? + .vectors + .get(start..start + self.dim) + } + + /// Every live entry as `(id, vector)`, ordered by id. + pub fn live_vectors(&self) -> Vec<(u32, Vec)> { + let mut out: Vec<(u32, Vec)> = Vec::with_capacity(self.live_count()); + for cell in &self.cells { + for (pos, &id) in cell.ids.iter().enumerate() { + if !self.deleted.contains(id) { + let start = pos * self.dim; + out.push((id, cell.vectors[start..start + self.dim].to_vec())); + } + } + } + out.sort_unstable_by_key(|(id, _)| *id); + out + } + + /// Drop every entry with id `next_id` or later, so the next + /// [`Self::add`] takes `next_id` again. The training stays. A rollback + /// uses it to withdraw the newest adds. + pub fn roll_back_to(&mut self, next_id: u32) { + self.retain_entries(|id| id < next_id); + self.deleted.remove_range(next_id..); + self.next_id = self.next_id.min(next_id); + } + + /// Remove every soft-deleted entry. Returns the number removed. Ids of + /// the remaining entries do not change. + pub fn compact(&mut self) -> usize { + let deleted = std::mem::take(&mut self.deleted); + self.retain_entries(|id| !deleted.contains(id)) + } + + /// Keep the entries whose id satisfies `keep` and rebuild the slot map. + /// Returns the number of entries dropped. + fn retain_entries(&mut self, keep: impl Fn(u32) -> bool) -> usize { + let m = self.pq.as_ref().map_or(0, |pq| pq.m); + let dim = self.dim; + let mut removed = 0; + self.slots.clear(); + for (cell_idx, cell) in self.cells.iter_mut().enumerate() { + let mut kept = IvfCell::default(); + for (pos, &id) in cell.ids.iter().enumerate() { + if !keep(id) { + removed += 1; + continue; + } + self.slots + .insert(id, (cell_idx as u32, kept.ids.len() as u32)); + kept.ids.push(id); + kept.codes + .extend_from_slice(&cell.codes[pos * m..(pos + 1) * m]); + kept.vectors + .extend_from_slice(&cell.vectors[pos * dim..(pos + 1) * dim]); + } + *cell = kept; + } + removed + } + + /// Entries held, live or soft-deleted. + pub fn len(&self) -> usize { + self.slots.len() + } + + /// Entries held and not soft-deleted. + pub fn live_count(&self) -> usize { + self.slots.len() - self.deleted.len() as usize + } + + /// Soft-deleted entries held. + pub fn tombstone_count(&self) -> usize { + self.deleted.len() as usize + } + + pub fn is_empty(&self) -> bool { + self.slots.is_empty() + } + + pub fn dim(&self) -> usize { + self.dim + } + + pub fn n_cells(&self) -> usize { + self.centroids.len() + } + + pub fn params(&self) -> &IvfPqParams { + &self.params + } + + /// Vectors the codebooks were trained on. `0` when untrained. + pub fn trained_on(&self) -> usize { + self.trained_on + } + + /// Unix milliseconds the caller stamped the training with. + pub fn trained_at_ms(&self) -> u64 { + self.trained_at_ms + } + + /// Stamp the training time, in Unix milliseconds. + pub fn set_trained_at_ms(&mut self, ms: u64) { + self.trained_at_ms = ms; + } + + /// Approximate heap bytes of the centroids, codes and vectors. + pub fn memory_bytes(&self) -> usize { + let f32_size = std::mem::size_of::(); + let centroids = self.centroids.len() * self.dim * f32_size; + let entries: usize = self + .cells + .iter() + .map(|c| { + c.ids.len() * std::mem::size_of::() + + c.codes.len() + + c.vectors.len() * f32_size + }) + .sum(); + centroids + entries + } +} + +/// Index of the centroid in `centroids` nearest to `vector`. `0` when there +/// are none. +fn nearest(centroids: &[Vec], vector: &[f32], params: &IvfPqParams) -> usize { + let mut best = 0; + let mut best_dist = f32::MAX; + for (i, c) in centroids.iter().enumerate() { + let d = distance(vector, c, params.metric); + if d < best_dist { + best_dist = d; + best = i; + } + } + best +} + +/// `vector - centroid`, component-wise. +pub(super) fn residual(vector: &[f32], centroid: &[f32]) -> Vec { + vector.iter().zip(centroid).map(|(a, b)| a - b).collect() +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::distance::DistanceMetric; + use crate::test_support::test_memory; + + fn make_vectors(n: usize, dim: usize) -> Vec> { + (0..n) + .map(|i| (0..dim).map(|d| ((i * dim + d) as f32) * 0.01).collect()) + .collect() + } + + fn small_params() -> IvfPqParams { + IvfPqParams { + n_cells: 4, + pq_m: 4, + pq_k: 8, + nprobe: 4, + metric: DistanceMetric::L2, + } + } + + fn trained(vecs: &[Vec]) -> IvfPqIndex { + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let mut idx = IvfPqIndex::new(8, small_params()); + idx.train(&refs, test_memory()).unwrap(); + idx + } + + #[test] + fn rolling_back_withdraws_every_vector_added_after_the_mark() { + let vecs = make_vectors(64, 8); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let mut idx = trained(&vecs); + idx.add_batch(&refs[..40]).unwrap(); + let mark = idx.len() as u32; + let before: Vec = idx + .search(&vecs[5], 40) + .unwrap() + .iter() + .map(|r| r.id) + .collect(); + + idx.add_batch(&refs[40..]).unwrap(); + idx.roll_back_to(mark); + + assert_eq!(idx.len(), 40); + assert!(idx.is_trained(), "the training the index held stays"); + let after_ids: Vec = idx + .search(&vecs[5], 64) + .unwrap() + .iter() + .map(|r| r.id) + .collect(); + assert!(after_ids.iter().all(|id| *id < mark)); + assert_eq!(after_ids, before, "the search reads as before the adds"); + + // The next add takes the first id past the mark again. + assert_eq!(idx.add(&vecs[63]).unwrap(), mark); + } + + #[test] + fn caller_ids_survive_delete_compact_and_reads() { + let vecs = make_vectors(16, 8); + let mut idx = trained(&vecs); + for (i, v) in vecs.iter().enumerate() { + idx.insert_with_id(100 + i as u32, v.clone()).unwrap(); + } + assert!(matches!( + idx.insert_with_id(100, vecs[0].clone()), + Err(VectorError::InvalidInput { .. }) + )); + assert!(idx.delete(103)); + assert!(!idx.delete(103), "a second delete finds nothing live"); + assert!(idx.get_vector(103).is_none()); + assert_eq!(idx.live_count(), 15); + + assert_eq!(idx.compact(), 1); + assert_eq!(idx.len(), 15); + assert!(!idx.contains(103)); + assert_eq!(idx.get_vector(104), Some(vecs[4].as_slice())); + assert_eq!(idx.add(&vecs[0]).unwrap(), 116); + } + + #[test] + fn a_trained_index_holding_vectors_refuses_to_retrain() { + let vecs = make_vectors(16, 8); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let mut idx = trained(&vecs); + idx.add(&vecs[0]).unwrap(); + assert!(matches!( + idx.train(&refs, test_memory()), + Err(VectorError::InvalidInput { .. }) + )); + assert_eq!(idx.len(), 1); + } + + #[test] + fn wrong_dimension_is_a_typed_error() { + let vecs: Vec> = (0..32) + .map(|i| (0..8).map(|d| ((i * 8 + d) % 17) as f32).collect()) + .collect(); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let mut idx = IvfPqIndex::new(8, small_params()); + assert!(matches!( + idx.add(&[0.0; 8]), + Err(VectorError::InvalidInput { .. }) + )); + idx.train(&refs, test_memory()).unwrap(); + idx.add_batch(&refs).unwrap(); + assert!(matches!( + idx.search(&[0.0; 3], 5), + Err(VectorError::DimensionMismatch { + expected: 8, + got: 3 + }) + )); + assert!(matches!( + idx.add(&[0.0; 3]), + Err(VectorError::DimensionMismatch { + expected: 8, + got: 3 + }) + )); + let short = [0.0_f32; 3]; + let mut untrained = IvfPqIndex::new(8, IvfPqParams::default()); + assert!(matches!( + untrained.train(&[&short], test_memory()), + Err(VectorError::DimensionMismatch { + expected: 8, + got: 3 + }) + )); + } +} diff --git a/nodedb-vector/src/ivf/kmeans.rs b/nodedb-vector/src/ivf/kmeans.rs new file mode 100644 index 000000000..31fa432ec --- /dev/null +++ b/nodedb-vector/src/ivf/kmeans.rs @@ -0,0 +1,99 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! k-means++ seeding and Lloyd iterations for the IVF coarse quantizer. + +use crate::distance::{DistanceMetric, distance}; + +/// Up to `k` centroids for `data`, seeded by k-means++ from a fixed seed so +/// the same training set always yields the same centroids. +pub(crate) fn kmeans_centroids( + data: &[&[f32]], + dim: usize, + k: usize, + max_iter: usize, +) -> Vec> { + let n = data.len(); + let k = k.min(n); + if k == 0 { + return Vec::new(); + } + + let mut centroids: Vec> = vec![data[0].to_vec()]; + let mut min_dists = vec![f32::MAX; n]; + + // Initialize min_dists against the first centroid. + for (i, point) in data.iter().enumerate() { + let d = distance(point, ¢roids[0], DistanceMetric::L2); + if d < min_dists[i] { + min_dists[i] = d; + } + } + + let mut rng = crate::hnsw::Xorshift64::new(0xC0FF_EEDE_ADBE_EF42); + for _ in 1..k { + let total: f64 = min_dists.iter().map(|&d| d as f64).sum(); + let next_idx = if total < f64::EPSILON { + 0 + } else { + let target = rng.next_f64() * total; + let mut acc = 0.0f64; + let mut chosen = n - 1; + for (i, &d) in min_dists.iter().enumerate() { + acc += d as f64; + if acc >= target { + chosen = i; + break; + } + } + chosen + }; + let last = data[next_idx]; + centroids.push(last.to_vec()); + for (i, point) in data.iter().enumerate() { + let d = distance(point, last, DistanceMetric::L2); + if d < min_dists[i] { + min_dists[i] = d; + } + } + } + + let mut assignments = vec![0usize; n]; + for _ in 0..max_iter { + let mut changed = false; + for (i, point) in data.iter().enumerate() { + let mut best = 0; + let mut best_d = f32::MAX; + for (c, centroid) in centroids.iter().enumerate() { + let d = distance(point, centroid, DistanceMetric::L2); + if d < best_d { + best_d = d; + best = c; + } + } + if assignments[i] != best { + assignments[i] = best; + changed = true; + } + } + if !changed { + break; + } + let mut sums = vec![vec![0.0f32; dim]; k]; + let mut counts = vec![0usize; k]; + for (i, point) in data.iter().enumerate() { + let c = assignments[i]; + counts[c] += 1; + for d in 0..dim { + sums[c][d] += point[d]; + } + } + for c in 0..k { + if counts[c] > 0 { + for d in 0..dim { + centroids[c][d] = sums[c][d] / counts[c] as f32; + } + } + } + } + centroids +} diff --git a/nodedb-vector/src/ivf/mod.rs b/nodedb-vector/src/ivf/mod.rs new file mode 100644 index 000000000..503cee1cc --- /dev/null +++ b/nodedb-vector/src/ivf/mod.rs @@ -0,0 +1,10 @@ +// SPDX-License-Identifier: Apache-2.0 + +pub mod checkpoint; +pub mod index; +pub mod kmeans; +pub mod params; +pub mod search; + +pub use index::IvfPqIndex; +pub use params::IvfPqParams; diff --git a/nodedb-vector/src/ivf/params.rs b/nodedb-vector/src/ivf/params.rs new file mode 100644 index 000000000..e3d681464 --- /dev/null +++ b/nodedb-vector/src/ivf/params.rs @@ -0,0 +1,40 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! IVF-PQ index parameters. + +use crate::distance::DistanceMetric; + +/// IVF-PQ index configuration. +#[derive(Debug, Clone, zerompk::ToMessagePack, zerompk::FromMessagePack)] +pub struct IvfPqParams { + /// Number of Voronoi cells (partitions). Typical: sqrt(N). + pub n_cells: usize, + /// Number of PQ subvectors. Must divide dimension evenly. + pub pq_m: usize, + /// Centroids per PQ subvector (at most 256 for u8 codes). + pub pq_k: usize, + /// Number of cells to probe at query time. Higher = better recall. + pub nprobe: usize, + /// Distance metric. + pub metric: DistanceMetric, +} + +impl IvfPqParams { + /// Vectors an index needs before it can train: one per coarse cell and + /// one per PQ centroid, since both k-means runs need at least `k` points. + pub fn training_threshold(&self) -> usize { + self.n_cells.max(self.pq_k).max(1) + } +} + +impl Default for IvfPqParams { + fn default() -> Self { + Self { + n_cells: 256, + pq_m: 8, + pq_k: 256, + nprobe: 16, + metric: DistanceMetric::L2, + } + } +} diff --git a/nodedb-vector/src/ivf/search.rs b/nodedb-vector/src/ivf/search.rs new file mode 100644 index 000000000..789a0f948 --- /dev/null +++ b/nodedb-vector/src/ivf/search.rs @@ -0,0 +1,203 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! IVF-PQ search: probe the nearest cells, rank their entries by PQ +//! distance, and rerank the best of them by exact FP32 distance. + +use roaring::RoaringBitmap; + +use crate::distance::{DistanceMetric, distance}; +use crate::error::{VectorError, check_dim}; +use crate::hnsw::SearchResult; + +use super::index::{IvfPqIndex, residual}; + +/// PQ-ranked candidates reranked exactly, per result asked for. +const RERANK_PER_RESULT: usize = 3; +/// Fewest PQ-ranked candidates reranked exactly. +const MIN_RERANK: usize = 20; + +/// One PQ-ranked entry: id, PQ distance, cell, position in the cell. +struct Candidate { + id: u32, + pq_distance: f32, + cell: usize, + pos: usize, +} + +impl IvfPqIndex { + /// The `top_k` live entries nearest to `query` under the index metric. + /// + /// A query without the index dimension fails with + /// [`VectorError::DimensionMismatch`]. A distance table over the memory + /// budget fails the search: skipping its cell would drop results. + pub fn search(&self, query: &[f32], top_k: usize) -> Result, VectorError> { + self.search_with(query, top_k, self.params.metric, None) + } + + /// The `top_k` live entries nearest to `query` under `metric`, keeping + /// only ids in `filter` when one is given. Fails as [`Self::search`]. + pub fn search_with( + &self, + query: &[f32], + top_k: usize, + metric: DistanceMetric, + filter: Option<&RoaringBitmap>, + ) -> Result, VectorError> { + check_dim(self.dim, query.len())?; + let Some(pq) = &self.pq else { + return Ok(Vec::new()); + }; + if top_k == 0 || self.slots.is_empty() { + return Ok(Vec::new()); + } + + let mut probe: Vec<(usize, f32)> = self + .centroids + .iter() + .enumerate() + .map(|(i, c)| (i, distance(query, c, self.params.metric))) + .collect(); + probe.sort_by(|a, b| a.1.total_cmp(&b.1)); + probe.truncate(self.params.nprobe.max(1)); + + let m = pq.m; + let mut candidates: Vec = Vec::new(); + for &(cell_idx, _) in &probe { + let cell = &self.cells[cell_idx]; + if cell.ids.is_empty() { + continue; + } + let table = pq.build_distance_table(&residual(query, &self.centroids[cell_idx]))?; + for (pos, &id) in cell.ids.iter().enumerate() { + if self.deleted.contains(id) || filter.is_some_and(|f| !f.contains(id)) { + continue; + } + let code = &cell.codes[pos * m..(pos + 1) * m]; + candidates.push(Candidate { + id, + pq_distance: pq.asymmetric_distance(&table, code), + cell: cell_idx, + pos, + }); + } + } + + let pool = top_k.saturating_mul(RERANK_PER_RESULT).max(MIN_RERANK); + if candidates.len() > pool { + candidates.select_nth_unstable_by(pool, |a, b| a.pq_distance.total_cmp(&b.pq_distance)); + candidates.truncate(pool); + } + + let dim = self.dim; + let mut results: Vec = candidates + .into_iter() + .map(|c| { + let start = c.pos * dim; + let vector = &self.cells[c.cell].vectors[start..start + dim]; + SearchResult { + id: c.id, + distance: distance(query, vector, metric), + } + }) + .collect(); + results.sort_by(|a, b| a.distance.total_cmp(&b.distance)); + results.truncate(top_k); + Ok(results) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::ivf::IvfPqParams; + use crate::test_support::test_memory; + + fn make_vectors(n: usize, dim: usize) -> Vec> { + (0..n) + .map(|i| (0..dim).map(|d| ((i * dim + d) as f32) * 0.01).collect()) + .collect() + } + + fn index_with(vecs: &[Vec], params: IvfPqParams) -> IvfPqIndex { + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let mut idx = IvfPqIndex::new(vecs[0].len(), params); + idx.train(&refs, test_memory()).unwrap(); + idx.add_batch(&refs).unwrap(); + idx + } + + #[test] + fn train_and_search_finds_the_exact_match_first() { + let vecs = make_vectors(1000, 16); + let idx = index_with( + &vecs, + IvfPqParams { + n_cells: 32, + pq_m: 4, + pq_k: 32, + nprobe: 8, + metric: DistanceMetric::L2, + }, + ); + assert_eq!(idx.len(), 1000); + let results = idx.search(&vecs[500], 5).unwrap(); + assert_eq!(results.len(), 5); + assert_eq!(results[0].id, 500, "the exact rerank puts the match first"); + assert_eq!(results[0].distance, 0.0); + } + + #[test] + fn probing_every_cell_returns_every_live_entry_once() { + let vecs = make_vectors(64, 8); + let mut idx = index_with( + &vecs, + IvfPqParams { + n_cells: 4, + pq_m: 4, + pq_k: 8, + nprobe: 4, + metric: DistanceMetric::L2, + }, + ); + idx.delete(7); + let mut ids: Vec = idx + .search(&vecs[0], 100) + .unwrap() + .iter() + .map(|r| r.id) + .collect(); + ids.sort_unstable(); + let expected: Vec = (0..64).filter(|id| *id != 7).collect(); + assert_eq!(ids, expected); + } + + #[test] + fn a_filter_keeps_only_its_ids() { + let vecs = make_vectors(64, 8); + let idx = index_with( + &vecs, + IvfPqParams { + n_cells: 4, + pq_m: 4, + pq_k: 8, + nprobe: 4, + metric: DistanceMetric::L2, + }, + ); + let filter: RoaringBitmap = [3u32, 40, 41].into_iter().collect(); + let mut ids: Vec = idx + .search_with(&vecs[0], 10, DistanceMetric::L2, Some(&filter)) + .unwrap() + .iter() + .map(|r| r.id) + .collect(); + ids.sort_unstable(); + assert_eq!(ids, vec![3, 40, 41]); + } + + #[test] + fn an_untrained_index_finds_nothing() { + let idx = IvfPqIndex::new(8, IvfPqParams::default()); + assert!(idx.search(&[0.0; 8], 5).unwrap().is_empty()); + } +} diff --git a/nodedb-vector/src/lib.rs b/nodedb-vector/src/lib.rs index 8495537c1..519da2d53 100644 --- a/nodedb-vector/src/lib.rs +++ b/nodedb-vector/src/lib.rs @@ -75,7 +75,7 @@ pub use adaptive_filter::{ #[cfg(not(target_arch = "wasm32"))] pub use builder::{BuildSender, CompleteReceiver}; #[cfg(not(target_arch = "wasm32"))] -pub use collection::{BuildComplete, BuildRequest, StorageTier, VectorCollection}; +pub use collection::{BuildComplete, BuildKind, BuildRequest, StorageTier, VectorCollection}; pub use flat::FlatIndex; pub use index_config::{IndexConfig, IndexType}; pub use ivf::{IvfPqIndex, IvfPqParams}; diff --git a/nodedb-vector/src/matryoshka.rs b/nodedb-vector/src/matryoshka.rs index d581b38a0..760d7f0bc 100644 --- a/nodedb-vector/src/matryoshka.rs +++ b/nodedb-vector/src/matryoshka.rs @@ -22,6 +22,7 @@ use std::collections::BinaryHeap; use crate::distance::distance; +use crate::error::VectorError; use nodedb_types::vector_distance::DistanceMetric; /// Per-collection Matryoshka configuration. @@ -124,21 +125,39 @@ impl Ord for HeapEntry { /// by ascending full-dim distance. /// /// # Notes -/// - `candidates` yields `(id, full_dim_vector)` pairs. Vectors shorter than -/// `full_dim` are accepted; truncation clips to available length. +/// - `candidates` yields `(id, vector)` pairs; each vector carries at least +/// `full_dim` components, and components past `full_dim` are ignored. /// - When `coarse_dim == full_dim` the method degenerates to a single-pass /// top-k scan with one distance call per candidate (no duplicated work). +/// +/// # Errors +/// - [`VectorError::InvalidInput`] when `coarse_dim` exceeds `full_dim`. +/// - [`VectorError::DimensionMismatch`] when `query` has fewer than +/// `full_dim` components. +/// - [`VectorError::StoredDimensionMismatch`] when a candidate vector has +/// fewer than `full_dim` components. pub fn matryoshka_search<'a, I>( candidates: I, query: &[f32], options: &MatryoshkaSearchOptions, metric: DistanceMetric, -) -> Vec<(u32, f32)> +) -> Result, VectorError> where I: Iterator, { let coarse = options.coarse_dim as usize; let full = options.full_dim as usize; + if coarse > full { + return Err(VectorError::InvalidInput { + detail: format!("Matryoshka coarse dimension {coarse} exceeds full dimension {full}"), + }); + } + if query.len() < full { + return Err(VectorError::DimensionMismatch { + expected: full, + got: query.len(), + }); + } let pool_size = (options.oversample as usize).max(1) * options.k.max(1); let query_coarse = truncate(query, coarse); @@ -151,6 +170,12 @@ where let mut survivor_vecs: Vec> = Vec::with_capacity(pool_size); for (id, vec) in candidates { + if vec.len() < full { + return Err(VectorError::StoredDimensionMismatch { + expected: full, + got: vec.len(), + }); + } let vec_coarse = truncate(vec, coarse); let d = distance(query_coarse, vec_coarse, metric); @@ -163,7 +188,7 @@ where if should_insert { let vec_idx = survivor_vecs.len(); - survivor_vecs.push(vec[..full.min(vec.len())].to_vec()); + survivor_vecs.push(truncate(vec, full).to_vec()); coarse_heap.push(HeapEntry { dist: d, @@ -193,7 +218,7 @@ where reranked.sort_unstable_by(|a, b| a.1.partial_cmp(&b.1).unwrap_or(std::cmp::Ordering::Equal)); reranked.truncate(options.k); - reranked + Ok(reranked) } #[cfg(test)] @@ -288,7 +313,7 @@ mod tests { k: 10, }; - let results = matryoshka_search(candidates, &query, &opts, DistanceMetric::L2); + let results = matryoshka_search(candidates, &query, &opts, DistanceMetric::L2).unwrap(); assert_eq!(results.len(), 10, "expected exactly k=10 results"); } @@ -317,7 +342,7 @@ mod tests { oversample: 1, k: 10, }; - let mrl = matryoshka_search(candidates, &query, &opts, DistanceMetric::L2); + let mrl = matryoshka_search(candidates, &query, &opts, DistanceMetric::L2).unwrap(); // Same IDs in same order. let direct_ids: Vec = direct.iter().map(|(id, _)| *id).collect(); @@ -327,4 +352,51 @@ mod tests { "coarse==full should equal direct search" ); } + + #[test] + fn short_query_or_candidate_is_a_typed_error() { + let vecs = make_vecs(4, 8); + let opts = MatryoshkaSearchOptions { + coarse_dim: 4, + full_dim: 8, + oversample: 2, + k: 2, + }; + let candidates = vecs + .iter() + .enumerate() + .map(|(i, v)| (i as u32, v.as_slice())); + assert!(matches!( + matryoshka_search(candidates, &[0.0; 6], &opts, DistanceMetric::L2), + Err(VectorError::DimensionMismatch { + expected: 8, + got: 6 + }) + )); + + let short = [0.0_f32; 5]; + let candidates = std::iter::once((0_u32, &short[..])); + assert!(matches!( + matryoshka_search(candidates, &[0.0; 8], &opts, DistanceMetric::L2), + Err(VectorError::StoredDimensionMismatch { + expected: 8, + got: 5 + }) + )); + + let inverted = MatryoshkaSearchOptions { + coarse_dim: 16, + full_dim: 8, + oversample: 2, + k: 2, + }; + let candidates = vecs + .iter() + .enumerate() + .map(|(i, v)| (i as u32, v.as_slice())); + assert!(matches!( + matryoshka_search(candidates, &[0.0; 8], &inverted, DistanceMetric::L2), + Err(VectorError::InvalidInput { .. }) + )); + } } diff --git a/nodedb-vector/src/multivec/meta_embed.rs b/nodedb-vector/src/multivec/meta_embed.rs index c1ef5ea56..67e896e2c 100644 --- a/nodedb-vector/src/multivec/meta_embed.rs +++ b/nodedb-vector/src/multivec/meta_embed.rs @@ -17,6 +17,7 @@ use nodedb_types::vector_distance::DistanceMetric; use super::plaid::PlaidPruner; use super::scoring::budgeted_maxsim; use super::storage::MultiVectorStore; +use crate::error::{VectorError, check_dim}; /// Search a `MultiVectorStore` using budgeted MaxSim with optional PLAID /// candidate pruning. @@ -30,7 +31,9 @@ use super::storage::MultiVectorStore; /// * `metric` — distance metric (Cosine recommended for MetaEmbed). /// /// # Returns -/// A `Vec<(doc_id, score)>` sorted descending by score, length ≤ `k`. +/// A `Vec<(doc_id, score)>` sorted descending by score, length ≤ `k`, or +/// [`VectorError::DimensionMismatch`] when a query vector does not have the +/// store dimension. pub fn meta_embed_search( store: &MultiVectorStore, plaid: Option<&PlaidPruner>, @@ -38,9 +41,12 @@ pub fn meta_embed_search( budget: u8, k: usize, metric: DistanceMetric, -) -> Vec<(u32, f32)> { +) -> Result, VectorError> { + for v in query { + check_dim(store.dim, v.len())?; + } if k == 0 || query.is_empty() { - return Vec::new(); + return Ok(Vec::new()); } // Effective budget: 0 means use all query vectors. @@ -52,7 +58,7 @@ pub fn meta_embed_search( // Determine candidate set. let candidate_ids: Vec = match plaid { - Some(pruner) => pruner.candidates(query), + Some(pruner) => pruner.candidates(query)?, None => store.iter().map(|doc| doc.doc_id).collect(), }; @@ -70,7 +76,7 @@ pub fn meta_embed_search( // Sort descending by score. scored.sort_unstable_by(|a, b| b.1.partial_cmp(&a.1).unwrap_or(std::cmp::Ordering::Equal)); scored.truncate(k); - scored + Ok(scored) } // --------------------------------------------------------------------------- @@ -108,7 +114,8 @@ mod tests { fn search_returns_at_most_k_results() { let store = build_store(10, 4, 2); let query = vec![vec![1.0f32, 0.0, 0.0, 0.0]]; - let results = meta_embed_search(&store, None, &query, 2, 3, DistanceMetric::Cosine); + let results = + meta_embed_search(&store, None, &query, 2, 3, DistanceMetric::Cosine).unwrap(); assert!(results.len() <= 3); } @@ -116,7 +123,8 @@ mod tests { fn search_results_sorted_descending() { let store = build_store(8, 4, 2); let query = vec![vec![1.0f32, 0.0, 0.0, 0.0]]; - let results = meta_embed_search(&store, None, &query, 2, 8, DistanceMetric::Cosine); + let results = + meta_embed_search(&store, None, &query, 2, 8, DistanceMetric::Cosine).unwrap(); for w in results.windows(2) { assert!(w[0].1 >= w[1].1, "not sorted: {:?}", results); } @@ -130,9 +138,10 @@ mod tests { let pruner = PlaidPruner::train(&store, 3, 10, 99); let query = vec![vec![1.0f32, 0.0f32]]; - let unfiltered = meta_embed_search(&store, None, &query, 2, 9, DistanceMetric::Cosine); + let unfiltered = + meta_embed_search(&store, None, &query, 2, 9, DistanceMetric::Cosine).unwrap(); let filtered = - meta_embed_search(&store, Some(&pruner), &query, 2, 9, DistanceMetric::Cosine); + meta_embed_search(&store, Some(&pruner), &query, 2, 9, DistanceMetric::Cosine).unwrap(); let unfiltered_ids: std::collections::HashSet = unfiltered.iter().map(|(id, _)| *id).collect(); @@ -148,7 +157,7 @@ mod tests { #[test] fn search_empty_query_returns_empty() { let store = build_store(5, 4, 2); - let results = meta_embed_search(&store, None, &[], 2, 5, DistanceMetric::Cosine); + let results = meta_embed_search(&store, None, &[], 2, 5, DistanceMetric::Cosine).unwrap(); assert!(results.is_empty()); } @@ -156,7 +165,8 @@ mod tests { fn search_k_zero_returns_empty() { let store = build_store(5, 4, 2); let query = vec![vec![1.0f32, 0.0, 0.0, 0.0]]; - let results = meta_embed_search(&store, None, &query, 2, 0, DistanceMetric::Cosine); + let results = + meta_embed_search(&store, None, &query, 2, 0, DistanceMetric::Cosine).unwrap(); assert!(results.is_empty()); } @@ -166,8 +176,29 @@ mod tests { // Query is also in direction 0 — doc 0 should rank first. let store = build_store(4, 4, 1); let query = vec![vec![1.0f32, 0.0, 0.0, 0.0]]; - let results = meta_embed_search(&store, None, &query, 1, 1, DistanceMetric::Cosine); + let results = + meta_embed_search(&store, None, &query, 1, 1, DistanceMetric::Cosine).unwrap(); assert_eq!(results.len(), 1); assert_eq!(results[0].0, 0, "expected doc_id=0 to be top result"); } + + #[test] + fn wrong_dimension_query_is_a_typed_error() { + let store = build_store(4, 4, 2); + let pruner = PlaidPruner::train(&store, 2, 3, 7); + let query = vec![vec![1.0f32, 0.0, 0.0]]; + for plaid in [None, Some(&pruner)] { + let result = meta_embed_search(&store, plaid, &query, 1, 2, DistanceMetric::Cosine); + assert!( + matches!( + result, + Err(VectorError::DimensionMismatch { + expected: 4, + got: 3 + }) + ), + "{result:?}" + ); + } + } } diff --git a/nodedb-vector/src/multivec/plaid.rs b/nodedb-vector/src/multivec/plaid.rs index f2ac46fb7..7a49dc5ea 100644 --- a/nodedb-vector/src/multivec/plaid.rs +++ b/nodedb-vector/src/multivec/plaid.rs @@ -13,6 +13,7 @@ use std::collections::{HashMap, HashSet}; use crate::distance::scalar::scalar_distance; +use crate::error::{VectorError, check_dim}; use nodedb_types::vector_distance::DistanceMetric; use super::storage::MultiVectorStore; @@ -258,9 +259,18 @@ impl PlaidPruner { /// /// The query centroid bag is the set of nearest centroids for each query /// vector. - pub fn candidates(&self, query: &[Vec]) -> Vec { - if self.centroids.is_empty() || query.is_empty() { - return Vec::new(); + /// + /// A query vector without the centroid dimension fails with + /// [`VectorError::DimensionMismatch`]. + pub fn candidates(&self, query: &[Vec]) -> Result, VectorError> { + let Some(first) = self.centroids.first() else { + return Ok(Vec::new()); + }; + for v in query { + check_dim(first.len(), v.len())?; + } + if query.is_empty() { + return Ok(Vec::new()); } // Build query centroid bag. @@ -277,11 +287,12 @@ impl PlaidPruner { .collect(); // Collect docs that share at least one centroid with the query. - self.doc_centroids + Ok(self + .doc_centroids .iter() .filter(|(_, doc_ids)| doc_ids.iter().any(|id| query_bag.contains(id))) .map(|(&doc_id, _)| doc_id) - .collect() + .collect()) } } @@ -351,7 +362,7 @@ mod tests { // A query near cluster A should return at least some candidates. let query = vec![vec![0.0f32, 0.0f32]]; - let cands = pruner.candidates(&query); + let cands = pruner.candidates(&query).unwrap(); assert!(!cands.is_empty(), "expected at least one candidate"); } @@ -361,7 +372,7 @@ mod tests { let store = MultiVectorStore::new(2, MultiVecMode::PerToken); let pruner = PlaidPruner::train(&store, 3, 5, 1); let query = vec![vec![0.0f32, 0.0f32]]; - assert!(pruner.candidates(&query).is_empty()); + assert!(pruner.candidates(&query).unwrap().is_empty()); } #[test] @@ -375,7 +386,7 @@ mod tests { vec![10.0f32, 0.0f32], vec![0.0f32, 10.0f32], ]; - let mut cands = pruner.candidates(&query); + let mut cands = pruner.candidates(&query).unwrap(); cands.sort_unstable(); cands.dedup(); assert_eq!(cands.len(), 9, "all docs should be candidates: {:?}", cands); diff --git a/nodedb-vector/src/navix/acorn.rs b/nodedb-vector/src/navix/acorn.rs index 6ec7f8444..2ea98cc76 100644 --- a/nodedb-vector/src/navix/acorn.rs +++ b/nodedb-vector/src/navix/acorn.rs @@ -17,6 +17,7 @@ mod inner { use roaring::RoaringBitmap; use crate::distance::distance; + use crate::error::{VectorError, check_dim}; use crate::hnsw::graph::{Candidate, HnswIndex}; use crate::navix::traversal::SearchResult; @@ -43,26 +44,34 @@ mod inner { /// ACORN-1 filtered search. /// /// Uses static 2-hop expansion: when a 1-hop neighbor is not in `allowed`, - /// expand to its 2-hop neighbors unconditionally. + /// expand to its 2-hop neighbors unconditionally. A query without the + /// index dimension fails with [`VectorError::DimensionMismatch`]. pub fn acorn_search( index: &HnswIndex, query: &[f32], options: &AcornSearchOptions, metric: nodedb_types::vector_distance::DistanceMetric, - ) -> Vec { + ) -> Result, VectorError> { + check_dim(index.dim(), query.len())?; if index.is_empty() || options.allowed.is_empty() || options.k == 0 { - return Vec::new(); + return Ok(Vec::new()); } let total = index.len(); let global_sel = options.allowed.len() as f64 / total as f64; if global_sel < options.brute_force_threshold { - return brute_force_on_allowed(index, query, options.k, &options.allowed, metric); + return Ok(brute_force_on_allowed( + index, + query, + options.k, + &options.allowed, + metric, + )); } let Some(ep) = index.entry_point() else { - return Vec::new(); + return Ok(Vec::new()); }; // Phase 1: greedy descent (unfiltered) to find best layer-0 entry. @@ -78,14 +87,14 @@ mod inner { let ef = options.ef_search.max(options.k); let results = acorn_search_layer_0(index, query, current_ep, ef, &options.allowed, metric); - results + Ok(results .into_iter() .take(options.k) .map(|c| SearchResult { id: c.id, distance: c.dist, }) - .collect() + .collect()) } /// Minimal greedy single-layer descent used for Phase-1 layer navigation. @@ -298,7 +307,7 @@ mod inner { brute_force_threshold: 0.001, }; - let res = acorn_search(&idx, &query, &opts, DistanceMetric::L2); + let res = acorn_search(&idx, &query, &opts, DistanceMetric::L2).unwrap(); assert!(!res.is_empty()); for r in &res { assert!( @@ -309,6 +318,24 @@ mod inner { } } + #[test] + fn acorn_wrong_dimension_query_is_a_typed_error() { + let idx = build_index(20); + let opts = AcornSearchOptions { + k: 3, + ef_search: 64, + allowed: (0..20u32).collect(), + brute_force_threshold: 0.001, + }; + assert!(matches!( + acorn_search(&idx, &[1.0], &opts, DistanceMetric::L2), + Err(VectorError::DimensionMismatch { + expected: 3, + got: 1 + }) + )); + } + /// Very low selectivity (1 ID out of 20) — result must be that single ID. #[test] fn acorn_single_allowed_id() { @@ -325,7 +352,7 @@ mod inner { brute_force_threshold: 0.001, }; - let res = acorn_search(&idx, &query, &opts, DistanceMetric::L2); + let res = acorn_search(&idx, &query, &opts, DistanceMetric::L2).unwrap(); assert!(res.len() <= 1); if let Some(r) = res.first() { assert_eq!(r.id, 7); diff --git a/nodedb-vector/src/navix/traversal.rs b/nodedb-vector/src/navix/traversal.rs index 257d68648..016ad8b9d 100644 --- a/nodedb-vector/src/navix/traversal.rs +++ b/nodedb-vector/src/navix/traversal.rs @@ -15,6 +15,7 @@ use std::collections::{BinaryHeap, HashSet}; use roaring::RoaringBitmap; use crate::distance::distance; +use crate::error::{VectorError, check_dim}; use crate::hnsw::graph::{Candidate, HnswIndex}; use crate::navix::selectivity::{NavixHeuristic, local_selectivity_at, pick_heuristic}; @@ -59,28 +60,38 @@ impl Default for NavixSearchOptions { /// Returns up to `options.k` nearest vectors from `index` to `query`, where /// candidate IDs must be present in `options.allowed`. /// +/// Returns an empty Vec when the index is empty or `options.allowed` is empty. +/// /// # Errors /// -/// Returns an empty Vec when the index is empty or `options.allowed` is empty. +/// [`VectorError::DimensionMismatch`] when `query` does not have the index +/// dimension. pub fn navix_search( index: &HnswIndex, query: &[f32], options: &NavixSearchOptions, metric: nodedb_types::vector_distance::DistanceMetric, -) -> Vec { +) -> Result, VectorError> { + check_dim(index.dim(), query.len())?; if index.is_empty() || options.allowed.is_empty() || options.k == 0 { - return Vec::new(); + return Ok(Vec::new()); } let total = index.len(); let global_sel = options.allowed.len() as f64 / total as f64; if global_sel < options.brute_force_threshold { - return brute_force_on_allowed(index, query, options.k, &options.allowed, metric); + return Ok(brute_force_on_allowed( + index, + query, + options.k, + &options.allowed, + metric, + )); } let Some(ep) = index.entry_point() else { - return Vec::new(); + return Ok(Vec::new()); }; // Phase 1: greedy descent from max_layer to layer 1 (unfiltered, as in @@ -97,14 +108,14 @@ pub fn navix_search( let ef = options.ef_search.max(options.k); let results = navix_search_layer_0(index, query, current_ep, ef, &options.allowed, metric); - results + Ok(results .into_iter() .take(options.k) .map(|c| SearchResult { id: c.id, distance: c.dist, }) - .collect() + .collect()) } // ── Internal helpers ────────────────────────────────────────────────────────── @@ -513,8 +524,8 @@ mod tests { brute_force_threshold: 0.001, }; - let navix_res = navix_search(&idx, &query, &opts, DistanceMetric::L2); - let hnsw_res = idx.search(&query, 5, 64); + let navix_res = navix_search(&idx, &query, &opts, DistanceMetric::L2).unwrap(); + let hnsw_res = idx.search(&query, 5, 64).unwrap(); assert!(!navix_res.is_empty()); // The best result should be id=10 (exact match) in both cases. @@ -536,7 +547,7 @@ mod tests { brute_force_threshold: 0.001, }; - let res = navix_search(&idx, &query, &opts, DistanceMetric::L2); + let res = navix_search(&idx, &query, &opts, DistanceMetric::L2).unwrap(); // With only one allowed ID, we get at most 1 result. assert!(res.len() <= 1); if let Some(r) = res.first() { @@ -562,7 +573,7 @@ mod tests { brute_force_threshold: 0.001, }; - let res = navix_search(&idx, &query, &opts, DistanceMetric::L2); + let res = navix_search(&idx, &query, &opts, DistanceMetric::L2).unwrap(); assert!(!res.is_empty()); for r in &res { assert!( @@ -593,7 +604,7 @@ mod tests { brute_force_threshold: 0.5, }; - let res = navix_search(&idx, &query, &opts, DistanceMetric::L2); + let res = navix_search(&idx, &query, &opts, DistanceMetric::L2).unwrap(); // Manual brute-force reference. let mut manual: Vec<(u32, f32)> = allowed @@ -634,7 +645,7 @@ mod tests { allowed, brute_force_threshold: 0.001, }; - let res = navix_search(&idx, &[1.0, 0.0, 0.0], &opts, DistanceMetric::L2); + let res = navix_search(&idx, &[1.0, 0.0, 0.0], &opts, DistanceMetric::L2).unwrap(); assert!(res.is_empty()); } @@ -648,7 +659,28 @@ mod tests { allowed: RoaringBitmap::new(), brute_force_threshold: 0.001, }; - let res = navix_search(&idx, &[5.0, 0.0, 0.0], &opts, DistanceMetric::L2); + let res = navix_search(&idx, &[5.0, 0.0, 0.0], &opts, DistanceMetric::L2).unwrap(); assert!(res.is_empty()); } + + #[test] + fn wrong_dimension_query_is_a_typed_error() { + let idx = build_index(20); + let opts = NavixSearchOptions { + k: 3, + allowed: (0..20u32).collect(), + ..NavixSearchOptions::default() + }; + let result = navix_search(&idx, &[1.0, 0.0], &opts, DistanceMetric::L2); + assert!( + matches!( + result, + Err(crate::error::VectorError::DimensionMismatch { + expected: 3, + got: 2 + }) + ), + "{result:?}" + ); + } } diff --git a/nodedb-vector/src/quantize/mod.rs b/nodedb-vector/src/quantize/mod.rs index 2bd207f56..0a2e03d5b 100644 --- a/nodedb-vector/src/quantize/mod.rs +++ b/nodedb-vector/src/quantize/mod.rs @@ -5,6 +5,7 @@ pub mod binary_codec; pub mod pq; pub mod pq_decode; +pub mod pq_kmeans; pub mod pq_codec; diff --git a/nodedb-vector/src/quantize/pq.rs b/nodedb-vector/src/quantize/pq.rs index dd143bfad..ed8a57a90 100644 --- a/nodedb-vector/src/quantize/pq.rs +++ b/nodedb-vector/src/quantize/pq.rs @@ -19,7 +19,9 @@ use std::mem::size_of; use nodedb_mem::{ReservationToken, ScopedMemory}; use nodedb_types::decode_bounds::checked_decode_capacity; -use crate::error::VectorError; +use crate::error::{VectorError, check_dim}; + +use super::pq_kmeans::{kmeans, l2_sub}; /// Hard ceiling for a decoded PQ vector. This bounds corrupted persisted /// configuration even when the codec has no scoped memory handle attached. @@ -99,24 +101,39 @@ impl PqCodec { k: usize, max_iter: usize, memory: ScopedMemory, - ) -> Self { - assert!(!vectors.is_empty()); - assert!( - dim > 0 - && dim <= MAX_PQ_DECODE_DIM - && m > 0 - && k > 0 - && k <= usize::from(u8::MAX) + 1 - && k <= vectors.len() - ); - assert!( - dim.is_multiple_of(m), - "dim ({dim}) must be divisible by m ({m})" - ); - + ) -> Result { + let invalid = |detail: String| Err(VectorError::InvalidInput { detail }); + if vectors.is_empty() { + return invalid("PQ training needs at least one vector".into()); + } + if dim == 0 || dim > MAX_PQ_DECODE_DIM { + return invalid(format!( + "PQ dimension {dim} is outside 1..={MAX_PQ_DECODE_DIM}" + )); + } + if m == 0 || !dim.is_multiple_of(m) { + return invalid(format!("PQ dimension {dim} must be divisible by m ({m})")); + } + if k == 0 || k > usize::from(u8::MAX) + 1 || k > vectors.len() { + return invalid(format!( + "PQ centroid count {k} must be in 1..=256 and at most the {} training vectors", + vectors.len() + )); + } let sub_dim = dim / m; - let codebook_bytes = pq_codebook_allocation_bytes(m, k, sub_dim); - assert!(codebook_bytes.is_some_and(|bytes| bytes <= MAX_PQ_CODEBOOK_BYTES)); + if pq_codebook_allocation_bytes(m, k, sub_dim) + .is_none_or(|bytes| bytes > MAX_PQ_CODEBOOK_BYTES) + { + return invalid(format!( + "PQ codebook for m={m}, k={k}, sub-dimension {sub_dim} exceeds \ + {MAX_PQ_CODEBOOK_BYTES} bytes" + )); + } + // Parameters are checked before any vector is read. + for v in vectors { + check_dim(dim, v.len())?; + } + let mut codebooks = Vec::with_capacity(m); for sub in 0..m { @@ -131,14 +148,14 @@ impl PqCodec { codebooks.push(centroids); } - Self { + Ok(Self { dim, m, k, sub_dim, codebooks, memory, - } + }) } /// Encode a vector: for each subvector, find the nearest centroid index. @@ -147,8 +164,11 @@ impl PqCodec { /// intentionally skipped here to avoid atomic overhead on every candidate /// during search; use [`encode_batch`] for bulk encoding with budget /// enforcement. + /// + /// Precondition: `vector.len() == self.dim`. This per-candidate hot path + /// does not re-check it; every caller checks the dimension first + /// (`encode_batch`, the IVF-PQ `add`, and the codec-index entry points). pub fn encode(&self, vector: &[f32]) -> Vec { - debug_assert_eq!(vector.len(), self.dim); let mut code = Vec::with_capacity(self.m); for sub in 0..self.m { let offset = sub * self.sub_dim; @@ -165,6 +185,9 @@ impl PqCodec { /// before allocating the output buffer. The guard is released at /// the end of this call — the buffer itself remains alive. pub fn encode_batch(&self, vectors: &[&[f32]]) -> Result, VectorError> { + for v in vectors { + check_dim(self.dim, v.len())?; + } let capacity = self.m * vectors.len(); let _g = try_reserve_or_skip(&self.memory, capacity * size_of::())?; let mut out = Vec::with_capacity(capacity); @@ -183,7 +206,7 @@ impl PqCodec { /// Charges `m * k * size_of::()` bytes to the bound budget (if set) /// before allocating the table. pub fn build_distance_table(&self, query: &[f32]) -> Result>, VectorError> { - debug_assert_eq!(query.len(), self.dim); + check_dim(self.dim, query.len())?; let total_bytes = self.m * self.k * size_of::(); let _g = try_reserve_or_skip(&self.memory, total_bytes)?; let mut table = Vec::with_capacity(self.m); @@ -223,12 +246,12 @@ impl PqCodec { .m .checked_mul(self.sub_dim) .filter(|&value| value == self.dim && value <= MAX_PQ_DECODE_DIM) - .ok_or(VectorError::DimensionMismatch { + .ok_or(VectorError::StoredDimensionMismatch { expected: self.dim, got: 0, })?; if code.len() != self.m { - return Err(VectorError::DimensionMismatch { + return Err(VectorError::StoredDimensionMismatch { expected: self.m, got: code.len(), }); @@ -246,7 +269,7 @@ impl PqCodec { MAX_PQ_DECODE_DIM, MAX_PQ_DECODE_DIM * size_of::(), ) - .ok_or(VectorError::DimensionMismatch { + .ok_or(VectorError::StoredDimensionMismatch { expected: self.dim, got: 0, })?; @@ -279,7 +302,11 @@ impl PqCodec { sub_dim: self.sub_dim, codebooks: self.codebooks.clone(), }; - let payload = zerompk::to_msgpack_vec(&data).unwrap_or_default(); + let payload = zerompk::to_msgpack_vec(&data).map_err(|e| { + VectorError::CheckpointSerializationError { + detail: format!("PQ codec encode: {e}"), + } + })?; let mut out = Vec::with_capacity(7 + payload.len()); out.extend_from_slice(MAGIC); out.push(VERSION); @@ -366,123 +393,53 @@ impl PqCodec { } } -/// L2 squared distance for sub-vectors (used in k-means and encoding). -#[inline] -fn l2_sub(a: &[f32], b: &[f32]) -> f32 { - let mut sum = 0.0f32; - for i in 0..a.len() { - let d = a[i] - b[i]; - sum += d * d; - } - sum -} - -/// Simple k-means clustering for PQ codebook training. -/// -/// Uses proper k-means++ initialization (weighted d² sampling) with a -/// deterministic seed so training is reproducible across runs. -fn kmeans(data: &[&[f32]], dim: usize, k: usize, max_iter: usize) -> Vec> { - let n = data.len(); - if n == 0 || k == 0 { - return Vec::new(); - } - let k = k.min(n); // Can't have more centroids than data points. - - // K-means++ initialization with deterministic xorshift. - let mut rng = crate::hnsw::Xorshift64::new(0xC0FF_EEDE_ADBE_EF42); - - let mut centroids: Vec> = Vec::with_capacity(k); - centroids.push(data[0].to_vec()); +#[cfg(test)] +mod tests { + use super::*; + use crate::test_support::test_memory; - let mut min_dists = vec![f32::MAX; n]; - // Update against the first centroid. - for (i, point) in data.iter().enumerate() { - let d = l2_sub(point, ¢roids[0]); - if d < min_dists[i] { - min_dists[i] = d; - } + fn assert_invalid(result: Result) { + assert!( + matches!(result, Err(VectorError::InvalidInput { .. })), + "expected InvalidInput" + ); } - for _ in 1..k { - let total: f64 = min_dists.iter().map(|&d| d as f64).sum(); - let next_idx = if total < f64::EPSILON { - // All points coincide with existing centroids. - 0 - } else { - let target = rng.next_f64() * total; - let mut acc = 0.0f64; - let mut chosen = n - 1; - for (i, &d) in min_dists.iter().enumerate() { - acc += d as f64; - if acc >= target { - chosen = i; - break; - } - } - chosen - }; - centroids.push(data[next_idx].to_vec()); - // Incrementally update min_dists against the new centroid. - let last = centroids.last().expect("just pushed"); - for (i, point) in data.iter().enumerate() { - let d = l2_sub(point, last); - if d < min_dists[i] { - min_dists[i] = d; - } - } + #[test] + fn train_rejects_a_vector_of_the_wrong_dimension() { + let good = [0.0_f32, 1.0]; + let short = [0.0_f32]; + let result = PqCodec::train(&[&good, &short], 2, 1, 1, 1, test_memory()); + assert!(matches!( + result, + Err(VectorError::DimensionMismatch { + expected: 2, + got: 1 + }) + )); } - // K-means iterations. - let mut assignments = vec![0usize; n]; - for _ in 0..max_iter { - // Assignment step. - let mut changed = false; - for (i, point) in data.iter().enumerate() { - let mut best = 0; - let mut best_d = f32::MAX; - for (c, centroid) in centroids.iter().enumerate() { - let d = l2_sub(point, centroid); - if d < best_d { - best_d = d; - best = c; - } - } - if assignments[i] != best { - assignments[i] = best; - changed = true; - } - } - if !changed { - break; - } - - // Update step: recompute centroids as means. - let mut sums = vec![vec![0.0f32; dim]; k]; - let mut counts = vec![0usize; k]; - for (i, point) in data.iter().enumerate() { - let c = assignments[i]; - counts[c] += 1; - for d in 0..dim { - sums[c][d] += point[d]; - } - } - for c in 0..k { - if counts[c] > 0 { - for d in 0..dim { - centroids[c][d] = sums[c][d] / counts[c] as f32; - } - } - } + #[test] + fn distance_table_rejects_a_query_of_the_wrong_dimension() { + let a = [0.0_f32, 1.0]; + let b = [1.0_f32, 0.0]; + let codec = PqCodec::train(&[&a, &b], 2, 1, 2, 1, test_memory()).unwrap(); + assert!(matches!( + codec.build_distance_table(&[1.0]), + Err(VectorError::DimensionMismatch { + expected: 2, + got: 1 + }) + )); + assert!(matches!( + codec.encode_batch(&[&[1.0]]), + Err(VectorError::DimensionMismatch { + expected: 2, + got: 1 + }) + )); } - centroids -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::test_support::test_memory; - fn make_clustered_data() -> Vec> { // 4 clusters in 4D space, 50 points each. let mut vecs = Vec::new(); @@ -501,40 +458,49 @@ mod tests { } #[test] - #[should_panic] fn train_rejects_dimension_above_decode_limit() { let vector = [0.0]; - PqCodec::train(&[&vector], MAX_PQ_DECODE_DIM + 1, 1, 1, 1, test_memory()); + assert_invalid(PqCodec::train( + &[&vector], + MAX_PQ_DECODE_DIM + 1, + 1, + 1, + 1, + test_memory(), + )); } #[test] - #[should_panic] fn train_rejects_raw_64_mib_codebook_once_container_overhead_is_counted() { let vector = [0.0]; let vectors = vec![vector.as_slice(); 256]; - PqCodec::train(&vectors, 65_536, 1, 256, 1, test_memory()); + assert_invalid(PqCodec::train(&vectors, 65_536, 1, 256, 1, test_memory())); } #[test] - #[should_panic] fn train_rejects_codebook_above_decode_limit() { let vector = [0.0]; let vectors = vec![vector.as_slice(); 17]; - PqCodec::train(&vectors, MAX_PQ_DECODE_DIM, 1, 17, 1, test_memory()); + assert_invalid(PqCodec::train( + &vectors, + MAX_PQ_DECODE_DIM, + 1, + 17, + 1, + test_memory(), + )); } #[test] - #[should_panic] fn train_rejects_more_centroids_than_training_vectors() { let vector = [0.0]; - PqCodec::train(&[&vector], 1, 1, 2, 1, test_memory()); + assert_invalid(PqCodec::train(&[&vector], 1, 1, 2, 1, test_memory())); } #[test] - #[should_panic] fn train_rejects_centroid_count_above_u8_encoding_range() { let vector = [0.0]; - PqCodec::train(&[&vector], 1, 1, 257, 1, test_memory()); + assert_invalid(PqCodec::train(&[&vector], 1, 1, 257, 1, test_memory())); } #[test] @@ -602,7 +568,7 @@ mod tests { fn encode_decode_roundtrip() { let vecs = make_clustered_data(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = PqCodec::train(&refs, 4, 2, 16, 10, test_memory()); + let codec = PqCodec::train(&refs, 4, 2, 16, 10, test_memory()).unwrap(); for v in &vecs { let code = codec.encode(v); @@ -616,7 +582,7 @@ mod tests { fn distance_table_gives_correct_ordering() { let vecs = make_clustered_data(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = PqCodec::train(&refs, 4, 2, 16, 10, test_memory()); + let codec = PqCodec::train(&refs, 4, 2, 16, 10, test_memory()).unwrap(); let codes: Vec> = vecs.iter().map(|v| codec.encode(v)).collect(); let query = &[5.0, 5.0, 5.0, 5.0]; @@ -650,7 +616,7 @@ mod tests { fn batch_encode() { let vecs = make_clustered_data(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = PqCodec::train(&refs, 4, 2, 16, 10, test_memory()); + let codec = PqCodec::train(&refs, 4, 2, 16, 10, test_memory()).unwrap(); let batch = codec.encode_batch(&refs).unwrap(); assert_eq!(batch.len(), 2 * 200); // M=2, N=200 @@ -661,7 +627,7 @@ mod tests { fn pq_codec_golden_format() { let vecs = make_clustered_data(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = PqCodec::train(&refs, 4, 2, 16, 10, test_memory()); + let codec = PqCodec::train(&refs, 4, 2, 16, 10, test_memory()).unwrap(); let bytes = codec.to_bytes().unwrap(); diff --git a/nodedb-vector/src/quantize/pq_codec.rs b/nodedb-vector/src/quantize/pq_codec.rs index 16b505091..c526aa30c 100644 --- a/nodedb-vector/src/quantize/pq_codec.rs +++ b/nodedb-vector/src/quantize/pq_codec.rs @@ -166,7 +166,7 @@ mod tests { }) .collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - PqCodec::train(&refs, 4, 2, 8, 10, test_memory()) + PqCodec::train(&refs, 4, 2, 8, 10, test_memory()).unwrap() } /// `encode` round-trip: packed_bits in the UQV must match the raw diff --git a/nodedb-vector/src/quantize/pq_kmeans.rs b/nodedb-vector/src/quantize/pq_kmeans.rs new file mode 100644 index 000000000..6e6d8a5d6 --- /dev/null +++ b/nodedb-vector/src/quantize/pq_kmeans.rs @@ -0,0 +1,115 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Sub-vector distance and k-means for PQ codebook training. + +/// L2 squared distance for sub-vectors (used in k-means and encoding). +#[inline] +pub(crate) fn l2_sub(a: &[f32], b: &[f32]) -> f32 { + let mut sum = 0.0f32; + for i in 0..a.len() { + let d = a[i] - b[i]; + sum += d * d; + } + sum +} + +/// Simple k-means clustering for PQ codebook training. +/// +/// Uses proper k-means++ initialization (weighted d² sampling) with a +/// deterministic seed so training is reproducible across runs. +pub(crate) fn kmeans(data: &[&[f32]], dim: usize, k: usize, max_iter: usize) -> Vec> { + let n = data.len(); + if n == 0 || k == 0 { + return Vec::new(); + } + let k = k.min(n); // Can't have more centroids than data points. + + // K-means++ initialization with deterministic xorshift. + let mut rng = crate::hnsw::Xorshift64::new(0xC0FF_EEDE_ADBE_EF42); + + let mut centroids: Vec> = Vec::with_capacity(k); + centroids.push(data[0].to_vec()); + + let mut min_dists = vec![f32::MAX; n]; + // Update against the first centroid. + for (i, point) in data.iter().enumerate() { + let d = l2_sub(point, ¢roids[0]); + if d < min_dists[i] { + min_dists[i] = d; + } + } + + for _ in 1..k { + let total: f64 = min_dists.iter().map(|&d| d as f64).sum(); + let next_idx = if total < f64::EPSILON { + // All points coincide with existing centroids. + 0 + } else { + let target = rng.next_f64() * total; + let mut acc = 0.0f64; + let mut chosen = n - 1; + for (i, &d) in min_dists.iter().enumerate() { + acc += d as f64; + if acc >= target { + chosen = i; + break; + } + } + chosen + }; + let last = data[next_idx]; + centroids.push(last.to_vec()); + // Incrementally update min_dists against the new centroid. + for (i, point) in data.iter().enumerate() { + let d = l2_sub(point, last); + if d < min_dists[i] { + min_dists[i] = d; + } + } + } + + // K-means iterations. + let mut assignments = vec![0usize; n]; + for _ in 0..max_iter { + // Assignment step. + let mut changed = false; + for (i, point) in data.iter().enumerate() { + let mut best = 0; + let mut best_d = f32::MAX; + for (c, centroid) in centroids.iter().enumerate() { + let d = l2_sub(point, centroid); + if d < best_d { + best_d = d; + best = c; + } + } + if assignments[i] != best { + assignments[i] = best; + changed = true; + } + } + if !changed { + break; + } + + // Update step: recompute centroids as means. + let mut sums = vec![vec![0.0f32; dim]; k]; + let mut counts = vec![0usize; k]; + for (i, point) in data.iter().enumerate() { + let c = assignments[i]; + counts[c] += 1; + for d in 0..dim { + sums[c][d] += point[d]; + } + } + for c in 0..k { + if counts[c] > 0 { + for d in 0..dim { + centroids[c][d] = sums[c][d] / counts[c] as f32; + } + } + } + } + + centroids +} diff --git a/nodedb-vector/src/quantize/sq8.rs b/nodedb-vector/src/quantize/sq8.rs index d54a2e6e9..b2e10c4d0 100644 --- a/nodedb-vector/src/quantize/sq8.rs +++ b/nodedb-vector/src/quantize/sq8.rs @@ -15,7 +15,7 @@ use serde::{Deserialize, Serialize}; -use crate::error::VectorError; +use crate::error::{VectorError, check_dim}; /// Magic bytes identifying a serialized [`Sq8Codec`] blob. /// @@ -47,15 +47,29 @@ impl Sq8Codec { /// At least 1000 vectors recommended for stable calibration; /// for fewer vectors the bounds may be tight, causing clipping /// on future inserts outside the calibration range. - pub fn calibrate(vectors: &[&[f32]], dim: usize) -> Self { - assert!(!vectors.is_empty(), "cannot calibrate on empty set"); - assert!(dim > 0); + /// + /// An empty set or a zero `dim` fails with [`VectorError::InvalidInput`]; + /// a vector without `dim` components fails with + /// [`VectorError::DimensionMismatch`]. + pub fn calibrate(vectors: &[&[f32]], dim: usize) -> Result { + if vectors.is_empty() { + return Err(VectorError::InvalidInput { + detail: "SQ8 calibration needs at least one vector".into(), + }); + } + if dim == 0 { + return Err(VectorError::InvalidInput { + detail: "SQ8 calibration needs a non-zero dimension".into(), + }); + } + for v in vectors { + check_dim(dim, v.len())?; + } let mut mins = vec![f32::MAX; dim]; let mut maxs = vec![f32::MIN; dim]; for v in vectors { - debug_assert_eq!(v.len(), dim); for d in 0..dim { if v[d] < mins[d] { mins[d] = v[d]; @@ -76,12 +90,24 @@ impl Sq8Codec { } } - Self { + Ok(Self { dim, mins, maxs, scales, inv_scales, + }) + } + + /// A codec over the unit range `[0, 1]` on every dimension, for + /// normalized embeddings before any calibration data exists. + pub fn unit_range(dim: usize) -> Self { + Self { + dim, + mins: vec![0.0; dim], + maxs: vec![1.0; dim], + scales: vec![1.0 / 255.0; dim], + inv_scales: vec![255.0; dim], } } @@ -223,7 +249,7 @@ mod tests { .map(|i| vec![i as f32 * 0.1, (i as f32).sin(), (i as f32).cos()]) .collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - Sq8Codec::calibrate(&refs, 3) + Sq8Codec::calibrate(&refs, 3).unwrap() } #[test] @@ -284,7 +310,7 @@ mod tests { fn quantize_dequantize_roundtrip() { let vecs = make_vectors(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = Sq8Codec::calibrate(&refs, 3); + let codec = Sq8Codec::calibrate(&refs, 3).unwrap(); for v in &vecs { let q = codec.quantize(v); @@ -306,7 +332,7 @@ mod tests { fn asymmetric_l2_close_to_exact() { let vecs = make_vectors(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = Sq8Codec::calibrate(&refs, 3); + let codec = Sq8Codec::calibrate(&refs, 3).unwrap(); let query = &[5.0, 0.5, -0.5]; for v in &vecs { @@ -330,7 +356,7 @@ mod tests { fn batch_quantize() { let vecs = make_vectors(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = Sq8Codec::calibrate(&refs, 3); + let codec = Sq8Codec::calibrate(&refs, 3).unwrap(); let batch = codec.quantize_batch(&refs); assert_eq!(batch.len(), 3 * 100); @@ -345,10 +371,31 @@ mod tests { // All vectors have the same value in dimension 0. let vecs: Vec> = (0..10).map(|i| vec![5.0, i as f32]).collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = Sq8Codec::calibrate(&refs, 2); + let codec = Sq8Codec::calibrate(&refs, 2).unwrap(); // Constant dimension should quantize to 0 without NaN/inf. let q = codec.quantize(&[5.0, 3.0]); assert_eq!(q[0], 0); // constant dim } + + #[test] + fn calibrate_rejects_bad_input_with_typed_errors() { + use crate::error::VectorError; + assert!(matches!( + Sq8Codec::calibrate(&[], 3), + Err(VectorError::InvalidInput { .. }) + )); + let v = [1.0_f32, 2.0]; + assert!(matches!( + Sq8Codec::calibrate(&[&v], 0), + Err(VectorError::InvalidInput { .. }) + )); + assert!(matches!( + Sq8Codec::calibrate(&[&v], 3), + Err(VectorError::DimensionMismatch { + expected: 3, + got: 2 + }) + )); + } } diff --git a/nodedb-vector/src/quantize/sq8_codec.rs b/nodedb-vector/src/quantize/sq8_codec.rs index 914fa624e..d9766d251 100644 --- a/nodedb-vector/src/quantize/sq8_codec.rs +++ b/nodedb-vector/src/quantize/sq8_codec.rs @@ -118,7 +118,7 @@ mod tests { .map(|i| vec![i as f32 * 0.1, -(i as f32) * 0.05, 1.0 + i as f32 * 0.02]) .collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - Sq8Codec::calibrate(&refs, 3) + Sq8Codec::calibrate(&refs, 3).unwrap() } /// `encode` round-trip: packed_bits in the UQV must match the raw diff --git a/nodedb-vector/src/rerank/codecs/pq.rs b/nodedb-vector/src/rerank/codecs/pq.rs index f2a3c3a60..8be53356b 100644 --- a/nodedb-vector/src/rerank/codecs/pq.rs +++ b/nodedb-vector/src/rerank/codecs/pq.rs @@ -227,7 +227,8 @@ impl RerankCodec for PqRerank { self.k, self.max_iter, self.memory.clone(), - ); + ) + .map_err(|e| RerankError::BadInput(format!("pq train: {e}")))?; self.codec = Some(codec); Ok(()) } diff --git a/nodedb-vector/src/rerank/codecs/sq8.rs b/nodedb-vector/src/rerank/codecs/sq8.rs index 5b615607d..53c1e27ff 100644 --- a/nodedb-vector/src/rerank/codecs/sq8.rs +++ b/nodedb-vector/src/rerank/codecs/sq8.rs @@ -40,13 +40,12 @@ impl Sq8Rerank { /// which is suitable for normalized embeddings. For best accuracy call /// `train()` with representative samples before encoding. pub fn new(dim: usize) -> Self { - // Build a minimal calibration over the unit range so encoding is - // functional before train() is called. - let lo = vec![0.0f32; dim]; - let hi = vec![1.0f32; dim]; - let samples: Vec<&[f32]> = vec![lo.as_slice(), hi.as_slice()]; - let codec = Sq8Codec::calibrate(&samples, dim); - Self { codec, dim } + // Calibrated over the unit range so encoding is functional before + // train() is called. + Self { + codec: Sq8Codec::unit_range(dim), + dim, + } } /// Wrap an already-trained `Sq8Codec`. @@ -133,7 +132,8 @@ impl RerankCodec for Sq8Rerank { "sq8 train: empty sample set".to_string(), )); } - self.codec = Sq8Codec::calibrate(samples, self.dim); + self.codec = Sq8Codec::calibrate(samples, self.dim) + .map_err(|e| RerankError::BadInput(format!("sq8 train: {e}")))?; Ok(()) } } diff --git a/nodedb-vector/src/sieve/collection.rs b/nodedb-vector/src/sieve/collection.rs index 3daf70866..ea7c10c9b 100644 --- a/nodedb-vector/src/sieve/collection.rs +++ b/nodedb-vector/src/sieve/collection.rs @@ -148,7 +148,7 @@ mod tests { .expect("build"); let idx = coll.get(&"tenant_id=1".to_string()).unwrap(); - let results = idx.search(&[2.0, 2.0, 2.0], 2, 32); + let results = idx.search(&[2.0, 2.0, 2.0], 2, 32).unwrap(); assert!(!results.is_empty()); } } diff --git a/nodedb-vector/src/sieve/router.rs b/nodedb-vector/src/sieve/router.rs index 063541bde..c3a905e12 100644 --- a/nodedb-vector/src/sieve/router.rs +++ b/nodedb-vector/src/sieve/router.rs @@ -39,7 +39,9 @@ impl<'a> SieveRouter<'a> { /// /// # Returns /// - /// Up to `k` nearest-neighbour results, sorted by ascending distance. + /// Up to `k` nearest-neighbour results, sorted by ascending distance, or + /// [`VectorError::DimensionMismatch`](crate::error::VectorError::DimensionMismatch) + /// when `query` does not have the index dimension. pub fn route( &self, query: &[f32], @@ -48,7 +50,7 @@ impl<'a> SieveRouter<'a> { k: usize, ef_search: usize, metric: DistanceMetric, - ) -> Vec { + ) -> Result, crate::error::VectorError> { // Fast path: subindex hit. if let Some(sig) = predicate_signature && let Some(subindex) = self.collection.get(sig) @@ -63,13 +65,13 @@ impl<'a> SieveRouter<'a> { allowed, brute_force_threshold: 0.001, }; - navix_search(self.fallback, query, &opts, metric) + Ok(navix_search(self.fallback, query, &opts, metric)? .into_iter() .map(|r| SearchResult { id: r.id, distance: r.distance, }) - .collect() + .collect()) } } @@ -124,14 +126,16 @@ mod tests { fallback: &fallback, }; - let results = router.route( - &[2.0, 0.0, 0.0], - Some(&"T".to_string()), - all_allowed(20), // bitmap irrelevant on subindex path - 3, - 32, - DistanceMetric::L2, - ); + let results = router + .route( + &[2.0, 0.0, 0.0], + Some(&"T".to_string()), + all_allowed(20), // bitmap irrelevant on subindex path + 3, + 32, + DistanceMetric::L2, + ) + .unwrap(); assert!(!results.is_empty()); // All result IDs must be within the subindex range [0..5). @@ -152,14 +156,16 @@ mod tests { }; let allowed = all_allowed(20); - let results = router.route( - &[10.0, 0.0, 0.0], - Some(&"unknown_sig".to_string()), - allowed, - 3, - 64, - DistanceMetric::L2, - ); + let results = router + .route( + &[10.0, 0.0, 0.0], + Some(&"unknown_sig".to_string()), + allowed, + 3, + 64, + DistanceMetric::L2, + ) + .unwrap(); assert!(!results.is_empty()); // The nearest vector to [10,0,0] in [0..20] is id=10. @@ -182,7 +188,9 @@ mod tests { }; let allowed = all_allowed(20); - let results = router.route(&[5.0, 0.0, 0.0], None, allowed, 3, 64, DistanceMetric::L2); + let results = router + .route(&[5.0, 0.0, 0.0], None, allowed, 3, 64, DistanceMetric::L2) + .unwrap(); assert!(!results.is_empty()); // Must include id=5 since fallback has all 20 vectors. diff --git a/nodedb-vector/src/vamana/build.rs b/nodedb-vector/src/vamana/build.rs index 56d873d7d..a039b83b3 100644 --- a/nodedb-vector/src/vamana/build.rs +++ b/nodedb-vector/src/vamana/build.rs @@ -15,6 +15,7 @@ use nodedb_codec::vector_quant::codec::VectorCodec; +use crate::error::{VectorError, check_dim}; use crate::vamana::graph::VamanaGraph; use crate::vamana::prune::alpha_prune; @@ -35,10 +36,12 @@ use crate::vamana::prune::alpha_prune; /// * `alpha` — pruning factor (typical: 1.2; must be > 1). /// * `l_build` — beam width during construction (typical: 100). /// -/// # Panics +/// # Errors /// -/// Panics if `vectors`, `ids`, and `quantized` do not all have the same -/// length, or if `vectors` is empty. +/// - [`VectorError::InvalidInput`] when `vectors` is empty, or `vectors`, +/// `ids` and `quantized` differ in length. +/// - [`VectorError::DimensionMismatch`] when a vector's length differs from +/// the first vector's. pub fn build_vamana( vectors: &[Vec], ids: &[u64], @@ -47,21 +50,28 @@ pub fn build_vamana( r: usize, alpha: f32, l_build: usize, -) -> VamanaGraph { - assert_eq!( - vectors.len(), - ids.len(), - "vectors and ids must have equal length" - ); - assert_eq!( - vectors.len(), - quantized.len(), - "vectors and quantized must have equal length" - ); - assert!(!vectors.is_empty(), "cannot build from empty vector set"); +) -> Result { + let Some(first) = vectors.first() else { + return Err(VectorError::InvalidInput { + detail: "Vamana build needs at least one vector".into(), + }); + }; + if vectors.len() != ids.len() || vectors.len() != quantized.len() { + return Err(VectorError::InvalidInput { + detail: format!( + "Vamana build got {} vectors, {} ids and {} quantized vectors; the counts must match", + vectors.len(), + ids.len(), + quantized.len() + ), + }); + } + let dim = first.len(); + for v in vectors { + check_dim(dim, v.len())?; + } let n = vectors.len(); - let dim = vectors[0].len(); let l = l_build.max(r); // --- Construct graph skeleton --- @@ -132,7 +142,7 @@ pub fn build_vamana( graph.set_neighbors(i, pruned); } - graph + Ok(graph) } // ------------------------------------------------------------------ @@ -375,7 +385,7 @@ mod tests { let ids: Vec = (0..n as u64).collect(); let quantized: Vec = vecs.iter().map(|v| codec.encode(v)).collect(); - let graph = build_vamana(&vecs, &ids, &codec, &quantized, 8, 1.2, 20); + let graph = build_vamana(&vecs, &ids, &codec, &quantized, 8, 1.2, 20).unwrap(); assert_eq!(graph.len(), n); @@ -402,7 +412,7 @@ mod tests { let ids: Vec = (0..n as u64).collect(); let quantized: Vec = vecs.iter().map(|v| codec.encode(v)).collect(); - let graph = build_vamana(&vecs, &ids, &codec, &quantized, r, 1.2, 15); + let graph = build_vamana(&vecs, &ids, &codec, &quantized, r, 1.2, 15).unwrap(); for i in 0..n { assert!( diff --git a/nodedb-vector/src/vamana/search.rs b/nodedb-vector/src/vamana/search.rs index 6c228985f..b3e2b77f9 100644 --- a/nodedb-vector/src/vamana/search.rs +++ b/nodedb-vector/src/vamana/search.rs @@ -23,6 +23,7 @@ use std::collections::{BinaryHeap, HashSet}; use nodedb_codec::vector_quant::codec::VectorCodec; use crate::distance::scalar::l2_squared; +use crate::error::{VectorError, check_dim}; use crate::vamana::graph::VamanaGraph; use crate::vamana::node_fetcher::NodeFetcher; @@ -82,6 +83,14 @@ impl Ord for Candidate { /// # Returns /// /// Up to `k` results sorted by ascending distance. +/// +/// The query arrives already prepared by the codec, so its dimension is +/// checked where the caller prepares it; [`rerank`] checks the FP32 query. +/// +/// # Errors +/// +/// [`VectorError::InvalidInput`] when `quantized` does not hold one vector +/// per graph node. pub fn beam_search( graph: &VamanaGraph, query: &C::Query, @@ -90,13 +99,22 @@ pub fn beam_search( fetcher: &mut F, k: usize, l_search: usize, -) -> Vec +) -> Result, VectorError> where C: VectorCodec, F: NodeFetcher, { if graph.is_empty() || quantized.is_empty() { - return Vec::new(); + return Ok(Vec::new()); + } + if quantized.len() != graph.len() { + return Err(VectorError::InvalidInput { + detail: format!( + "Vamana search got {} quantized vectors for {} graph nodes", + quantized.len(), + graph.len() + ), + }); } let l = l_search.max(k); @@ -173,12 +191,13 @@ where }); out.truncate(k); - out.into_iter() + Ok(out + .into_iter() .map(|c| BeamSearchResult { id: graph.external_id(c.idx as usize), distance: c.dist, }) - .collect() + .collect()) } /// Rerank a candidate list using full-precision FP32 vectors from `fetcher`. @@ -192,13 +211,21 @@ where /// them without re-submitting. /// /// This is the "SSD fetch + rerank" step described in the DiskANN paper. +/// +/// # Errors +/// +/// - [`VectorError::DimensionMismatch`] when `query_fp32` does not have the +/// graph dimension. +/// - [`VectorError::StoredDimensionMismatch`] when a fetched vector does not +/// have the graph dimension. pub fn rerank( candidates: Vec, query_fp32: &[f32], fetcher: &mut F, graph: &VamanaGraph, k: usize, -) -> Vec { +) -> Result, VectorError> { + check_dim(graph.dim, query_fp32.len())?; // Build id → internal index map. let id_to_idx: std::collections::HashMap = graph.iter().map(|(idx, node)| (node.id, idx)).collect(); @@ -210,18 +237,25 @@ pub fn rerank( .collect(); fetcher.prefetch_batch(&candidate_indices); - let mut reranked: Vec = candidates - .into_iter() - .filter_map(|c| { - let idx = *id_to_idx.get(&c.id)?; - let vec = fetcher.fetch_fp32(idx as u32)?; - let d = l2_squared(query_fp32, &vec); - Some(BeamSearchResult { - id: c.id, - distance: d, - }) - }) - .collect(); + let mut reranked: Vec = Vec::with_capacity(candidates.len()); + for c in candidates { + let Some(&idx) = id_to_idx.get(&c.id) else { + continue; + }; + let Some(vec) = fetcher.fetch_fp32(idx as u32) else { + continue; + }; + if vec.len() != graph.dim { + return Err(VectorError::StoredDimensionMismatch { + expected: graph.dim, + got: vec.len(), + }); + } + reranked.push(BeamSearchResult { + id: c.id, + distance: l2_squared(query_fp32, &vec), + }); + } reranked.sort_by(|a, b| { a.distance @@ -229,7 +263,7 @@ pub fn rerank( .unwrap_or(std::cmp::Ordering::Equal) }); reranked.truncate(k); - reranked + Ok(reranked) } #[cfg(test)] @@ -293,14 +327,14 @@ mod tests { let ids: Vec = (0..n as u64).collect(); let quantized: Vec = vecs.iter().map(|v| codec.encode(v)).collect(); - let graph = build_vamana(&vecs, &ids, &codec, &quantized, 8, 1.2, 20); + let graph = build_vamana(&vecs, &ids, &codec, &quantized, 8, 1.2, 20).unwrap(); // Query with the vector at index 7; it should be the nearest result. let query_vec = vecs[7].clone(); let query = codec.prepare_query(&query_vec); let mut fetcher = InMemoryFetcher::new(dim, vecs.clone()); - let results = beam_search(&graph, &query, &codec, &quantized, &mut fetcher, 5, 20); + let results = beam_search(&graph, &query, &codec, &quantized, &mut fetcher, 5, 20).unwrap(); assert!( !results.is_empty(), @@ -315,4 +349,40 @@ mod tests { "distance to self must be near zero" ); } + + #[test] + fn wrong_dimension_is_a_typed_error() { + let dim = 4; + let codec = L2Codec; + let vecs = random_vecs(10, dim, 7); + let ids: Vec = (0..10).collect(); + let quantized: Vec = vecs.iter().map(|v| codec.encode(v)).collect(); + let graph = build_vamana(&vecs, &ids, &codec, &quantized, 4, 1.2, 8).unwrap(); + let mut fetcher = InMemoryFetcher::new(dim, vecs.clone()); + let query = codec.prepare_query(&vecs[0]); + let candidates = + beam_search(&graph, &query, &codec, &quantized, &mut fetcher, 3, 8).unwrap(); + assert!(matches!( + rerank(candidates, &[0.0; 3], &mut fetcher, &graph, 3), + Err(VectorError::DimensionMismatch { + expected: 4, + got: 3 + }) + )); + + let mut ragged = vecs.clone(); + ragged[3] = vec![0.0; 2]; + let ragged_q: Vec = ragged.iter().map(|v| codec.encode(v)).collect(); + assert!(matches!( + build_vamana(&ragged, &ids, &codec, &ragged_q, 4, 1.2, 8), + Err(VectorError::DimensionMismatch { + expected: 4, + got: 2 + }) + )); + assert!(matches!( + beam_search(&graph, &query, &codec, &quantized[..5], &mut fetcher, 3, 8), + Err(VectorError::InvalidInput { .. }) + )); + } } diff --git a/nodedb-vector/tests/vector_suite/cases/collection_bitmap_filter.rs b/nodedb-vector/tests/vector_suite/cases/collection_bitmap_filter.rs index 9f7696821..4fdc1bece 100644 --- a/nodedb-vector/tests/vector_suite/cases/collection_bitmap_filter.rs +++ b/nodedb-vector/tests/vector_suite/cases/collection_bitmap_filter.rs @@ -28,7 +28,7 @@ fn params() -> HnswParams { /// so the next inserts land at `base_id == seal_count`. fn seal_one(coll: &mut VectorCollection, count: usize) { for i in 0..count { - coll.insert(vec![i as f32, 0.0]); + coll.insert(vec![i as f32, 0.0]).unwrap(); } let req = coll.seal("k").expect("seal produced request"); let mut idx = HnswIndex::new(req.dim, req.params.clone()); @@ -59,7 +59,9 @@ fn bitmap_filter_targets_second_segment_global_ids() { // segment's bitmap lookup tests local id 25 against a bitmap that // contains global 75 → zero matches. let bytes = bitmap_bytes([75u32]); - let results = coll.search_with_bitmap_bytes(&[75.0, 0.0], 1, 64, &bytes); + let results = coll + .search_with_bitmap_bytes(&[75.0, 0.0], 1, 64, &bytes) + .unwrap(); assert_eq!( results.len(), @@ -79,7 +81,9 @@ fn bitmap_filter_recovers_many_globals_across_segments() { let wanted: Vec = (60..70).collect(); let bytes = bitmap_bytes(wanted.iter().copied()); - let results = coll.search_with_bitmap_bytes(&[65.0, 0.0], 10, 128, &bytes); + let results = coll + .search_with_bitmap_bytes(&[65.0, 0.0], 10, 128, &bytes) + .unwrap(); assert_eq!( results.len(), @@ -107,7 +111,9 @@ fn bitmap_filter_first_segment_still_works() { seal_one(&mut coll, 50); let bytes = bitmap_bytes([10u32, 20, 30]); - let results = coll.search_with_bitmap_bytes(&[20.0, 0.0], 3, 64, &bytes); + let results = coll + .search_with_bitmap_bytes(&[20.0, 0.0], 3, 64, &bytes) + .unwrap(); let got: std::collections::HashSet = results.iter().map(|r| r.id).collect(); let expected: std::collections::HashSet = [10u32, 20, 30].into_iter().collect(); assert_eq!(got, expected); diff --git a/nodedb-vector/tests/vector_suite/cases/collection_checkpoint_tombstones.rs b/nodedb-vector/tests/vector_suite/cases/collection_checkpoint_tombstones.rs index ebf6f1cde..f63c2446d 100644 --- a/nodedb-vector/tests/vector_suite/cases/collection_checkpoint_tombstones.rs +++ b/nodedb-vector/tests/vector_suite/cases/collection_checkpoint_tombstones.rs @@ -34,7 +34,7 @@ fn params() -> HnswParams { fn growing_segment_tombstones_survive_checkpoint_roundtrip() { let mut coll = VectorCollection::new(2, params()); for i in 0..10u32 { - coll.insert(vec![i as f32, 0.0]); + coll.insert(vec![i as f32, 0.0]).unwrap(); } assert!(coll.delete(3), "delete on live growing vector must succeed"); assert!(coll.delete(7), "delete on live growing vector must succeed"); @@ -51,7 +51,7 @@ fn growing_segment_tombstones_survive_checkpoint_roundtrip() { "tombstoned growing-segment vectors resurrected on restore" ); - let results = restored.search(&[3.0, 0.0], 10, 64); + let results = restored.search(&[3.0, 0.0], 10, 64).unwrap(); let ids: std::collections::HashSet = results.iter().map(|r| r.id).collect(); assert!( !ids.contains(&3), @@ -69,7 +69,7 @@ fn building_segment_tombstones_survive_checkpoint_roundtrip() { // snapshot time, exercising the `building_segments` encode path. let mut coll = VectorCollection::with_seal_threshold(2, params(), 20); for i in 0..20u32 { - coll.insert(vec![i as f32, 0.0]); + coll.insert(vec![i as f32, 0.0]).unwrap(); } let _req = coll.seal("k").expect("seal produced request"); // Intentionally do NOT complete the build — vectors now sit in the @@ -89,7 +89,7 @@ fn building_segment_tombstones_survive_checkpoint_roundtrip() { "tombstoned building-segment vectors resurrected on restore" ); - let results = restored.search(&[5.0, 0.0], 20, 64); + let results = restored.search(&[5.0, 0.0], 20, 64).unwrap(); let ids: std::collections::HashSet = results.iter().map(|r| r.id).collect(); assert!( !ids.contains(&5), diff --git a/nodedb-vector/tests/vector_suite/cases/collection_compact_doc_map.rs b/nodedb-vector/tests/vector_suite/cases/collection_compact_doc_map.rs index 65ea25343..bb0d48349 100644 --- a/nodedb-vector/tests/vector_suite/cases/collection_compact_doc_map.rs +++ b/nodedb-vector/tests/vector_suite/cases/collection_compact_doc_map.rs @@ -30,7 +30,8 @@ fn build_collection_with_docs() -> VectorCollection { let mut coll = VectorCollection::with_seal_threshold(2, params(), 6); // Six docs, one vector each. Global ids 0..6, surrogates 1..=6. for i in 0..6u32 { - coll.insert_with_surrogate(vec![i as f32, 0.0], Surrogate::new(i + 1)); + coll.insert_with_surrogate(vec![i as f32, 0.0], Surrogate::new(i + 1)) + .unwrap(); } let req = coll.seal("k").expect("seal produced request"); let mut idx = HnswIndex::new(req.dim, req.params.clone()); @@ -56,7 +57,7 @@ fn surrogate_map_stays_correct_after_compact() { let removed = coll.compact(); assert_eq!(removed, 2, "compact should remove 2 tombstoned nodes"); - let results = coll.search(&[0.0, 0.0], 4, 64); + let results = coll.search(&[0.0, 0.0], 4, 64).unwrap(); let ids: Vec = results.iter().map(|r| r.id).collect(); assert_eq!(ids.len(), 4, "expected 4 live vectors post-compact"); @@ -81,12 +82,12 @@ fn multi_doc_map_stays_correct_after_compact() { let a_vecs: Vec> = (0..3u32).map(|i| vec![i as f32, 0.0]).collect(); let a_refs: Vec<&[f32]> = a_vecs.iter().map(|v| v.as_slice()).collect(); - let a_ids = coll.insert_multi_vector(&a_refs, doc_a); + let a_ids = coll.insert_multi_vector(&a_refs, doc_a).unwrap(); assert_eq!(a_ids, vec![0, 1, 2]); let b_vecs: Vec> = (3..6u32).map(|i| vec![i as f32, 0.0]).collect(); let b_refs: Vec<&[f32]> = b_vecs.iter().map(|v| v.as_slice()).collect(); - let b_ids = coll.insert_multi_vector(&b_refs, doc_b); + let b_ids = coll.insert_multi_vector(&b_refs, doc_b).unwrap(); assert_eq!(b_ids, vec![3, 4, 5]); let req = coll.seal("k").expect("seal produced request"); diff --git a/nodedb-vector/tests/vector_suite/cases/collection_pq_config.rs b/nodedb-vector/tests/vector_suite/cases/collection_pq_config.rs index 512769ea8..c8cfd146e 100644 --- a/nodedb-vector/tests/vector_suite/cases/collection_pq_config.rs +++ b/nodedb-vector/tests/vector_suite/cases/collection_pq_config.rs @@ -40,7 +40,7 @@ fn make_built_collection_with_pq_config() -> VectorCollection { for (d, slot) in v.iter_mut().enumerate() { *slot = ((i as f32) * 0.01 + (d as f32) * 0.1).sin(); } - coll.insert(v); + coll.insert(v).unwrap(); } let req = coll.seal("pq").expect("seal produced request"); let mut idx = HnswIndex::new(req.dim, req.params.clone()); diff --git a/nodedb-vector/tests/vector_suite/cases/hnsw_checkpoint_encryption.rs b/nodedb-vector/tests/vector_suite/cases/hnsw_checkpoint_encryption.rs index d2070269e..999b759d9 100644 --- a/nodedb-vector/tests/vector_suite/cases/hnsw_checkpoint_encryption.rs +++ b/nodedb-vector/tests/vector_suite/cases/hnsw_checkpoint_encryption.rs @@ -30,7 +30,7 @@ fn make_key() -> WalEncryptionKey { fn make_collection() -> VectorCollection { let mut coll = VectorCollection::new(DIM, params()); for i in 0u32..10 { - coll.insert(vec![i as f32, 0.0, 0.0, 0.0]); + coll.insert(vec![i as f32, 0.0, 0.0, 0.0]).unwrap(); } coll } @@ -72,7 +72,7 @@ fn hnsw_checkpoint_encrypted_at_rest() { assert_eq!(restored.dim(), DIM); // Nearest neighbour to [5.0, 0, 0, 0] must be vector id=5. - let results = restored.search(&[5.0, 0.0, 0.0, 0.0], 1, 64); + let results = restored.search(&[5.0, 0.0, 0.0, 0.0], 1, 64).unwrap(); assert!(!results.is_empty(), "search must return a result"); assert_eq!( results[0].id, 5, diff --git a/nodedb-vector/tests/vector_suite/cases/quantize_kmeans_distribution.rs b/nodedb-vector/tests/vector_suite/cases/quantize_kmeans_distribution.rs index b7cf74218..a5c036c6a 100644 --- a/nodedb-vector/tests/vector_suite/cases/quantize_kmeans_distribution.rs +++ b/nodedb-vector/tests/vector_suite/cases/quantize_kmeans_distribution.rs @@ -64,7 +64,7 @@ fn unique_centroid_count(codec: &PqCodec, vectors: &[Vec]) -> usize { fn pq_kmeans_produces_diverse_centroids_on_duplicate_heavy_data() { let vecs = clustered_with_duplicates(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = PqCodec::train(&refs, 4, 2, 16, 20, test_memory()); + let codec = PqCodec::train(&refs, 4, 2, 16, 20, test_memory()).unwrap(); let unique = unique_centroid_count(&codec, &vecs); assert!( @@ -83,7 +83,7 @@ fn pq_distance_table_separates_duplicates_from_outliers() { // codebook entries alias to one point so all distances look similar. let vecs = clustered_with_duplicates(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = PqCodec::train(&refs, 4, 2, 16, 20, test_memory()); + let codec = PqCodec::train(&refs, 4, 2, 16, 20, test_memory()).unwrap(); let query = [0.0f32, 0.0, 0.0, 0.0]; let table = codec @@ -121,15 +121,15 @@ fn ivf_pq_training_does_not_collapse_on_duplicate_heavy_data() { metric: DistanceMetric::L2, }, ); - idx.train(&refs, test_memory()); + idx.train(&refs, test_memory()).unwrap(); for v in &vecs { - idx.add(v); + idx.add(v).unwrap(); } // Query at the origin. Correct training assigns near-duplicates to // one cell and outliers to another; the nearest result must come // from the duplicate cluster (original indices 0..190). - let results = idx.search(&[0.0, 0.0, 0.0, 0.0], 5); + let results = idx.search(&[0.0, 0.0, 0.0, 0.0], 5).unwrap(); assert!(!results.is_empty(), "IVF-PQ returned no results"); for r in &results { assert!( diff --git a/nodedb-wal/Cargo.toml b/nodedb-wal/Cargo.toml index c6e084769..db6938905 100644 --- a/nodedb-wal/Cargo.toml +++ b/nodedb-wal/Cargo.toml @@ -17,8 +17,9 @@ default = [] io-uring = ["dep:io-uring"] # File structured corruption / invariant / durability reports through the # `faultbox` black-box recorder. Inert until the host application calls -# `faultbox::init`; a no-op on wasm32 regardless. Off by default so the crate -# stays dependency-light for embedders that report failures their own way. +# `faultbox::init`. Pulls nothing on wasm32, where the reporting modules do not +# build. Off by default so the crate stays dependency-light for embedders that +# report failures their own way. diagnostics = ["dep:faultbox"] # Arm the crash-injection points used by the durability tests. Without it every # `fail_point!` compiles to nothing, so production builds pay no cost. @@ -35,21 +36,22 @@ hkdf = { workspace = true } sha2 = { workspace = true } getrandom = { workspace = true } zeroize = { workspace = true } -faultbox = { workspace = true, optional = true } [target.'cfg(target_os = "linux")'.dependencies] io-uring = { workspace = true, optional = true } +# The writer, readers, segments, and double-write buffer build only off wasm32. +# On wasm32 the crate builds the crypto module and what it needs. [target.'cfg(not(target_arch = "wasm32"))'.dependencies] memmap2 = { workspace = true } +faultbox = { workspace = true, optional = true } # `getrandom` reaches this crate several times and no copy picks a backend on its # own for `wasm32-unknown-unknown`: the workspace version arrives directly, 0.3 # via sonic-rs/rand, 0.2 via aes-gcm's `rand_core`. Naming each here with its # JS-backed feature lets Cargo's feature unification apply them to the # transitive copies, which is the same approach the WASM bindings crate already -# uses. Without this the crate does not compile for the target at all, despite -# carrying wasm32 code paths. +# uses. Without this the crypto module does not compile for the target. # # The unaliased entry tracks the workspace version rather than pinning one: # pinning it here means a workspace bump leaves this entry naming a version the diff --git a/nodedb-wal/src/diag/inert.rs b/nodedb-wal/src/diag/inert.rs index e1aecfb99..d6583e2dc 100644 --- a/nodedb-wal/src/diag/inert.rs +++ b/nodedb-wal/src/diag/inert.rs @@ -2,8 +2,7 @@ //! The non-recording implementation of the WAL's report sites. //! -//! Compiled when the `diagnostics` feature is off, and unconditionally on -//! wasm32 where there is no filesystem to write a report to. Every entry point +//! Compiled when the `diagnostics` feature is off. Every entry point //! keeps the signature of its recording counterpart so call sites are free of //! `cfg`, and every one is empty so the WAL behaves byte-for-byte as it did //! before the recorder existed. diff --git a/nodedb-wal/src/diag/mod.rs b/nodedb-wal/src/diag/mod.rs index e83acc5d8..5339248fc 100644 --- a/nodedb-wal/src/diag/mod.rs +++ b/nodedb-wal/src/diag/mod.rs @@ -17,25 +17,24 @@ //! redactor belongs to the binary; everything here is inert until the host //! application initializes the recorder, so a library emitting these costs //! nothing on its own. The real implementation compiles only under the -//! `diagnostics` feature and off wasm32 (where there is no filesystem to write -//! a report to); otherwise every entry point is a no-op with the same +//! `diagnostics` feature; otherwise every entry point is a no-op with the same //! signature, so call sites never need a `cfg`. -#[cfg(all(feature = "diagnostics", not(target_arch = "wasm32")))] +#[cfg(feature = "diagnostics")] mod context; -#[cfg(all(feature = "diagnostics", not(target_arch = "wasm32")))] +#[cfg(feature = "diagnostics")] mod recording; -#[cfg(not(all(feature = "diagnostics", not(target_arch = "wasm32"))))] +#[cfg(not(feature = "diagnostics"))] mod inert; -#[cfg(all(feature = "diagnostics", not(target_arch = "wasm32")))] +#[cfg(feature = "diagnostics")] pub use recording::{ durability_lost, encrypted_record_without_key, mid_file_corruption, out_of_space, replay_below_retained_floor, segment_lsn_gap, }; -#[cfg(not(all(feature = "diagnostics", not(target_arch = "wasm32"))))] +#[cfg(not(feature = "diagnostics"))] pub use inert::{ durability_lost, encrypted_record_without_key, mid_file_corruption, out_of_space, replay_below_retained_floor, segment_lsn_gap, diff --git a/nodedb-wal/src/double_write/raw_io.rs b/nodedb-wal/src/double_write/raw_io.rs index 11a3f570f..49f4bd331 100644 --- a/nodedb-wal/src/double_write/raw_io.rs +++ b/nodedb-wal/src/double_write/raw_io.rs @@ -31,37 +31,27 @@ pub(crate) fn full_capacity_slice(buf: &AlignedBuf) -> &[u8] { /// `pwrite`-retry helper that handles short writes. pub(crate) fn pwrite_all(file: &File, data: &[u8], offset: u64) -> Result<()> { - #[cfg(not(target_arch = "wasm32"))] - { - use std::os::unix::io::AsRawFd as _; - let fd = file.as_raw_fd(); - let mut remaining = data; - let mut write_offset = offset; - while !remaining.is_empty() { - // SAFETY: `remaining` is a live slice of `remaining.len()` bytes - // and `fd` is owned by the borrowed `file`. - let n = unsafe { - libc::pwrite( - fd, - remaining.as_ptr() as *const libc::c_void, - remaining.len(), - write_offset as libc::off_t, - ) - }; - if n < 0 { - return Err(WalError::Io(std::io::Error::last_os_error())); - } - let n = n as usize; - remaining = &remaining[n..]; - write_offset += n as u64; + use std::os::unix::io::AsRawFd as _; + let fd = file.as_raw_fd(); + let mut remaining = data; + let mut write_offset = offset; + while !remaining.is_empty() { + // SAFETY: `remaining` is a live slice of `remaining.len()` bytes + // and `fd` is owned by the borrowed `file`. + let n = unsafe { + libc::pwrite( + fd, + remaining.as_ptr() as *const libc::c_void, + remaining.len(), + write_offset as libc::off_t, + ) + }; + if n < 0 { + return Err(WalError::Io(std::io::Error::last_os_error())); } - Ok(()) - } - #[cfg(target_arch = "wasm32")] - { - let _ = (file, data, offset); - Err(WalError::Unsupported { - detail: "O_DIRECT pwrite not available on wasm32", - }) + let n = n as usize; + remaining = &remaining[n..]; + write_offset += n as u64; } + Ok(()) } diff --git a/nodedb-wal/src/double_write/recover.rs b/nodedb-wal/src/double_write/recover.rs index 70600e69a..91d986b05 100644 --- a/nodedb-wal/src/double_write/recover.rs +++ b/nodedb-wal/src/double_write/recover.rs @@ -11,10 +11,7 @@ use crate::error::Result; use crate::record::WalRecord; use super::buffer::DoubleWriteBuffer; -use super::layout::SlotPrefix; - -#[cfg(not(target_arch = "wasm32"))] -use super::layout::{DWB_CAPACITY, DWB_SLOT_STRIDE, SLOT_PREFIX_SIZE, slot_offset}; +use super::layout::{DWB_CAPACITY, DWB_SLOT_STRIDE, SLOT_PREFIX_SIZE, SlotPrefix, slot_offset}; impl DoubleWriteBuffer { /// Try to recover a WAL record by LSN from the double-write buffer. @@ -25,31 +22,19 @@ impl DoubleWriteBuffer { /// highest slot sequence number wins. Returning the older one would /// resurrect a payload that was never acknowledged to any client. pub fn recover_record(&mut self, target_lsn: u64) -> Result> { - // Tail expressions, not early returns: exactly one arm compiles per - // target, so a `return` here is redundant and `-D warnings` rejects it - // on the wasm build. - #[cfg(target_arch = "wasm32")] - { - let _ = target_lsn; - Ok(None) - } - - #[cfg(not(target_arch = "wasm32"))] - { - let mut best: Option<(u64, WalRecord)> = None; - for_each_slot(self, |prefix, record_bytes| { - let Some(record) = decode_record(record_bytes) else { - return; - }; - if record.header.lsn != target_lsn || record.verify_checksum().is_err() { - return; - } - if best.as_ref().is_none_or(|(seq, _)| prefix.seq > *seq) { - best = Some((prefix.seq, record)); - } - })?; - Ok(best.map(|(_, record)| record)) - } + let mut best: Option<(u64, WalRecord)> = None; + for_each_slot(self, |prefix, record_bytes| { + let Some(record) = decode_record(record_bytes) else { + return; + }; + if record.header.lsn != target_lsn || record.verify_checksum().is_err() { + return; + } + if best.as_ref().is_none_or(|(seq, _)| prefix.seq > *seq) { + best = Some((prefix.seq, record)); + } + })?; + Ok(best.map(|(_, record)| record)) } } @@ -58,45 +43,13 @@ impl DoubleWriteBuffer { /// The write path resumes above this so a reused sequence number can never let /// a stale copy tie with the record that replaced it. pub(super) fn scan_max_seq(dwb: &mut DoubleWriteBuffer) -> Result { - #[cfg(not(target_arch = "wasm32"))] - { - let mut max = 0u64; - for_each_slot(dwb, |prefix, _| max = max.max(prefix.seq))?; - Ok(max) - } - - #[cfg(target_arch = "wasm32")] - { - use std::io::{Read as _, Seek as _, SeekFrom}; - - use super::layout::{DWB_CAPACITY, SLOT_PREFIX_SIZE, slot_offset}; - use crate::error::WalError; - - let mut max = 0u64; - let mut prefix = [0u8; SLOT_PREFIX_SIZE]; - for i in 0..DWB_CAPACITY as u32 { - if dwb - .file - .seek(SeekFrom::Start(slot_offset(i))) - .map_err(WalError::Io) - .is_err() - { - continue; - } - if dwb.file.read_exact(&mut prefix).is_err() { - continue; - } - if let Some(decoded) = SlotPrefix::decode(&prefix) { - max = max.max(decoded.seq); - } - } - Ok(max) - } + let mut max = 0u64; + for_each_slot(dwb, |prefix, _| max = max.max(prefix.seq))?; + Ok(max) } /// Visit every slot that carries a well-formed prefix, handing the callback /// the prefix and the record bytes (WAL header + payload) it frames. -#[cfg(not(target_arch = "wasm32"))] fn for_each_slot(dwb: &DoubleWriteBuffer, mut visit: F) -> Result<()> where F: FnMut(&SlotPrefix, &[u8]), @@ -141,7 +94,6 @@ where /// Rebuild a record from the bytes a slot frames. `None` when the header is /// not a WAL header at all. -#[cfg(not(target_arch = "wasm32"))] fn decode_record(bytes: &[u8]) -> Option { use crate::record::{HEADER_SIZE, RecordHeader, WAL_MAGIC}; diff --git a/nodedb-wal/src/error.rs b/nodedb-wal/src/error.rs index 9eb422b57..8cedd3fe6 100644 --- a/nodedb-wal/src/error.rs +++ b/nodedb-wal/src/error.rs @@ -100,10 +100,6 @@ pub enum WalError { #[error("filesystem holding {path} does not support O_DIRECT")] DirectIoUnsupported { path: String }, - /// Operation is not supported on the current platform (e.g. wasm32). - #[error("WAL operation not supported on this platform: {detail}")] - Unsupported { detail: &'static str }, - /// The filesystem ran out of space while appending to the WAL (ENOSPC). /// /// Distinct from a generic [`WalError::Io`] so callers can stop diff --git a/nodedb-wal/src/lazy_reader.rs b/nodedb-wal/src/lazy_reader.rs index 3e175dd72..bb4316480 100644 --- a/nodedb-wal/src/lazy_reader.rs +++ b/nodedb-wal/src/lazy_reader.rs @@ -121,7 +121,7 @@ impl LazyWalReader { Ok(None) } - /// Read the next record header (54 bytes) without reading the payload. + /// Read the next record header without reading the payload. /// /// Returns `None` at EOF or first corruption. After this call, use /// either `read_payload()` to get the payload or `skip_payload()` to @@ -460,7 +460,8 @@ mod tests { vshard_id: 0, payload_len: (MAX_WAL_PAYLOAD_SIZE + 1) as u32, database_id: 0, - reserved: [0; 8], + apply_key: 0, + event_source: crate::record::NO_EVENT_SOURCE, crc32c: 0, }; std::fs::write(&path, header.to_bytes()).unwrap(); diff --git a/nodedb-wal/src/lib.rs b/nodedb-wal/src/lib.rs index f61e167e8..361b071ac 100644 --- a/nodedb-wal/src/lib.rs +++ b/nodedb-wal/src/lib.rs @@ -22,54 +22,88 @@ //! //! Sustain 100,000+ async writes/sec with sub-millisecond p99 latency. //! `free -m` cached memory must not move during the benchmark. +//! +//! ## Targets +//! +//! On wasm32 the crate builds only [`crypto`], the key and envelope code it +//! needs, [`error`], and the record-header constants in [`record::header`]. +//! The writer, readers, segments, double-write buffer, replay, and recovery +//! build only on native targets. +#[cfg(not(target_arch = "wasm32"))] pub mod align; pub mod crypto; +#[cfg(not(target_arch = "wasm32"))] pub mod diag; +#[cfg(not(target_arch = "wasm32"))] pub mod double_write; pub mod error; +#[cfg(not(target_arch = "wasm32"))] pub mod lazy_reader; #[cfg(not(target_arch = "wasm32"))] pub mod mmap_reader; +#[cfg(not(target_arch = "wasm32"))] pub mod preamble; +#[cfg(not(target_arch = "wasm32"))] pub mod reader; pub mod record; +#[cfg(not(target_arch = "wasm32"))] pub mod recovery; +#[cfg(not(target_arch = "wasm32"))] pub mod replay; pub mod secure_mem; +#[cfg(not(target_arch = "wasm32"))] pub mod segment; mod segment_envelope; +#[cfg(not(target_arch = "wasm32"))] pub mod segmented; +#[cfg(not(target_arch = "wasm32"))] pub mod temporal_purge; +#[cfg(not(target_arch = "wasm32"))] pub mod tombstone; +#[cfg(not(target_arch = "wasm32"))] pub mod torn_tail; #[cfg(all(feature = "io-uring", target_os = "linux"))] pub mod uring_writer; +#[cfg(not(target_arch = "wasm32"))] pub mod writer; +#[cfg(not(target_arch = "wasm32"))] pub use double_write::{ DoubleWriteBuffer, DwbDegradation, DwbMirror, DwbMode, DwbProtection, DwbSkipReason, wal_dwb_bytes_written_total, wal_dwb_degradations_total, wal_dwb_unprotected_records_total, }; pub use error::{Result, WalError}; +#[cfg(not(target_arch = "wasm32"))] pub use lazy_reader::LazyWalReader; +#[cfg(not(target_arch = "wasm32"))] pub use preamble::{ CIPHER_AES_256_GCM, PREAMBLE_SIZE, PREAMBLE_VERSION, SEG_PREAMBLE_MAGIC, SegmentPreamble, WAL_PREAMBLE_MAGIC, }; +#[cfg(not(target_arch = "wasm32"))] pub use reader::{StopReason, WalReader}; +#[cfg(not(target_arch = "wasm32"))] pub use record::{ - CalvinAppliedPayload, FtsDeletePayload, FtsIndexPayload, RecordHeader, RecordType, + CalvinAppliedPayload, FtsDeletePayload, FtsIndexPayload, RecordStamp, RecordTarget, RecordType, SpatialDeletePayload, SpatialPutPayload, WalRecord, WalRecordArgs, WriteAbortedPayload, }; +pub use record::{NO_EVENT_SOURCE, RecordHeader}; +#[cfg(not(target_arch = "wasm32"))] pub use recovery::{RecoveryInfo, recover}; +#[cfg(not(target_arch = "wasm32"))] pub use replay::{ AbortedWrites, DatabaseTombstones, ReplayFilters, TombstoneSet, drop_aborted_records, extract_replay_filters, extract_tombstones, }; pub use secure_mem::SecureKey; +#[cfg(not(target_arch = "wasm32"))] pub use segmented::{SegmentedWal, SegmentedWalConfig}; +#[cfg(not(target_arch = "wasm32"))] pub use temporal_purge::{TemporalPurgeEngine, TemporalPurgePayload}; +#[cfg(not(target_arch = "wasm32"))] pub use tombstone::{CollectionTombstonePayload, MAX_COLLECTION_NAME_LEN}; +#[cfg(not(target_arch = "wasm32"))] pub use torn_tail::{TailVerdict, verify_committed_prefix}; +#[cfg(not(target_arch = "wasm32"))] pub use writer::WalWriter; diff --git a/nodedb-wal/src/mmap_reader/reader.rs b/nodedb-wal/src/mmap_reader/reader.rs index 0ca585489..6a6041aa0 100644 --- a/nodedb-wal/src/mmap_reader/reader.rs +++ b/nodedb-wal/src/mmap_reader/reader.rs @@ -479,7 +479,8 @@ mod tests { vshard_id: 0, payload_len: 1, database_id: 0, - reserved: [0; 8], + apply_key: 0, + event_source: crate::record::NO_EVENT_SOURCE, crc32c: 0, }; std::fs::write(&path, header.to_bytes()).unwrap(); @@ -501,7 +502,8 @@ mod tests { vshard_id: 0, payload_len: (crate::record::MAX_WAL_PAYLOAD_SIZE + 1) as u32, database_id: 0, - reserved: [0; 8], + apply_key: 0, + event_source: crate::record::NO_EVENT_SOURCE, crc32c: 0, }; std::fs::write(&path, header.to_bytes()).unwrap(); diff --git a/nodedb-wal/src/reader.rs b/nodedb-wal/src/reader.rs index edaade37c..f63b03205 100644 --- a/nodedb-wal/src/reader.rs +++ b/nodedb-wal/src/reader.rs @@ -471,7 +471,8 @@ mod tests { vshard_id: 0, payload_len: (crate::record::MAX_WAL_PAYLOAD_SIZE + 1) as u32, database_id: 0, - reserved: [0; 8], + apply_key: 0, + event_source: crate::record::NO_EVENT_SOURCE, crc32c: 0, }; std::fs::write(&path, header.to_bytes()).unwrap(); diff --git a/nodedb-wal/src/record/header.rs b/nodedb-wal/src/record/header.rs index 4dd9bf225..499d2cb24 100644 --- a/nodedb-wal/src/record/header.rs +++ b/nodedb-wal/src/record/header.rs @@ -1,6 +1,6 @@ // SPDX-License-Identifier: Apache-2.0 -//! WAL record header: fixed 54-byte prefix + constants. +//! WAL record header: fixed 55-byte prefix + constants. use crate::error::{Result, WalError}; @@ -22,7 +22,10 @@ pub const WAL_MAGIC: u32 = 0x5359_4E57; // "SYNW" /// /// v1 is the initial shipped format with 54-byte headers (u64 tenant_id, /// u16 vshard_id, u32 payload_len, u16 reserved, u32 crc32c). -pub const WAL_FORMAT_VERSION: u16 = 1; +/// +/// v2 adds the one-byte event source at offset 50 and grows the header to +/// 55 bytes. A v1 record does not open. +pub const WAL_FORMAT_VERSION: u16 = 2; /// Maximum WAL record payload size (64 MiB). Distinct from cluster RPC's limit. pub const MAX_WAL_PAYLOAD_SIZE: usize = 64 * 1024 * 1024; @@ -31,13 +34,19 @@ pub const MAX_WAL_PAYLOAD_SIZE: usize = 64 * 1024 * 1024; /// /// Layout (all little-endian): /// magic(4) | format_version(2) | record_type(4) | lsn(8) | tenant_id(8) -/// | vshard_id(4) | payload_len(4) | database_id(8) | reserved(8) | crc32c(4) +/// | vshard_id(4) | payload_len(4) | database_id(8) | apply_key(8) +/// | event_source(1) | crc32c(4) /// /// `database_id` occupies bytes 34–41 (previously part of the 16-byte reserved -/// field). `reserved` occupies bytes 42–49. Bytes 34–41 were zero-filled in +/// field). `apply_key` occupies bytes 42–49. Bytes 34–41 were zero-filled in /// prior records, so `database_id == 0` maps to `DatabaseId(0)` (the default /// database), preserving backward compatibility without a format-version bump. -pub const HEADER_SIZE: usize = 54; +pub const HEADER_SIZE: usize = 55; + +/// The event source of a record that carries no row write. Replay rebuilds +/// no write event from it. A write record carries the code of the source its +/// write ran with; the WAL stores the code and does not interpret it. +pub const NO_EVENT_SOURCE: u8 = 0; /// Bit 14 in `record_type` signals the payload is AES-256-GCM encrypted. /// Separate from bit 15 (required flag). Both bits keep their positions; @@ -48,7 +57,7 @@ pub const ENCRYPTED_FLAG: u32 = 0x0000_4000; /// must not be silently skipped. pub const REQUIRED_FLAG: u32 = 0x0000_8000; -/// WAL record header (fixed 54 bytes). +/// WAL record header (fixed 55 bytes). #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub struct RecordHeader { pub magic: u32, @@ -64,9 +73,16 @@ pub struct RecordHeader { /// /// Occupies bytes 34–41 of the on-disk header (previously part of reserved). pub database_id: u64, - /// Reserved for future use; must be zero on write; ignored on read - /// (but covered by CRC32C). Occupies bytes 42–49. - pub reserved: [u8; 8], + /// The idempotency key of the replicated proposal whose apply appended + /// this record, `0` for a record no proposal apply appended. The record + /// and the key are durable together, so a node recovers which proposals + /// it applied from the records themselves. Covered by CRC32C. Occupies + /// bytes 42–49. + pub apply_key: u64, + /// The event source of the row write this record carries, as the writer's + /// code. [`NO_EVENT_SOURCE`] for a record that carries no row write. + /// Covered by CRC32C. Occupies byte 50. + pub event_source: u8, pub crc32c: u32, } @@ -81,14 +97,13 @@ impl RecordHeader { buf[26..30].copy_from_slice(&self.vshard_id.to_le_bytes()); buf[30..34].copy_from_slice(&self.payload_len.to_le_bytes()); buf[34..42].copy_from_slice(&self.database_id.to_le_bytes()); - buf[42..50].copy_from_slice(&self.reserved); - buf[50..54].copy_from_slice(&self.crc32c.to_le_bytes()); + buf[42..50].copy_from_slice(&self.apply_key.to_le_bytes()); + buf[50] = self.event_source; + buf[51..55].copy_from_slice(&self.crc32c.to_le_bytes()); buf } pub fn from_bytes(buf: &[u8; HEADER_SIZE]) -> Self { - let mut reserved = [0u8; 8]; - reserved.copy_from_slice(&buf[42..50]); Self { magic: u32::from_le_bytes([buf[0], buf[1], buf[2], buf[3]]), format_version: u16::from_le_bytes([buf[4], buf[5]]), @@ -104,15 +119,18 @@ impl RecordHeader { database_id: u64::from_le_bytes([ buf[34], buf[35], buf[36], buf[37], buf[38], buf[39], buf[40], buf[41], ]), - reserved, - crc32c: u32::from_le_bytes([buf[50], buf[51], buf[52], buf[53]]), + apply_key: u64::from_le_bytes([ + buf[42], buf[43], buf[44], buf[45], buf[46], buf[47], buf[48], buf[49], + ]), + event_source: buf[50], + crc32c: u32::from_le_bytes([buf[51], buf[52], buf[53], buf[54]]), } } /// CRC32C over header (excluding the crc32c field) + payload. /// - /// The 16 reserved bytes are included in the CRC so they cannot be - /// silently modified without detection. + /// The apply key is included in the CRC so it cannot be silently + /// modified without detection. pub fn compute_checksum(&self, payload: &[u8]) -> u32 { let header_bytes = self.to_bytes(); let mut digest = crc32c::crc32c(&header_bytes[..HEADER_SIZE - 4]); @@ -168,7 +186,8 @@ mod tests { vshard_id, payload_len: 100, database_id: 0, - reserved: [0u8; 8], + apply_key: 0, + event_source: NO_EVENT_SOURCE, crc32c: 0xDEAD_BEEF, } } @@ -181,11 +200,11 @@ mod tests { } #[test] - fn header_golden_54_bytes_exact_offsets() { + fn header_golden_55_bytes_exact_offsets() { // magic at 0..4, format_version at 4..6, record_type at 6..10, // lsn at 10..18, tenant_id at 18..26, vshard_id at 26..30, - // payload_len at 30..34, database_id at 34..42, reserved at 42..50, - // crc32c at 50..54. + // payload_len at 30..34, database_id at 34..42, apply_key at 42..50, + // event_source at 50, crc32c at 51..55. let header = RecordHeader { magic: WAL_MAGIC, format_version: WAL_FORMAT_VERSION, @@ -195,11 +214,12 @@ mod tests { vshard_id: 0xCAFE_BABE, payload_len: 256, database_id: 0xABCD_0000_1234_5678, - reserved: [0u8; 8], + apply_key: 0, + event_source: NO_EVENT_SOURCE, crc32c: 0x1234_5678, }; let b = header.to_bytes(); - assert_eq!(b.len(), 54); + assert_eq!(b.len(), 55); // magic assert_eq!(&b[0..4], &WAL_MAGIC.to_le_bytes()); // format_version @@ -216,10 +236,12 @@ mod tests { assert_eq!(&b[30..34], &256u32.to_le_bytes()); // database_id assert_eq!(&b[34..42], &0xABCD_0000_1234_5678u64.to_le_bytes()); - // reserved — all zero + // apply_key — zero assert_eq!(&b[42..50], &[0u8; 8]); + // event_source + assert_eq!(b[50], NO_EVENT_SOURCE); // crc32c - assert_eq!(&b[50..54], &0x1234_5678u32.to_le_bytes()); + assert_eq!(&b[51..55], &0x1234_5678u32.to_le_bytes()); } #[test] @@ -234,7 +256,8 @@ mod tests { vshard_id: 0, payload_len: 0, database_id: 7, - reserved: [0u8; 8], + apply_key: 0, + event_source: NO_EVENT_SOURCE, crc32c: 0, }; let bytes = header.to_bytes(); @@ -268,7 +291,8 @@ mod tests { vshard_id: 0, payload_len: 0, database_id: 0, - reserved: [0u8; 8], + apply_key: 0, + event_source: NO_EVENT_SOURCE, crc32c: 0, }; let bytes = header.to_bytes(); @@ -353,4 +377,15 @@ mod tests { ); assert_eq!(decoded2.logical_record_type(), 0x0001_0001 | REQUIRED_FLAG); } + + #[test] + fn every_event_source_code_roundtrips() { + for code in 0..=u8::MAX { + let mut header = make_header(1, 0); + header.event_source = code; + let decoded = RecordHeader::from_bytes(&header.to_bytes()); + assert_eq!(decoded.event_source, code); + assert_eq!(decoded, header); + } + } } diff --git a/nodedb-wal/src/record/mod.rs b/nodedb-wal/src/record/mod.rs index 6e7a0e4bf..addd1e768 100644 --- a/nodedb-wal/src/record/mod.rs +++ b/nodedb-wal/src/record/mod.rs @@ -1,26 +1,46 @@ // SPDX-License-Identifier: Apache-2.0 +#[cfg(not(target_arch = "wasm32"))] pub mod aborted; +#[cfg(not(target_arch = "wasm32"))] pub mod anchor; +#[cfg(not(target_arch = "wasm32"))] pub mod calvin; +#[cfg(not(target_arch = "wasm32"))] pub mod fts_spatial; pub mod header; +#[cfg(not(target_arch = "wasm32"))] pub mod padding; +#[cfg(not(target_arch = "wasm32"))] pub mod surrogate; +#[cfg(not(target_arch = "wasm32"))] pub mod sync_seq; +#[cfg(not(target_arch = "wasm32"))] pub mod types; +#[cfg(not(target_arch = "wasm32"))] pub mod wal_record; +#[cfg(not(target_arch = "wasm32"))] pub use aborted::{WRITE_ABORTED_PAYLOAD_SIZE, WriteAbortedPayload}; +#[cfg(not(target_arch = "wasm32"))] pub use anchor::{ANCHOR_PAYLOAD_SIZE, LsnMsAnchorPayload}; +#[cfg(not(target_arch = "wasm32"))] pub use calvin::CalvinAppliedPayload; +#[cfg(not(target_arch = "wasm32"))] pub use fts_spatial::{FtsDeletePayload, FtsIndexPayload, SpatialDeletePayload, SpatialPutPayload}; pub use header::{ - ENCRYPTED_FLAG, HEADER_SIZE, MAX_WAL_PAYLOAD_SIZE, RecordHeader, WAL_FORMAT_VERSION, WAL_MAGIC, + ENCRYPTED_FLAG, HEADER_SIZE, MAX_WAL_PAYLOAD_SIZE, NO_EVENT_SOURCE, RecordHeader, + WAL_FORMAT_VERSION, WAL_MAGIC, }; +#[cfg(not(target_arch = "wasm32"))] pub(crate) use padding::pad_buffer_to_alignment; +#[cfg(not(target_arch = "wasm32"))] pub use padding::{MIN_PADDING_RECORD_SIZE, padding_record, padding_span}; +#[cfg(not(target_arch = "wasm32"))] pub use surrogate::{SURROGATE_PAYLOAD_SIZE, SurrogateAllocPayload, SurrogateBindPayload}; +#[cfg(not(target_arch = "wasm32"))] pub use sync_seq::{SYNC_SEQ_ADVANCE_PAYLOAD_SIZE, SyncSeqAdvancePayload}; +#[cfg(not(target_arch = "wasm32"))] pub use types::RecordType; -pub use wal_record::{WalRecord, WalRecordArgs}; +#[cfg(not(target_arch = "wasm32"))] +pub use wal_record::{RecordStamp, RecordTarget, WalRecord, WalRecordArgs}; diff --git a/nodedb-wal/src/record/types.rs b/nodedb-wal/src/record/types.rs index 89a8af891..b3540cd97 100644 --- a/nodedb-wal/src/record/types.rs +++ b/nodedb-wal/src/record/types.rs @@ -345,6 +345,20 @@ pub enum RecordType { /// exists to prevent, so an older binary pointed at a WAL containing one /// must fail to start loudly rather than readmit refused data. WriteAborted = 61 | 0x8000, + + /// Marks one replicated Raft proposal as applied on this node when its + /// apply writes no record of its own (a `wal=false` timeseries ingest). + /// Payload: empty. The proposal's idempotency key is the header's + /// `apply_key`, as on every record a proposal's apply appends. + /// + /// A proposal re-proposed after a leader change can commit at two log + /// indexes. The data-group apply loop recovers every keyed record at boot + /// and skips the second copy, so a write applies once per proposal. Never + /// replayed into any engine. + /// + /// Required: skipping this record re-applies a duplicate proposal, which + /// double-counts every non-idempotent effect (a timeseries append). + ProposalApplied = 62 | 0x8000, } impl RecordType { @@ -399,6 +413,7 @@ impl RecordType { x if x == 59 | 0x8000 => Some(Self::GraphNodeLabelSet), x if x == 60 | 0x8000 => Some(Self::GraphNodeLabelRemove), x if x == 61 | 0x8000 => Some(Self::WriteAborted), + x if x == 62 | 0x8000 => Some(Self::ProposalApplied), _ => None, } } @@ -483,6 +498,7 @@ mod tests { RecordType::GraphNodeLabelSet, RecordType::GraphNodeLabelRemove, RecordType::WriteAborted, + RecordType::ProposalApplied, ] { assert_eq!(RecordType::from_raw(ty as u32), Some(ty)); } diff --git a/nodedb-wal/src/record/wal_record.rs b/nodedb-wal/src/record/wal_record.rs index a88c089fc..e34facdc4 100644 --- a/nodedb-wal/src/record/wal_record.rs +++ b/nodedb-wal/src/record/wal_record.rs @@ -3,7 +3,8 @@ //! `WalRecord` — header + payload with encryption + checksum helpers. use super::header::{ - ENCRYPTED_FLAG, HEADER_SIZE, MAX_WAL_PAYLOAD_SIZE, RecordHeader, WAL_FORMAT_VERSION, WAL_MAGIC, + ENCRYPTED_FLAG, HEADER_SIZE, MAX_WAL_PAYLOAD_SIZE, NO_EVENT_SOURCE, RecordHeader, + WAL_FORMAT_VERSION, WAL_MAGIC, }; use crate::error::{Result, WalError}; use crate::preamble::PREAMBLE_SIZE; @@ -15,6 +16,37 @@ pub struct WalRecord { pub payload: Vec, } +/// The header fields of a record its caller decides: its type, scope and +/// event source. The writer assigns the LSN. +#[derive(Debug, Clone, Copy)] +pub struct RecordTarget { + pub record_type: u32, + pub tenant_id: u64, + pub vshard_id: u32, + pub database_id: u64, + /// The event source code of the row write the record carries. + /// [`NO_EVENT_SOURCE`] for a record that carries no row write. + pub event_source: u8, +} + +/// The header fields that tie a record to the write that appended it. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct RecordStamp { + /// The idempotency key of the replicated proposal whose apply appended + /// the record. `0` when no proposal apply appended it. + pub apply_key: u64, + /// The event source code of the row write the record carries. + pub event_source: u8, +} + +impl RecordStamp { + /// No proposal key and no row write. + pub const NONE: Self = Self { + apply_key: 0, + event_source: NO_EVENT_SOURCE, + }; +} + /// Parameters for [`WalRecord::new`]. pub struct WalRecordArgs<'a> { pub record_type: u32, @@ -43,6 +75,17 @@ impl WalRecord { /// zero-filled). Pre-existing records with zeros decode to `DatabaseId(0)` /// (the default database), preserving backward compatibility. pub fn new(args: WalRecordArgs<'_>) -> Result { + Self::new_stamped(args, RecordStamp::NONE) + } + + /// [`Self::new`] with the proposal key and event source of `stamp`. Both + /// ride the header, inside the CRC and the encryption AAD, so the record + /// and its stamp are durable together. + pub fn new_stamped(args: WalRecordArgs<'_>, stamp: RecordStamp) -> Result { + let RecordStamp { + apply_key, + event_source, + } = stamp; let WalRecordArgs { record_type, lsn, @@ -70,7 +113,8 @@ impl WalRecord { vshard_id, payload_len: 0, database_id, - reserved: [0u8; 8], + apply_key, + event_source, crc32c: 0, }; let header_bytes = temp_header.to_bytes(); @@ -98,7 +142,8 @@ impl WalRecord { vshard_id, payload_len: final_payload.len() as u32, database_id, - reserved: [0u8; 8], + apply_key, + event_source, crc32c: 0, }; @@ -110,6 +155,18 @@ impl WalRecord { }) } + /// The idempotency key of the proposal whose apply appended this record, + /// `0` when no proposal apply appended it. + pub fn apply_key(&self) -> u64 { + self.header.apply_key + } + + /// The event source code of the row write this record carries. + /// [`NO_EVENT_SOURCE`] for a record that carries no row write. + pub fn event_source(&self) -> u8 { + self.header.event_source + } + /// Decrypt the payload if the record is encrypted. /// /// `epoch` must come from the on-disk segment preamble, not from the @@ -340,4 +397,34 @@ mod tests { let decoded = LsnMsAnchorPayload::from_bytes(&record.payload).unwrap(); assert_eq!(decoded, anchor); } + + #[test] + fn a_stamped_record_keeps_its_event_source_and_checksum() { + for code in [NO_EVENT_SOURCE, 1, 2, 3, 4, 5, 6, u8::MAX] { + let record = WalRecord::new_stamped( + WalRecordArgs { + record_type: 1, + lsn: 9, + tenant_id: 1, + vshard_id: 0, + database_id: 0, + payload: b"row".to_vec(), + encryption_key: None, + preamble_bytes: None, + }, + RecordStamp { + apply_key: 7, + event_source: code, + }, + ) + .expect("record"); + assert_eq!(record.event_source(), code); + assert_eq!(record.apply_key(), 7); + record + .verify_checksum() + .expect("checksum covers the source"); + let decoded = RecordHeader::from_bytes(&record.header.to_bytes()); + assert_eq!(decoded.event_source, code); + } + } } diff --git a/nodedb-wal/src/replay/filter.rs b/nodedb-wal/src/replay/filter.rs index 65b6560ca..2d38cff8a 100644 --- a/nodedb-wal/src/replay/filter.rs +++ b/nodedb-wal/src/replay/filter.rs @@ -2,11 +2,22 @@ //! Collection-tombstone replay filter. //! -//! A [`TombstoneSet`] records, per `(database_id, tenant_id, collection)` tuple, -//! the highest `purge_lsn` observed in any `RecordType::CollectionTombstoned` -//! record. Replay consumers query [`TombstoneSet::is_tombstoned`] after -//! decoding the collection field from their payload; any record with -//! `lsn < purge_lsn` for a tombstoned pair MUST be skipped. +//! A [`TombstoneSet`] records, per `(database_id, tenant_id, collection)`, the +//! highest `purge_lsn` observed in any `RecordType::CollectionTombstoned` +//! record or persisted tombstone row. Replay consumers query +//! [`TombstoneSet::is_tombstoned`] after decoding the collection field from +//! their payload. A record with `lsn < purge_lsn` for a tombstoned collection +//! MUST be skipped. +//! +//! Two name forms meet here: +//! - Tombstones name a collection by its bare catalog name. They enter +//! through [`CollectionKey`]. +//! - Data records name it by the database-qualified storage name +//! ([`nodedb_types::QualifiedCollection`]), `"{database_id}/{name}"` outside the default +//! database. +//! +//! The set keys every entry by the storage name, derived from the key at +//! insert. A data record's name then matches with no parsing. //! //! Rationale: the WAL crate is payload-schema agnostic. Each engine's //! payload (vector / KV / document / graph / ...) has its own MessagePack @@ -15,18 +26,28 @@ use std::collections::HashMap; +use nodedb_types::{CollectionKey, DatabaseId}; + use crate::record::{RecordType, WalRecord, WriteAbortedPayload}; use crate::replay::aborted::AbortedWrites; use crate::tombstone::CollectionTombstonePayload; +/// One tombstoned collection: its bare catalog name and its purge boundary. +#[derive(Debug, Clone)] +struct Tombstone { + bare_name: String, + purge_lsn: u64, +} + /// In-memory index of active collection tombstones. /// -/// Keyed by `(database_id, tenant_id, collection)`; value is the `purge_lsn` -/// written at tombstone time. If the same object is tombstoned more than once -/// (re-create → re-drop in the same log), the highest `purge_lsn` wins. +/// Keyed by `(database_id, tenant_id, storage name)`. The value keeps the bare +/// catalog name and the `purge_lsn` written at tombstone time. If the same +/// collection is tombstoned more than once (re-create → re-drop in the same +/// log), the highest `purge_lsn` wins. #[derive(Debug, Default, Clone)] pub struct TombstoneSet { - entries: HashMap<(u64, u64, String), u64>, + entries: HashMap<(u64, u64, String), Tombstone>, } impl TombstoneSet { @@ -43,39 +64,73 @@ impl TombstoneSet { } } - /// Record a tombstone. If the object already has a higher `purge_lsn`, - /// the existing value is kept (idempotent, order-independent). - pub fn insert(&mut self, database_id: u64, tenant_id: u64, collection: String, purge_lsn: u64) { + /// Record a tombstone for the collection `key` names. If the collection + /// already has a higher `purge_lsn`, the existing value is kept + /// (idempotent, order-independent). + pub fn insert(&mut self, key: CollectionKey<'_>, tenant_id: u64, purge_lsn: u64) { + let storage = key.qualified(); + self.merge( + ( + key.database_id().as_u64(), + tenant_id, + storage.as_str().to_string(), + ), + Tombstone { + bare_name: key.name().to_string(), + purge_lsn, + }, + ); + } + + fn merge(&mut self, slot: (u64, u64, String), tombstone: Tombstone) { self.entries - .entry((database_id, tenant_id, collection)) + .entry(slot) .and_modify(|existing| { - if purge_lsn > *existing { - *existing = purge_lsn; + if tombstone.purge_lsn > existing.purge_lsn { + existing.purge_lsn = tombstone.purge_lsn; } }) - .or_insert(purge_lsn); + .or_insert(tombstone); } - /// Return `true` iff a write at `lsn` for the database-scoped object - /// is shadowed by a later tombstone and therefore must be skipped. + /// Return `true` iff a data record at `lsn` is shadowed by a later + /// tombstone and therefore must be skipped. + /// + /// `storage_name` is the collection name as the data record carries it: + /// the database-qualified storage name. pub fn is_tombstoned( &self, database_id: u64, tenant_id: u64, - collection: &str, + storage_name: &str, lsn: u64, ) -> bool { self.entries - .get(&(database_id, tenant_id, collection.to_string())) - .is_some_and(|&purge_lsn| lsn < purge_lsn) + .get(&(database_id, tenant_id, storage_name.to_string())) + .is_some_and(|tombstone| lsn < tombstone.purge_lsn) } - /// Return the `purge_lsn` for an object, if any. Primarily used by redb - /// persistence to serialize the current set after a replay pass. - pub fn purge_lsn(&self, database_id: u64, tenant_id: u64, collection: &str) -> Option { + /// [`Self::is_tombstoned`] for a record that names its collection by the + /// bare catalog name. + pub fn is_key_tombstoned(&self, key: CollectionKey<'_>, tenant_id: u64, lsn: u64) -> bool { + self.is_tombstoned( + key.database_id().as_u64(), + tenant_id, + key.qualified().as_str(), + lsn, + ) + } + + /// Return the `purge_lsn` for the collection `key` names, if any. + pub fn purge_lsn(&self, key: CollectionKey<'_>, tenant_id: u64) -> Option { + let storage = key.qualified(); self.entries - .get(&(database_id, tenant_id, collection.to_string())) - .copied() + .get(&( + key.database_id().as_u64(), + tenant_id, + storage.as_str().to_string(), + )) + .map(|tombstone| tombstone.purge_lsn) } pub fn len(&self) -> usize { @@ -86,18 +141,18 @@ impl TombstoneSet { self.entries.is_empty() } - /// Iterate over every `(database_id, tenant_id, collection, purge_lsn)`. + /// Iterate over every `(database_id, tenant_id, bare name, purge_lsn)`. pub fn iter(&self) -> impl Iterator + '_ { - self.entries - .iter() - .map(|((db, tid, name), lsn)| (*db, *tid, name.as_str(), *lsn)) + self.entries.iter().map(|((db, tid, _storage), tombstone)| { + (*db, *tid, tombstone.bare_name.as_str(), tombstone.purge_lsn) + }) } /// Merge another tombstone set into this one. Used when loading /// persisted tombstones from redb at startup before a fresh WAL pass. pub fn extend(&mut self, other: TombstoneSet) { - for ((db, tid, name), lsn) in other.entries { - self.insert(db, tid, name, lsn); + for (slot, tombstone) in other.entries { + self.merge(slot, tombstone); } } } @@ -109,9 +164,11 @@ pub struct DatabaseTombstones<'a> { } impl DatabaseTombstones<'_> { - pub fn is_tombstoned(&self, tenant_id: u64, collection: &str, lsn: u64) -> bool { + /// [`TombstoneSet::is_tombstoned`] in the bound database. `storage_name` + /// is the database-qualified name the data record carries. + pub fn is_tombstoned(&self, tenant_id: u64, storage_name: &str, lsn: u64) -> bool { self.set - .is_tombstoned(self.database_id, tenant_id, collection, lsn) + .is_tombstoned(self.database_id, tenant_id, storage_name, lsn) } } @@ -178,10 +235,13 @@ pub fn extract_replay_filters(records: &[WalRecord]) -> crate::Result { reject_if_encrypted(record, "collection-tombstone extraction")?; let payload = CollectionTombstonePayload::from_bytes(&record.payload)?; + // The payload names the collection by its bare catalog name. filters.tombstones.insert( - record.header.database_id, + CollectionKey::from_bare( + DatabaseId::new(record.header.database_id), + &payload.collection, + ), record.header.tenant_id, - payload.collection, payload.purge_lsn, ); } @@ -234,6 +294,8 @@ pub fn drop_aborted_records(records: Vec, aborted: &AbortedWrites) -> #[cfg(test)] mod tests { + use nodedb_types::QualifiedCollection; + use super::*; use crate::record::{WalRecord, WalRecordArgs}; @@ -260,26 +322,88 @@ mod tests { .unwrap() } + fn key(database: u64, name: &str) -> CollectionKey<'_> { + CollectionKey::from_bare(DatabaseId::new(database), name) + } + #[test] fn is_tombstoned_shadows_older_writes() { let mut set = TombstoneSet::new(); - set.insert(7, 1, "users".into(), 100); - assert!(set.is_tombstoned(7, 1, "users", 50)); - assert!(!set.is_tombstoned(7, 1, "users", 100)); - assert!(!set.is_tombstoned(7, 1, "users", 200)); - assert!(!set.is_tombstoned(7, 1, "other", 50)); - assert!(!set.is_tombstoned(7, 2, "users", 50)); - assert!(!set.is_tombstoned(8, 1, "users", 50)); + set.insert(key(7, "users"), 1, 100); + // Database 7 is a named database: its records carry "7/users". + assert!(set.is_tombstoned(7, 1, "7/users", 50)); + assert!(!set.is_tombstoned(7, 1, "7/users", 100)); + assert!(!set.is_tombstoned(7, 1, "7/users", 200)); + assert!(!set.is_tombstoned(7, 1, "7/other", 50)); + assert!(!set.is_tombstoned(7, 2, "7/users", 50)); + assert!(!set.is_tombstoned(8, 1, "8/users", 50)); + } + + /// A tombstone enters by its bare catalog name. A data record in a named + /// database carries the qualified storage name. The two must match. + #[test] + fn bare_tombstone_shadows_qualified_record_in_named_database() { + let db = DatabaseId::new(1024); + let mut set = TombstoneSet::new(); + set.insert(CollectionKey::from_bare(db, "orders"), 1, 100); + let storage = QualifiedCollection::new(db, "orders"); + assert_eq!(storage.as_str(), "1024/orders"); + assert!(set.is_tombstoned(1024, 1, storage.as_str(), 99)); + assert!( + set.for_database(1024) + .is_tombstoned(1, storage.as_str(), 99) + ); + assert!(!set.is_tombstoned(1024, 1, storage.as_str(), 100)); + // The bare string is not a storage name in a named database. + assert!(!set.is_tombstoned(1024, 1, "orders", 99)); + assert!(set.is_key_tombstoned(CollectionKey::from_bare(db, "orders"), 1, 99)); + } + + /// The default database stores names unqualified, so bare and storage + /// names are the same string there. + #[test] + fn default_database_tombstone_shadows_bare_record() { + let mut set = TombstoneSet::new(); + set.insert( + CollectionKey::from_bare(DatabaseId::DEFAULT, "orders"), + 1, + 100, + ); + assert!(set.is_tombstoned(0, 1, "orders", 99)); + assert!(set.for_database(0).is_tombstoned(1, "orders", 99)); + assert!(!set.is_tombstoned(0, 1, "orders", 100)); + assert!(set.is_key_tombstoned( + CollectionKey::from_bare(DatabaseId::DEFAULT, "orders"), + 1, + 99 + )); + } + + /// A WAL tombstone record carries the bare name. Extraction must shadow + /// the named database's qualified data records. + #[test] + fn extracted_tombstone_shadows_qualified_records() { + let set = extract_tombstones(&[tombstone_record(1024, 1, "orders", 100, 101)]).unwrap(); + assert!(set.is_tombstoned(1024, 1, "1024/orders", 50)); + assert!(!set.is_tombstoned(1024, 1, "orders", 50)); + } + + #[test] + fn iter_yields_bare_names() { + let mut set = TombstoneSet::new(); + set.insert(key(1024, "orders"), 1, 100); + let rows: Vec<_> = set.iter().collect(); + assert_eq!(rows, vec![(1024, 1, "orders", 100)]); } #[test] fn insert_keeps_highest_purge_lsn() { let mut set = TombstoneSet::new(); - set.insert(7, 1, "users".into(), 100); - set.insert(7, 1, "users".into(), 50); - assert_eq!(set.purge_lsn(7, 1, "users"), Some(100)); - set.insert(7, 1, "users".into(), 200); - assert_eq!(set.purge_lsn(7, 1, "users"), Some(200)); + set.insert(key(7, "users"), 1, 100); + set.insert(key(7, "users"), 1, 50); + assert_eq!(set.purge_lsn(key(7, "users"), 1), Some(100)); + set.insert(key(7, "users"), 1, 200); + assert_eq!(set.purge_lsn(key(7, "users"), 1), Some(200)); } #[test] @@ -291,9 +415,9 @@ mod tests { ]; let set = extract_tombstones(&records).unwrap(); assert_eq!(set.len(), 3); - assert_eq!(set.purge_lsn(7, 1, "users"), Some(100)); - assert_eq!(set.purge_lsn(7, 1, "orders"), Some(150)); - assert_eq!(set.purge_lsn(8, 1, "users"), Some(200)); + assert_eq!(set.purge_lsn(key(7, "users"), 1), Some(100)); + assert_eq!(set.purge_lsn(key(7, "orders"), 1), Some(150)); + assert_eq!(set.purge_lsn(key(8, "users"), 1), Some(200)); } #[test] @@ -388,7 +512,7 @@ mod tests { abort_record(12, 14), ]; let filters = extract_replay_filters(&records).unwrap(); - assert_eq!(filters.tombstones.purge_lsn(0, 1, "users"), Some(100)); + assert_eq!(filters.tombstones.purge_lsn(key(0, "users"), 1), Some(100)); assert_eq!(filters.aborted.len(), 2); assert!(filters.aborted.contains(10)); assert!(filters.aborted.contains(12)); @@ -440,12 +564,12 @@ mod tests { #[test] fn extend_merges_sets() { let mut a = TombstoneSet::new(); - a.insert(7, 1, "users".into(), 100); + a.insert(key(7, "users"), 1, 100); let mut b = TombstoneSet::new(); - b.insert(7, 1, "users".into(), 150); - b.insert(8, 1, "orders".into(), 200); + b.insert(key(7, "users"), 1, 150); + b.insert(key(8, "orders"), 1, 200); a.extend(b); - assert_eq!(a.purge_lsn(7, 1, "users"), Some(150)); - assert_eq!(a.purge_lsn(8, 1, "orders"), Some(200)); + assert_eq!(a.purge_lsn(key(7, "users"), 1), Some(150)); + assert_eq!(a.purge_lsn(key(8, "orders"), 1), Some(200)); } } diff --git a/nodedb-wal/src/segmented.rs b/nodedb-wal/src/segmented.rs index 1ee5bbe4f..4a31e212e 100644 --- a/nodedb-wal/src/segmented.rs +++ b/nodedb-wal/src/segmented.rs @@ -23,7 +23,7 @@ use tracing::info; use crate::crypto::KeyRing; use crate::error::{Result, WalError}; -use crate::record::WalRecord; +use crate::record::{RecordTarget, WalRecord}; use crate::segment::{ DEFAULT_SEGMENT_TARGET_SIZE, SegmentContinuity, SegmentMeta, TruncateResult, check_retained_floor, discover_segments, segment_path, truncate_segments, @@ -190,14 +190,35 @@ impl SegmentedWal { vshard_id: u32, database_id: u64, payload: &[u8], + ) -> Result { + self.append_keyed( + RecordTarget { + record_type, + tenant_id, + vshard_id, + database_id, + event_source: crate::record::NO_EVENT_SOURCE, + }, + payload, + 0, + ) + } + + /// [`Self::append`] for a record appended by the apply of the replicated + /// proposal `apply_key`, carrying the event source in `target` (see + /// [`crate::WalRecord::new_stamped`]). + pub fn append_keyed( + &mut self, + target: RecordTarget, + payload: &[u8], + apply_key: u64, ) -> Result { // Check if we need to roll to a new segment. if self.writer.file_offset() >= self.segment_target_size { self.roll_segment()?; } - self.writer - .append(record_type, tenant_id, vshard_id, database_id, payload) + self.writer.append_keyed(target, payload, apply_key) } /// Flush all buffered records and fsync the active segment. diff --git a/nodedb-wal/src/writer/core.rs b/nodedb-wal/src/writer/core.rs index 9dde0f137..a311adef2 100644 --- a/nodedb-wal/src/writer/core.rs +++ b/nodedb-wal/src/writer/core.rs @@ -9,7 +9,7 @@ use crate::align::AlignedBuf; use crate::double_write::{DoubleWriteBuffer, DwbProtection}; use crate::error::{Result, WalError}; use crate::preamble::SegmentPreamble; -use crate::record::{HEADER_SIZE, MIN_PADDING_RECORD_SIZE, WalRecord, WalRecordArgs}; +use crate::record::{HEADER_SIZE, MIN_PADDING_RECORD_SIZE, RecordTarget, WalRecord, WalRecordArgs}; use super::config::{WalWriterConfig, open_dwb_for, resume_offset}; use super::durability::{DurabilityState, fsync_and_track}; @@ -254,6 +254,35 @@ impl WalWriter { database_id: u64, payload: &[u8], ) -> Result { + self.append_keyed( + RecordTarget { + record_type, + tenant_id, + vshard_id, + database_id, + event_source: crate::record::NO_EVENT_SOURCE, + }, + payload, + 0, + ) + } + + /// [`Self::append`] for a record appended by the apply of the replicated + /// proposal `apply_key`, carrying the event source in `target` (see + /// [`WalRecord::new_stamped`]). + pub fn append_keyed( + &mut self, + target: RecordTarget, + payload: &[u8], + apply_key: u64, + ) -> Result { + let RecordTarget { + record_type, + tenant_id, + vshard_id, + database_id, + event_source, + } = target; if self.sealed { return Err(WalError::Sealed); } @@ -261,16 +290,22 @@ impl WalWriter { let lsn = self.next_lsn.load(Ordering::Relaxed); let preamble_bytes = self.segment_preamble.as_ref().map(|p| p.to_bytes()); - let record = WalRecord::new(WalRecordArgs { - record_type, - lsn, - tenant_id, - vshard_id, - database_id, - payload: payload.to_vec(), - encryption_key: self.encryption_ring.as_ref().map(|r| r.current()), - preamble_bytes: preamble_bytes.as_ref(), - })?; + let record = WalRecord::new_stamped( + WalRecordArgs { + record_type, + lsn, + tenant_id, + vshard_id, + database_id, + payload: payload.to_vec(), + encryption_key: self.encryption_ring.as_ref().map(|r| r.current()), + preamble_bytes: preamble_bytes.as_ref(), + }, + crate::record::RecordStamp { + apply_key, + event_source, + }, + )?; let header_bytes = record.header.to_bytes(); let total_size = HEADER_SIZE + record.payload.len(); diff --git a/nodedb-wal/src/writer/flush.rs b/nodedb-wal/src/writer/flush.rs index aac2c1cdb..b410b18d5 100644 --- a/nodedb-wal/src/writer/flush.rs +++ b/nodedb-wal/src/writer/flush.rs @@ -1,10 +1,8 @@ // SPDX-License-Identifier: Apache-2.0 -use crate::error::Result; -// Only the unix `pwrite` path constructs an error value directly; elsewhere -// failures propagate as `Result` from calls that build their own. -#[cfg(unix)] -use crate::error::WalError; +use std::os::unix::io::AsRawFd; + +use crate::error::{Result, WalError}; use super::core::WalWriter; @@ -43,40 +41,36 @@ impl WalWriter { }; // Use pwrite to write at the exact offset, retrying on short writes. - #[cfg(unix)] - { - use std::os::unix::io::AsRawFd; - let fd = self.file.as_raw_fd(); - let mut remaining = data; - let mut write_offset = self.file_offset; - while !remaining.is_empty() { - let written = unsafe { - libc::pwrite( - fd, - remaining.as_ptr() as *const libc::c_void, - remaining.len(), - write_offset as libc::off_t, - ) - }; - if written < 0 { - return Err(write_error( - "WAL segment append", - write_offset, - remaining.len() as u64, - )); - } - let n = written as usize; - if n == 0 { - // A zero-length write makes no progress; retrying would - // spin forever. - return Err(WalError::Io(std::io::Error::new( - std::io::ErrorKind::WriteZero, - "WAL pwrite made no progress", - ))); - } - remaining = &remaining[n..]; - write_offset += n as u64; + let fd = self.file.as_raw_fd(); + let mut remaining = data; + let mut write_offset = self.file_offset; + while !remaining.is_empty() { + let written = unsafe { + libc::pwrite( + fd, + remaining.as_ptr() as *const libc::c_void, + remaining.len(), + write_offset as libc::off_t, + ) + }; + if written < 0 { + return Err(write_error( + "WAL segment append", + write_offset, + remaining.len() as u64, + )); + } + let n = written as usize; + if n == 0 { + // A zero-length write makes no progress; retrying would + // spin forever. + return Err(WalError::Io(std::io::Error::new( + std::io::ErrorKind::WriteZero, + "WAL pwrite made no progress", + ))); } + remaining = &remaining[n..]; + write_offset += n as u64; } self.file_offset += data.len() as u64; @@ -97,19 +91,12 @@ impl WalWriter { /// writes rather than treat it as a passing error. `offset` and `pending` say /// where the batch stalled and how much of it never reached the file, which is /// what a report needs to describe the write that could not complete. -/// -/// Gated to match its only call site: the `pwrite` loop is unix-only, so on -/// other targets (wasm32) this would be dead code and a `-D warnings` build -/// would reject it. -#[cfg(unix)] fn write_error(context: &'static str, offset: u64, pending: u64) -> WalError { let err = std::io::Error::last_os_error(); - #[cfg(unix)] if err.raw_os_error() == Some(libc::ENOSPC) { let out_of_space = WalError::OutOfSpace { context }; crate::diag::out_of_space(&out_of_space, context, offset, pending); return out_of_space; } - let _ = (context, offset, pending); WalError::Io(err) } diff --git a/nodedb-wal/tests/wal_suite/cases/encrypted_replay.rs b/nodedb-wal/tests/wal_suite/cases/encrypted_replay.rs index 1b890a232..4ad9223e3 100644 --- a/nodedb-wal/tests/wal_suite/cases/encrypted_replay.rs +++ b/nodedb-wal/tests/wal_suite/cases/encrypted_replay.rs @@ -9,6 +9,7 @@ //! and the free `replay_*` drivers) rather than decrypting by hand, so they pin //! the behaviour where consumers actually observe it. +use nodedb_types::{CollectionKey, DatabaseId}; use nodedb_wal::crypto::{KeyRing, WalEncryptionKey}; use nodedb_wal::mmap_reader::replay_segments_mmap; use nodedb_wal::record::RecordType; @@ -110,7 +111,10 @@ fn tombstones_are_extracted_from_an_encrypted_wal() { let set = extract_tombstones(&records) .expect("a tombstone written to an encrypted WAL must still be extractable"); - assert_eq!(set.purge_lsn(0, 1, "users"), Some(42)); + assert_eq!( + set.purge_lsn(CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), 1), + Some(42) + ); } #[test] diff --git a/nodedb-wal/tests/wal_suite/cases/faultbox_disabled.rs b/nodedb-wal/tests/wal_suite/cases/faultbox_disabled.rs index c43d89583..01bb72184 100644 --- a/nodedb-wal/tests/wal_suite/cases/faultbox_disabled.rs +++ b/nodedb-wal/tests/wal_suite/cases/faultbox_disabled.rs @@ -4,9 +4,8 @@ //! did before it existed. //! //! The report sites sit on error paths that must keep working for embedders who -//! never enable `diagnostics`, including wasm32 builds where the feature is -//! inert even when it is on. A recorder that changed an error, swallowed one, or -//! panicked at a detection site would be worse than no recorder at all. +//! never enable `diagnostics`. A recorder that changed an error, swallowed one, +//! or panicked at a detection site would be worse than no recorder at all. use std::io::{Seek, SeekFrom, Write}; use std::path::Path; diff --git a/nodedb-wal/tests/wal_suite/cases/mod.rs b/nodedb-wal/tests/wal_suite/cases/mod.rs index 515e7ce35..a8ace704b 100644 --- a/nodedb-wal/tests/wal_suite/cases/mod.rs +++ b/nodedb-wal/tests/wal_suite/cases/mod.rs @@ -12,7 +12,6 @@ mod encrypted_replay; mod faultbox_corruption_report; #[cfg(not(feature = "diagnostics"))] mod faultbox_disabled; -#[cfg(not(target_arch = "wasm32"))] mod mmap_reader_madvise; #[cfg(all(feature = "io-uring", target_os = "linux"))] mod o_direct_alignment; diff --git a/nodedb-wal/tests/wal_suite/cases/wal_collection_tombstone.rs b/nodedb-wal/tests/wal_suite/cases/wal_collection_tombstone.rs index 08feb85f8..8c2cef296 100644 --- a/nodedb-wal/tests/wal_suite/cases/wal_collection_tombstone.rs +++ b/nodedb-wal/tests/wal_suite/cases/wal_collection_tombstone.rs @@ -12,6 +12,7 @@ //! 4. [`TombstoneSet::is_tombstoned`] against realistic `(tenant, //! collection, lsn)` tuples. +use nodedb_types::{CollectionKey, DatabaseId}; use nodedb_wal::reader::WalReader; use nodedb_wal::record::RecordType; use nodedb_wal::writer::WalWriter; @@ -98,7 +99,10 @@ fn extract_and_shadow_writes_before_purge_lsn() { let set: TombstoneSet = extract_tombstones(&records).unwrap(); assert_eq!(set.len(), 1); - assert_eq!(set.purge_lsn(0, 1, "users"), Some(purge_lsn)); + assert_eq!( + set.purge_lsn(CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), 1), + Some(purge_lsn) + ); for lsn in &put_lsns { assert!( @@ -160,7 +164,10 @@ fn multiple_tombstones_keep_highest_purge_lsn() { let records = read_all(&path); let set = extract_tombstones(&records).unwrap(); - assert_eq!(set.purge_lsn(0, 1, "users"), Some(500)); + assert_eq!( + set.purge_lsn(CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), 1), + Some(500) + ); assert!(set.is_tombstoned(0, 1, "users", 499)); assert!(!set.is_tombstoned(0, 1, "users", 500)); } diff --git a/nodedb/src/bootstrap/cluster_ready.rs b/nodedb/src/bootstrap/cluster_ready.rs index 731445079..b4e549d6a 100644 --- a/nodedb/src/bootstrap/cluster_ready.rs +++ b/nodedb/src/bootstrap/cluster_ready.rs @@ -159,6 +159,17 @@ pub async fn await_cluster_ready( )); } + // Grants and hierarchy edges live in collections the data groups just + // finished replaying. Load them before the gateway opens, so no statement + // plans against an empty permission cache. + if let Err(error) = crate::bootstrap::permission_tree_load::load_permission_trees(shared).await + { + data_groups_gate.fail(format!("permission tree load failed: {error}")); + return Err(anyhow::anyhow!( + "permission tree load failed during startup: {error}" + )); + } + data_groups_gate.fire(); transport_gate.fire(); @@ -192,6 +203,26 @@ pub async fn await_cluster_ready( } warm_peers_gate.fire(); health_loop_gate.fire(); + + // In a cluster, a node plans permission-checked statements only under + // an authorization lease. The renewal loop runs from Raft start, and + // every input of a grant is live by now: the Raft groups, the replayed + // data groups, the permission cache and the Event Plane. The gateway + // opens once the first lease is granted, so the first statements are + // not refused. + if let Some(timing) = shared.authorization_fence.timing() + && let Err(error) = crate::control::security::auth_lease::await_planning_admitted( + shared, + RAFT_READY_STALL_TIMEOUT, + timing.renew_every, + ) + .await + { + gateway_enable_gate.fail(format!("authorization lease not granted: {error}")); + return Err(anyhow::anyhow!( + "authorization lease not granted during startup: {error}" + )); + } gateway_enable_gate.fire(); Ok(()) diff --git a/nodedb/src/bootstrap/constraint_reconcile.rs b/nodedb/src/bootstrap/constraint_reconcile.rs index d766099a5..5feb92867 100644 --- a/nodedb/src/bootstrap/constraint_reconcile.rs +++ b/nodedb/src/bootstrap/constraint_reconcile.rs @@ -153,7 +153,9 @@ pub async fn reconcile_once( continue; } - let vshard_id = nodedb_cluster::routing::vshard_for_collection(database_id, &stored.name); + let vshard_id = nodedb_cluster::routing::vshard_for_collection( + nodedb_types::CollectionKey::from_bare(database_id, &stored.name), + ); let entry = ReplicatedEntry::new( stored.tenant_id, database_id.as_u64(), diff --git a/nodedb/src/bootstrap/data_plane.rs b/nodedb/src/bootstrap/data_plane.rs index c6c5803ff..f731db951 100644 --- a/nodedb/src/bootstrap/data_plane.rs +++ b/nodedb/src/bootstrap/data_plane.rs @@ -338,6 +338,7 @@ pub fn spawn_data_plane_cores( query: config.tuning.query.clone(), graph: config.tuning.graph.clone(), timeseries: config.tuning.timeseries.clone(), + vector: config.tuning.vector.clone(), checkpoint_interval: std::time::Duration::from_secs(config.checkpoint.interval_secs), }; diff --git a/nodedb/src/bootstrap/listeners.rs b/nodedb/src/bootstrap/listeners.rs index 49d3509b9..1200fc300 100644 --- a/nodedb/src/bootstrap/listeners.rs +++ b/nodedb/src/bootstrap/listeners.rs @@ -13,18 +13,19 @@ use crate::control::cluster::ClusterHandle; use crate::control::server::ilp_listener::IlpListener; use crate::control::server::listener::Listener; use crate::control::server::pgwire::listener::PgListener; +use crate::control::server::reserved_socket::ReservedSocket; use crate::control::server::resp::RespListener; use crate::control::shutdown::ShutdownBus; -use crate::control::startup::StartupGate; +use crate::control::startup::{ReadyGate, StartupGate}; use crate::control::state::SharedState; -/// The pre-bound protocol listeners passed to [`spawn_protocol_listeners`]. +/// The listening client-protocol sockets passed to +/// [`spawn_protocol_listeners`]. /// -/// Every socket here is already bound by [`bind_listeners`], so spawning -/// cannot fail on a port conflict. +/// Every socket here was bound by [`bind_listeners`] and opened by +/// [`open_listeners`], so spawning cannot fail on a port conflict. pub struct ProtocolListeners { pub pg_listener: PgListener, - pub http_listener: TcpListener, pub sync_listener: TcpListener, pub ilp_listener: Option, pub resp_listener: Option, @@ -37,14 +38,16 @@ pub struct ListenerInfra { pub shutdown_bus: ShutdownBus, } -/// Spawn all non-native protocol listeners as background tasks. +/// Spawn all non-native client-protocol listeners as background tasks. /// /// The native listener is not spawned here — it is run on the main task -/// by the caller after this returns. +/// by the caller after this returns. The HTTP server is spawned earlier by +/// [`spawn_http_listener`]. /// /// Infallible by construction: every socket was bound by [`bind_listeners`] -/// before this point, so a port conflict has already aborted boot while -/// nothing was exposed. Nothing here may silently swallow a bind failure. +/// and opened by [`open_listeners`] before this point, so a port conflict +/// has already aborted boot. Nothing here may silently swallow a bind +/// failure. pub async fn spawn_protocol_listeners( listeners: ProtocolListeners, shared: Arc, @@ -55,7 +58,6 @@ pub async fn spawn_protocol_listeners( ) { let ProtocolListeners { pg_listener, - http_listener, sync_listener, ilp_listener, resp_listener, @@ -70,7 +72,6 @@ pub async fn spawn_protocol_listeners( }; let tls_flags = config.server.tls.as_ref(); let pgwire_tls_enabled = tls_flags.is_some_and(|t| t.pgwire); - let http_tls_enabled = tls_flags.is_some_and(|t| t.http); let resp_tls_enabled = tls_flags.is_some_and(|t| t.resp); let ilp_tls_enabled = tls_flags.is_some_and(|t| t.ilp); @@ -97,29 +98,6 @@ pub async fn spawn_protocol_listeners( } }); - // HTTP API server (on the socket bound by `bind_listeners`). - let shared_http = Arc::clone(&shared); - let http_auth_mode = config.auth.mode.clone(); - let http_tls = if http_tls_enabled { - config.server.tls.clone() - } else { - None - }; - let bus_http = shutdown_bus.clone(); - tokio::spawn(async move { - if let Err(e) = crate::control::server::http::server::run( - http_listener, - shared_http, - http_auth_mode, - http_tls.as_ref(), - bus_http, - ) - .await - { - tracing::error!(error = %e, "HTTP API server failed"); - } - }); - // ILP TCP listener (if configured). if let Some(ilp) = ilp_listener { let shared_ilp = Arc::clone(&shared); @@ -194,11 +172,61 @@ pub async fn spawn_protocol_listeners( nodedb_cluster::readiness::notify_ready(); } -/// Every protocol socket, bound before any accept loop starts. +/// Spawn the HTTP API server on `http_listener`. +/// +/// Boot calls this before it waits for the node to become ready, so +/// orchestrator probes can watch startup. Until the `Serving` phase only the +/// probe and metrics routes answer (see +/// `control::server::http::startup_gate`). +pub fn spawn_http_listener( + http_listener: TcpListener, + shared: Arc, + config: &ServerConfig, + shutdown_bus: ShutdownBus, +) { + let http_auth_mode = config.auth.mode.clone(); + let http_tls = config.server.tls.as_ref().filter(|tls| tls.http).cloned(); + tokio::spawn(async move { + if let Err(e) = crate::control::server::http::server::run( + http_listener, + shared, + http_auth_mode, + http_tls.as_ref(), + shutdown_bus, + ) + .await + { + tracing::error!(error = %e, "HTTP API server failed"); + } + }); +} + +/// Every protocol socket, bound before the node waits to become ready. +/// +/// The HTTP socket listens from the start, so probes answer during boot. The +/// client-protocol sockets are bound but not listening. pub struct BoundListeners { + pub http: TcpListener, + pub clients: ClientSockets, +} + +/// The client-protocol sockets, bound but not yet listening. +/// +/// A bound socket that does not listen refuses each connection attempt at +/// once. Boot listens through [`open_listeners`] only once the node can +/// serve, so no client waits in a kernel accept queue through boot. +pub struct ClientSockets { + pub native: ReservedSocket, + pub pgwire: ReservedSocket, + pub sync: ReservedSocket, + pub ilp: Option, + pub resp: Option, +} + +/// Every client-protocol socket, listening. +pub struct OpenListeners { pub native: Listener, pub pgwire: PgListener, - pub http: TcpListener, pub sync: TcpListener, pub ilp: Option, pub resp: Option, @@ -209,35 +237,82 @@ pub struct BoundListeners { /// This is the single fail-fast point for listener setup: it runs before the /// node waits on cluster readiness and before any accept loop is spawned, so /// a port conflict on *any* protocol — including HTTP and sync, which serve -/// from detached tasks — aborts boot while nothing is exposed yet. Never -/// move a bind out of here into a spawned task; that is how a server ends up -/// running for days missing a listener behind one warning line. -pub async fn bind_listeners(config: &ServerConfig) -> anyhow::Result { - let native = crate::control::server::listener::Listener::bind(config.native_addr()).await?; - let pgwire = - crate::control::server::pgwire::listener::PgListener::bind(config.pgwire_addr()).await?; - let http = TcpListener::bind(config.http_addr()) - .await - .with_context(|| format!("bind HTTP API listener to {}", config.http_addr()))?; - let sync = crate::control::server::sync::listener::bind_sync_listener(config.sync_addr()) - .await - .context("sync listener failed to bind")?; - let ilp = if let Some(ilp_addr) = config.ilp_addr() { - Some(crate::control::server::ilp_listener::IlpListener::bind(ilp_addr).await?) - } else { - None - }; - let resp = if let Some(resp_addr) = config.resp_addr() { - Some(crate::control::server::resp::RespListener::bind(resp_addr).await?) - } else { - None +/// from detached tasks — aborts boot early. Never move a bind out of here +/// into a spawned task; that is how a server ends up running for days +/// missing a listener behind one warning line. +pub fn bind_listeners(config: &ServerConfig) -> anyhow::Result { + let reserve = |name: &str, addr: std::net::SocketAddr| { + ReservedSocket::bind(addr).with_context(|| format!("bind {name} listener to {addr}")) }; + let http = reserve("HTTP API", config.http_addr())? + .listen() + .context("listen on the HTTP API address")?; + let native = reserve("native protocol", config.native_addr())?; + let pgwire = reserve("pgwire", config.pgwire_addr())?; + let sync = crate::control::server::sync::listener::reserve_sync_listener(config.sync_addr()) + .context("sync listener failed to bind")?; + let ilp = config + .ilp_addr() + .map(|addr| reserve("ILP", addr)) + .transpose()?; + let resp = config + .resp_addr() + .map(|addr| reserve("RESP", addr)) + .transpose()?; Ok(BoundListeners { + http, + clients: ClientSockets { + native, + pgwire, + sync, + ilp, + resp, + }, + }) +} + +/// Start listening on every client-protocol socket, then fire +/// `serving_gate`. +/// +/// Boot calls this once the node can serve and before any accept loop is +/// spawned. The gate advances the startup sequencer to +/// [`StartupPhase::Serving`](crate::control::startup::StartupPhase::Serving), +/// which opens every HTTP route and lets `/healthz` report `ok`, at the same +/// point the client protocols start listening. A socket that cannot listen +/// fails the gate and aborts boot: another process started listening on its +/// address after the bind. +pub fn open_listeners( + clients: ClientSockets, + serving_gate: ReadyGate, +) -> anyhow::Result { + let ClientSockets { native, pgwire, - http, sync, ilp, resp, - }) + } = clients; + let open = || -> crate::Result { + Ok(OpenListeners { + native: Listener::from_listener(native.listen()?)?, + pgwire: PgListener::from_listener(pgwire.listen()?)?, + sync: sync.listen()?, + ilp: ilp + .map(|socket| IlpListener::from_listener(socket.listen()?)) + .transpose()?, + resp: resp + .map(|socket| RespListener::from_listener(socket.listen()?)) + .transpose()?, + }) + }; + match open() { + Ok(open) => { + serving_gate.fire(); + Ok(open) + } + Err(error) => { + serving_gate.fail(error.to_string()); + Err(error.into()) + } + } } diff --git a/nodedb/src/bootstrap/mod.rs b/nodedb/src/bootstrap/mod.rs index 8faa98793..7b0ea34aa 100644 --- a/nodedb/src/bootstrap/mod.rs +++ b/nodedb/src/bootstrap/mod.rs @@ -12,6 +12,7 @@ pub mod diagnostics; pub mod index_registry_seed; pub mod listeners; pub mod panic_hook; +pub mod permission_tree_load; pub mod quota_replay; pub mod schema_rehydrate; pub mod signal; diff --git a/nodedb/src/bootstrap/permission_tree_load.rs b/nodedb/src/bootstrap/permission_tree_load.rs new file mode 100644 index 000000000..4e8272f92 --- /dev/null +++ b/nodedb/src/bootstrap/permission_tree_load.rs @@ -0,0 +1,24 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Load permission-tree state before the gateway serves. +//! +//! The permission cache is in-memory. After a restart it holds the tree +//! definitions from the catalog, but none of the hierarchy edges or grants +//! stored in the source collections. This step reads them, once every data +//! group finished its replay, so the first statement plans against the +//! grants that were in force before the restart. + +use std::sync::Arc; + +use crate::control::security::permission_tree::reload; +use crate::control::state::SharedState; + +/// Apply the tree-definition changes the metadata replay committed, then +/// load every tree's edges and grants from this node's cores. +pub async fn load_permission_trees(shared: &Arc) -> crate::Result<()> { + { + let mut cache = shared.permission_cache.write().await; + shared.authorization_fence.tree_defs().apply_to(&mut cache); + } + reload::reload_all(shared, None).await +} diff --git a/nodedb/src/bootstrap/state_wiring.rs b/nodedb/src/bootstrap/state_wiring.rs index 0b5034c87..4c3bb1173 100644 --- a/nodedb/src/bootstrap/state_wiring.rs +++ b/nodedb/src/bootstrap/state_wiring.rs @@ -43,10 +43,16 @@ pub async fn wire_state( array_catalog, maintenance_budget, } = components; - // Install startup gate. - if let Some(state) = Arc::get_mut(shared) { - state.startup = Arc::clone(startup_gate); - } + // Install startup gate. `/healthz` and the HTTP startup gate read it, so + // a state left on the test helpers' pre-fired gate would report ready + // and open every route during boot. + Arc::get_mut(shared) + .ok_or_else(|| { + anyhow::anyhow!( + "startup gate: SharedState is already shared before the gate was installed" + ) + })? + .startup = Arc::clone(startup_gate); // Replay surrogate WAL records. // Note: wal_records are not passed here — caller must handle surrogate replay @@ -244,24 +250,10 @@ pub async fn wire_state( state.scheduler_config = config.scheduler.clone(); } - // Construct and install the gateway + DDL plan-cache invalidator. - // - // `Gateway` holds a `Weak` back-reference to its own - // `SharedState`, so it cannot be installed via `Arc::get_mut` (which - // requires strong count 1 AND weak count 0 — the gateway's own `Weak` - // violates the latter). `gateway`/`gateway_invalidator` are therefore - // `OnceLock`s, set through `&self` exactly once here at boot. - // - // That weak reference outlives this call, so every `Arc::get_mut` install - // above depends on running BEFORE this block: one placed after it no-ops. - { - let gateway = Arc::new(crate::control::gateway::Gateway::new(Arc::clone(shared))); - let invalidator = Arc::new(crate::control::gateway::PlanCacheInvalidator::new( - &gateway.plan_cache, - )); - let _ = shared.gateway.set(gateway); - let _ = shared.gateway_invalidator.set(invalidator); - } + // The gateway's weak back-reference outlives this call, so every + // `Arc::get_mut` install above must run BEFORE it: one placed after + // it no-ops. + install_gateway(shared); // Hydrate bitemporal retention registry from array catalog. { @@ -301,3 +293,21 @@ pub async fn wire_state( Ok(()) } + +/// Construct and install the gateway and the DDL plan-cache invalidator. +/// +/// `Gateway` holds a `Weak` back-reference to its own +/// `SharedState`. `Arc::get_mut` requires strong count 1 and weak count 0, +/// so it cannot install the gateway. `gateway`/`gateway_invalidator` are +/// `OnceLock`s instead, set through `&self` exactly once. +/// +/// Every `Arc::get_mut` install on `shared` must run before this call. +/// A later `get_mut` sees the weak reference and no-ops. +pub fn install_gateway(shared: &Arc) { + let gateway = Arc::new(crate::control::gateway::Gateway::new(Arc::clone(shared))); + let invalidator = Arc::new(crate::control::gateway::PlanCacheInvalidator::new( + &gateway.plan_cache, + )); + let _ = shared.gateway.set(gateway); + let _ = shared.gateway_invalidator.set(invalidator); +} diff --git a/nodedb/src/bridge/admission_chokepoint.rs b/nodedb/src/bridge/admission_chokepoint.rs index ab130da81..4927c17b9 100644 --- a/nodedb/src/bridge/admission_chokepoint.rs +++ b/nodedb/src/bridge/admission_chokepoint.rs @@ -108,6 +108,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }) } @@ -216,6 +217,7 @@ mod tests { rls_write_check: check, returning: None, rls_filters: Vec::new(), + provenance: None, }) } diff --git a/nodedb/src/bridge/dispatch/closed_lsns.rs b/nodedb/src/bridge/dispatch/closed_lsns.rs new file mode 100644 index 000000000..bc21c9757 --- /dev/null +++ b/nodedb/src/bridge/dispatch/closed_lsns.rs @@ -0,0 +1,128 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The LSNs above the outcome floor whose last owner closed, kept as +//! disjoint ranges. +//! +//! A closed LSN's outcome is final, so a resend of it is refused. The floor +//! prunes every range at or below it. A held window keeps the floor down for +//! the rest of the process, so the set must stay bounded without the floor. +//! +//! A range grows across a gap of LSNs no live window owns. Such an LSN is +//! either closed already, or was never owned: its record was minted outside +//! a write window and must never reach a core, so a resend of it is refused +//! too. The ranges are therefore split only at LSNs a live window owns, and +//! their count is at most one more than the number of live owned LSNs. + +use std::collections::BTreeMap; + +/// Closed LSNs as `start -> end` ranges, both ends inclusive. +#[derive(Debug, Default)] +pub(super) struct ClosedLsns { + ranges: BTreeMap, +} + +impl ClosedLsns { + /// Record that `lsn`'s last owner closed. `owners` holds every LSN a live + /// window owns. + pub(super) fn insert(&mut self, lsn: u64, owners: &BTreeMap) { + if self.contains(lsn) { + return; + } + let mut start = lsn; + let mut end = lsn; + if let Some((&prev_start, &prev_end)) = self.ranges.range(..lsn).next_back() + && owners + .range(prev_end.saturating_add(1)..lsn) + .next() + .is_none() + { + self.ranges.remove(&prev_start); + start = prev_start; + } + if let Some((&next_start, &next_end)) = self.ranges.range(lsn.saturating_add(1)..).next() + && owners + .range(lsn.saturating_add(1)..next_start) + .next() + .is_none() + { + self.ranges.remove(&next_start); + end = next_end; + } + self.ranges.insert(start, end); + } + + /// Whether `lsn` lies in a closed range. + pub(super) fn contains(&self, lsn: u64) -> bool { + self.ranges + .range(..=lsn) + .next_back() + .is_some_and(|(_, end)| lsn <= *end) + } + + /// Drop every LSN at or below `floor`: the floor refuses those itself. + pub(super) fn prune_through(&mut self, floor: u64) { + let above = floor.saturating_add(1); + let straddling = self + .ranges + .range(..above) + .next_back() + .map(|(_, end)| *end) + .filter(|end| *end >= above); + self.ranges = self.ranges.split_off(&above); + if let Some(end) = straddling { + self.ranges.insert(above, end); + } + } +} + +#[cfg(test)] +impl ClosedLsns { + /// Number of stored ranges. + pub(super) fn range_count(&self) -> usize { + self.ranges.len() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn adjacent_and_unowned_gaps_merge_into_one_range() { + let owners = BTreeMap::new(); + let mut closed = ClosedLsns::default(); + for lsn in [5, 7, 6, 10] { + closed.insert(lsn, &owners); + } + assert_eq!(closed.range_count(), 1); + assert!( + closed.contains(8), + "an unowned gap LSN is refused like a closed one" + ); + assert!(!closed.contains(11)); + } + + #[test] + fn a_live_owned_lsn_splits_the_ranges() { + let owners = BTreeMap::from([(8, 1)]); + let mut closed = ClosedLsns::default(); + closed.insert(5, &owners); + closed.insert(10, &owners); + assert_eq!(closed.range_count(), 2); + assert!(!closed.contains(8)); + } + + #[test] + fn pruning_keeps_only_lsns_above_the_floor() { + let owners = BTreeMap::new(); + let mut closed = ClosedLsns::default(); + closed.insert(5, &owners); + closed.insert(9, &owners); + closed.prune_through(7); + assert!(!closed.contains(7)); + assert!(closed.contains(8)); + assert!(closed.contains(9)); + closed.prune_through(9); + assert_eq!(closed.range_count(), 0); + } +} diff --git a/nodedb/src/bridge/dispatch/core_channel.rs b/nodedb/src/bridge/dispatch/core_channel.rs index c462d4b04..e093d978b 100644 --- a/nodedb/src/bridge/dispatch/core_channel.rs +++ b/nodedb/src/bridge/dispatch/core_channel.rs @@ -11,6 +11,7 @@ use nodedb_bridge::wfq::WeightedFairQueue; use crate::bridge::envelope; use crate::data::eventfd::EventFdNotifier; +use crate::types::Lsn; use super::dispatcher::{BridgeRequest, BridgeResponse}; @@ -88,7 +89,10 @@ impl CoreChannel { /// /// The disconnect is also checked before the first pop, so a core known /// to be dead never has another request moved into a doomed `try_push`. - pub(super) fn flush_wfq(&mut self) -> usize { + /// + /// Every pushed request carries `outcome_floor`, the floor the caller read + /// just before this flush. + pub(super) fn flush_wfq(&mut self, outcome_floor: Lsn) -> usize { let mut flushed = 0; if self.request_tx.is_disconnected() { return 0; @@ -99,7 +103,10 @@ impl CoreChannel { }; let db_id = req.database_id.as_u64(); let req_id = req.request_id.as_u64(); - match self.request_tx.try_push(BridgeRequest { inner: req }) { + match self.request_tx.try_push(BridgeRequest { + inner: req, + outcome_floor, + }) { Ok(()) => { flushed += 1; self.update_db_pressure(db_id); diff --git a/nodedb/src/bridge/dispatch/dispatched_lsns.rs b/nodedb/src/bridge/dispatch/dispatched_lsns.rs new file mode 100644 index 000000000..708931ab3 --- /dev/null +++ b/nodedb/src/bridge/dispatch/dispatched_lsns.rs @@ -0,0 +1,42 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The dispatcher's hold on the outcome floor for every accepted request that +//! carries a WAL LSN. +//! +//! A request holds its window from the moment the dispatcher accepts it until +//! the core's final response arrives. A core that died, or a drain that gave +//! up on a core, also settles the window: that core never publishes a +//! watermark again. + +use std::collections::HashMap; +use std::sync::Arc; + +use crate::types::Lsn; + +use super::outcome_floor::{OutcomeFloor, WriteWindow}; + +/// Open windows of dispatched requests, by request id. +#[derive(Debug, Default)] +pub(super) struct DispatchedLsns { + windows: HashMap, +} + +impl DispatchedLsns { + /// Hold the floor below `lsn` until request `request_id` is answered. + pub(super) fn track(&mut self, floor: &Arc, request_id: u64, lsn: Lsn) { + self.windows.insert(request_id, floor.open_dispatched(lsn)); + } + + /// Release the hold of request `request_id`, if it has one. + pub(super) fn settle(&mut self, request_id: u64) { + if let Some(window) = self.windows.remove(&request_id) { + window.settle(); + } + } + + /// Number of requests holding the floor. + #[cfg(test)] + pub(super) fn len(&self) -> usize { + self.windows.len() + } +} diff --git a/nodedb/src/bridge/dispatch/dispatcher.rs b/nodedb/src/bridge/dispatch/dispatcher.rs index 7e1783f3e..35981fd38 100644 --- a/nodedb/src/bridge/dispatch/dispatcher.rs +++ b/nodedb/src/bridge/dispatch/dispatcher.rs @@ -1,23 +1,34 @@ // SPDX-License-Identifier: BUSL-1.1 -use std::collections::{HashMap, HashSet}; +//! The bridge [`Dispatcher`]: its per-core channels, construction, and +//! read-only accessors. +//! +//! Admission and enqueue live in `enqueue`. Response polling and dead-core +//! synthesis live in `response_poll`. The shutdown drain lives in `drain`. -use tracing::warn; +use std::collections::{HashMap, HashSet}; +use std::sync::Arc; use nodedb_bridge::backpressure::{BackpressureConfig, BackpressureController, PressureState}; use nodedb_bridge::buffer::RingBuffer; use nodedb_bridge::wfq::WeightedFairQueue; use nodedb_types::PriorityClass; +use tokio::sync::Notify; use crate::bridge::envelope; -use crate::bridge::envelope::{ErrorCode, Payload, Status}; use crate::control::router::vshard::VShardRouter; use crate::data::eventfd::EventFdNotifier; -use crate::types::{Lsn, RequestId}; - -use crate::bridge::admission_chokepoint::{assert_write_admitted, reject_uninjected_write}; +use crate::types::Lsn; use super::core_channel::{CoreChannel, CoreChannelDataSide}; +use super::dispatched_lsns::DispatchedLsns; +use super::outcome_floor::OutcomeFloor; + +/// Per-core request queue capacity of the server's bridge dispatcher. +/// +/// Each core's weighted-fair queue and SPSC rings hold at most this many +/// requests. Every request for one vShard routes to the same core. +pub const DATA_PLANE_QUEUE_CAPACITY: usize = 1024; /// Serialized form of a request that goes through the SPSC ring buffer. /// @@ -28,6 +39,20 @@ use super::core_channel::{CoreChannel, CoreChannelDataSide}; pub struct BridgeRequest { /// The full typed request envelope. pub inner: envelope::Request, + /// The outcome floor when this request entered the ring: every record at + /// or below it that any core receives has a final outcome. + pub outcome_floor: Lsn, +} + +impl BridgeRequest { + /// A request that carries no outcome floor. A core that reads it learns + /// nothing about the floor. + pub fn unfloored(inner: envelope::Request) -> Self { + Self { + inner, + outcome_floor: Lsn::ZERO, + } + } } /// Serialized form of a response coming back from the Data Plane. @@ -73,7 +98,7 @@ pub struct Dispatcher { pub(super) cores: Vec, /// Routes vShards to core IDs. - router: VShardRouter, + pub(super) router: VShardRouter, /// Per-tenant in-flight request count across all cores. pub(super) tenant_inflight: HashMap, @@ -82,13 +107,13 @@ pub struct Dispatcher { pub(super) request_tenant: HashMap, /// Maximum in-flight requests per tenant (0 = unlimited). - max_per_tenant_inflight: u32, + pub(super) max_per_tenant_inflight: u32, /// Per-core queue capacity (used in tenant fairness recalculation). - per_core_capacity: u32, + pub(super) per_core_capacity: u32, /// Resolves priority class for a database_id (consulted on enqueue). - priority_resolver: Box, + pub(super) priority_resolver: Box, /// True once the shutdown bus has entered `DrainingDataPlane`. /// @@ -96,6 +121,16 @@ pub struct Dispatcher { /// enqueue takes, so a dispatch that observed `false` has already pushed by /// the time the drain starts. Nothing reaches a core after that. pub(super) data_plane_draining: bool, + + /// Capacity freed on the bridge dispatcher. Every path that releases an + /// in-flight slot wakes all waiters once per call. + pub(super) capacity_freed: Arc, + + /// The node's outcome floor. Every ring push carries its current value. + pub(super) outcome_floor: Arc, + + /// The floor windows of accepted requests that carry a WAL LSN. + pub(super) dispatched_lsns: DispatchedLsns, } impl Dispatcher { @@ -149,171 +184,27 @@ impl Dispatcher { per_core_capacity: queue_capacity as u32, priority_resolver, data_plane_draining: false, + capacity_freed: Arc::new(Notify::new()), + outcome_floor: OutcomeFloor::new(), + dispatched_lsns: DispatchedLsns::default(), }, data_sides, ) } - /// Dispatch a request to the correct Data Plane core. + /// The signal fired whenever the dispatcher frees at least one in-flight + /// slot. /// - /// Enqueues into the per-core weighted-fair queue keyed by `DatabaseId`, - /// then flushes WFQ → physical ring. Returns `Err` when the WFQ itself is - /// full (total capacity reached across all active databases on that core). - pub fn dispatch(&mut self, request: envelope::Request) -> crate::Result<()> { - reject_uninjected_write(&request)?; - assert_write_admitted(&request); - self.reject_if_draining()?; - let tenant_id = request.tenant_id.as_u64(); - let req_id = request.request_id.as_u64(); - let database_id = request.database_id.as_u64(); - - // Per-tenant fairness: reject if this tenant has too many in-flight requests. - if self.max_per_tenant_inflight > 0 { - let inflight = self.tenant_inflight.get(&tenant_id).copied().unwrap_or(0); - if inflight >= self.max_per_tenant_inflight { - return Err(crate::Error::Dispatch { - detail: format!( - "tenant {tenant_id}: queue full ({inflight}/{} in-flight)", - self.max_per_tenant_inflight - ), - }); - } - } - - let core_id = - self.router - .resolve(request.vshard_id) - .ok_or_else(|| crate::Error::Dispatch { - detail: format!("no core for vshard {}", request.vshard_id), - })?; - - let channel = &mut self.cores[core_id]; - - // Refresh priority for this DB in the WFQ. - let cls = self.priority_resolver.priority_for(database_id); - channel.wfq.set_priority(database_id, cls); - - // Check per-DB suspended state (≥95% of fair share). - if channel.wfq.is_suspended_for(database_id) { - return Err(crate::Error::Dispatch { - detail: format!( - "database {database_id}: virtual queue suspended (≥95% of fair share on core {core_id})" - ), - }); - } - - // Enqueue into the WFQ — returns Err if total capacity is full. - channel - .wfq - .try_enqueue(database_id, request) - .map_err(|_| crate::Error::Dispatch { - detail: format!("core {core_id}: total WFQ capacity exhausted"), - })?; - - // Update per-DB pressure. - channel.update_db_pressure(database_id); - - // Flush WFQ → physical ring. - channel.flush_wfq(); - - // Update global backpressure based on ring utilization. - let util = channel.request_tx.utilization(); - if let Some(new_state) = channel.backpressure.update(util) { - warn!( - core_id, - utilization = util, - state = ?new_state, - "backpressure transition" - ); - } - - // Track the request as outstanding on this core, so a later core death - // can fail it instead of stranding the caller's waiter. - channel.outstanding.insert(req_id); - - // Track per-tenant in-flight + request→tenant mapping for response routing. - *self.tenant_inflight.entry(tenant_id).or_insert(0) += 1; - self.request_tenant.insert(req_id, tenant_id); - - // Wake the Data Plane core via eventfd. - if let Some(ref notifier) = channel.wake_notifier { - notifier.notify(); - } - - Ok(()) - } - - /// Record a response received for a tenant (decrements in-flight count). - pub fn tenant_response_received(&mut self, tenant_id: u64) { - if let Some(count) = self.tenant_inflight.get_mut(&tenant_id) { - *count = count.saturating_sub(1); - } + /// A caller refused with [`crate::Error::DispatchCapacity`] waits on it + /// before it retries. It wakes only registered waiters, so a caller + /// enables its `notified()` future before it checks for refused work. + pub fn capacity_freed(&self) -> Arc { + Arc::clone(&self.capacity_freed) } - /// Recalculate the per-tenant in-flight limit based on active tenants. - pub fn recalculate_tenant_limits(&mut self) { - let active = self.tenant_inflight.len().max(1) as u32; - let total_capacity: u32 = self.cores.len() as u32 * self.per_core_capacity; - self.max_per_tenant_inflight = (total_capacity / active).max(2); - self.tenant_inflight.retain(|_, count| *count > 0); - } - - /// Dispatch a request directly to a specific core by index. - /// - /// Bypasses vShard routing. Used by the checkpoint manager to send - /// checkpoint requests to every core regardless of vShard assignment. - pub fn dispatch_to_core( - &mut self, - core_id: usize, - request: envelope::Request, - ) -> crate::Result<()> { - reject_uninjected_write(&request)?; - assert_write_admitted(&request); - self.reject_if_draining()?; - if core_id >= self.cores.len() { - return Err(crate::Error::Dispatch { - detail: format!("core {core_id} out of range (have {})", self.cores.len()), - }); - } - - let tenant_id = request.tenant_id.as_u64(); - let req_id = request.request_id.as_u64(); - let database_id = request.database_id.as_u64(); - let channel = &mut self.cores[core_id]; - - let cls = self.priority_resolver.priority_for(database_id); - channel.wfq.set_priority(database_id, cls); - - channel - .wfq - .try_enqueue(database_id, request) - .map_err(|_| crate::Error::Dispatch { - detail: format!("core {core_id}: total WFQ capacity exhausted"), - })?; - - channel.update_db_pressure(database_id); - channel.flush_wfq(); - - let util = channel.request_tx.utilization(); - if let Some(new_state) = channel.backpressure.update(util) { - warn!( - core_id, - utilization = util, - state = ?new_state, - "backpressure transition" - ); - } - - channel.outstanding.insert(req_id); - - *self.tenant_inflight.entry(tenant_id).or_insert(0) += 1; - self.request_tenant.insert(req_id, tenant_id); - - if let Some(ref notifier) = channel.wake_notifier { - notifier.notify(); - } - - Ok(()) + /// The node's outcome floor, shared with every write that opens a window. + pub fn outcome_floor(&self) -> Arc { + Arc::clone(&self.outcome_floor) } /// Maximum SPSC request queue utilization across all cores (0-100). @@ -336,97 +227,6 @@ impl Dispatcher { .unwrap_or(PressureState::Normal) } - /// Poll responses from all Data Plane cores. - /// - /// A core whose channel has been observed dead contributes a synthesized - /// error `Response` for every request still outstanding on it: the one a - /// failed `try_push` consumed, everything still staged in its WFQ, and - /// everything dispatched earlier that it never answered. Those travel back - /// with the real responses so the single completion loop in the caller - /// finishes each waiter, and the loop below releases each request's - /// `tenant_inflight` slot exactly as a real response would — without which - /// one dead core ratchets the tenant's in-flight count until the tenant is - /// rejected on healthy cores too. - pub fn poll_responses(&mut self) -> Vec { - let mut responses = Vec::new(); - for (core_id, channel) in self.cores.iter_mut().enumerate() { - let mut batch = Vec::new(); - let (_drained, producer_gone) = channel.response_rx.drain_into(&mut batch, 64); - for br in batch { - let rid = br.inner.request_id.as_u64(); - // A streaming scan answers with many partials before its final - // response. The request is still executing on the core until - // that final one arrives, so releasing it here would let the - // shutdown drain call a live scan finished and would drop the - // tenant's in-flight slot mid-stream. - if !br.inner.partial { - channel.outstanding.remove(&rid); - if let Some(tid) = self.request_tenant.remove(&rid) - && let Some(count) = self.tenant_inflight.get_mut(&tid) - { - *count = count.saturating_sub(1); - } - } - responses.push(br.inner); - } - - if !(producer_gone || channel.request_tx.is_disconnected()) { - // Opportunistically flush WFQ after draining responses to fill headroom. - channel.flush_wfq(); - continue; - } - - // The core is gone. Collect every request it can no longer answer: - // items still staged in the WFQ first (dispatch order), then the - // rest of the outstanding set. A staged item is also in - // `outstanding`, so `seen` keeps each id to a single response. - let mut seen = HashSet::new(); - let mut lost = Vec::new(); - for staged in channel.wfq.drain() { - let rid = staged.request_id.as_u64(); - if seen.insert(rid) { - lost.push(rid); - } - } - for rid in channel.outstanding.drain() { - if seen.insert(rid) { - lost.push(rid); - } - } - - // Idempotence: both sources are emptied here — `wfq.drain` leaves - // the staging queue empty and `outstanding.drain` clears the set — - // and `flush_wfq` refuses to stage anything new onto a - // disconnected producer. A later poll therefore finds both empty - // and emits nothing, so a permanently dead core costs one pass - // over two empty containers rather than a repeating failure storm. - for rid in lost { - if let Some(tid) = self.request_tenant.remove(&rid) - && let Some(count) = self.tenant_inflight.get_mut(&tid) - { - *count = count.saturating_sub(1); - } - responses.push(envelope::Response { - request_id: RequestId::new(rid), - status: Status::Error, - attempt: 1, - partial: false, - payload: Payload::empty(), - watermark_lsn: Lsn::ZERO, - error_code: Some(Box::new(ErrorCode::Internal { - detail: format!( - "core-{core_id} is gone; the request can never be executed" - ), - })), - read_set_valid: None, - read_version_lsn: Lsn::ZERO, - write_set: Vec::new(), - }); - } - } - responses - } - /// Number of Data Plane cores. pub fn num_cores(&self) -> usize { self.cores.len() @@ -444,395 +244,3 @@ impl Dispatcher { &self.router } } - -#[cfg(test)] -mod tests { - use super::*; - use crate::bridge::envelope::*; - use crate::types::*; - use nodedb_physical::physical_plan::DocumentOp; - use std::time::{Duration, Instant}; - - fn make_request(vshard: u32) -> envelope::Request { - envelope::Request { - request_id: RequestId::new(1), - tenant_id: TenantId::new(1), - database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::new(vshard), - plan: PhysicalPlan::Document(DocumentOp::PointGet { - collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "users"), - document_id: "u1".into(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - rls_filters: Vec::new(), - system_time: nodedb_types::SystemTimeScope::Current, - valid_at_ms: None, - }), - deadline: Instant::now() + Duration::from_secs(5), - priority: Priority::Normal, - trace_id: TraceId::ZERO, - consistency: ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission: Admission::Exempt(ExemptReason::Read), - } - } - - fn make_request_for_db(vshard: u32, db: u64, req_id: u64) -> envelope::Request { - envelope::Request { - request_id: RequestId::new(req_id), - tenant_id: TenantId::new(1), - database_id: DatabaseId::new(db), - vshard_id: VShardId::new(vshard), - plan: PhysicalPlan::Document(DocumentOp::PointGet { - collection: nodedb_types::QualifiedCollection::new(DatabaseId::new(db), "c"), - document_id: "d".into(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - rls_filters: Vec::new(), - system_time: nodedb_types::SystemTimeScope::Current, - valid_at_ms: None, - }), - deadline: Instant::now() + Duration::from_secs(5), - priority: Priority::Normal, - trace_id: TraceId::ZERO, - consistency: ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission: Admission::Exempt(ExemptReason::Read), - } - } - - #[test] - fn dispatch_routes_to_correct_core() { - let (mut dispatcher, data_sides) = Dispatcher::new(4, 64); - - dispatcher.dispatch(make_request(0)).unwrap(); - dispatcher.dispatch(make_request(1)).unwrap(); - dispatcher.dispatch(make_request(4)).unwrap(); // Wraps to core 0. - - assert_eq!(data_sides[0].request_rx.len(), 2); - assert_eq!(data_sides[1].request_rx.len(), 1); - assert_eq!(data_sides[2].request_rx.len(), 0); - } - - #[test] - fn response_roundtrip() { - let (mut dispatcher, mut data_sides) = Dispatcher::new(2, 64); - - dispatcher.dispatch(make_request(0)).unwrap(); - - let _req = data_sides[0].request_rx.try_pop().unwrap(); - data_sides[0] - .response_tx - .try_push(BridgeResponse { - inner: envelope::Response { - request_id: RequestId::new(1), - status: Status::Ok, - attempt: 1, - partial: false, - payload: Payload::from_vec(b"result".to_vec()), - watermark_lsn: Lsn::new(42), - error_code: None, - read_set_valid: None, - read_version_lsn: crate::types::Lsn::ZERO, - write_set: Vec::new(), - }, - }) - .unwrap(); - - let responses = dispatcher.poll_responses(); - assert_eq!(responses.len(), 1); - assert_eq!(responses[0].status, Status::Ok); - assert_eq!(&*responses[0].payload, b"result"); - } - - #[test] - fn full_queue_returns_error() { - // With WFQ capacity == ring capacity, filling WFQ should eventually - // cause total-capacity exhaustion. - let (mut dispatcher, _data_sides) = Dispatcher::new(1, 4); - - for i in 0..4u64 { - dispatcher - .dispatch(make_request_for_db(0, i + 1, i + 1)) - .unwrap(); - } - - // Next dispatch should fail — WFQ total capacity exhausted. - let result = dispatcher.dispatch(make_request_for_db(0, 99, 99)); - assert!(result.is_err()); - } - - #[test] - fn dispatch_to_core_tracks_request_lifecycle() { - let (mut dispatcher, mut data_sides) = Dispatcher::new(2, 64); - let request = make_request(0); - let tenant_id = request.tenant_id.as_u64(); - let request_id = request.request_id.as_u64(); - - dispatcher.dispatch_to_core(1, request).unwrap(); - - assert_eq!(dispatcher.tenant_inflight.get(&tenant_id), Some(&1)); - assert_eq!(dispatcher.request_tenant.get(&request_id), Some(&tenant_id)); - assert_eq!(data_sides[1].request_rx.len(), 1); - - let _req = data_sides[1].request_rx.try_pop().unwrap(); - data_sides[1] - .response_tx - .try_push(BridgeResponse { - inner: envelope::Response { - request_id: RequestId::new(request_id), - status: Status::Ok, - attempt: 1, - partial: false, - payload: Payload::empty(), - watermark_lsn: Lsn::ZERO, - error_code: None, - read_set_valid: None, - read_version_lsn: crate::types::Lsn::ZERO, - write_set: Vec::new(), - }, - }) - .unwrap(); - - let responses = dispatcher.poll_responses(); - assert_eq!(responses.len(), 1); - assert_eq!(dispatcher.tenant_inflight.get(&tenant_id), Some(&0)); - assert!(!dispatcher.request_tenant.contains_key(&request_id)); - } - - #[test] - fn per_db_pressure_reported() { - let (mut dispatcher, _) = Dispatcher::new(1, 8); - // Fill fair share for DB 1 using 4 of 8 slots. - // With one DB initially, fair share = 8. With two DBs = 4 each. - // First enqueue DB1 + DB2, so fair_share = 4. - for i in 0..4u64 { - dispatcher - .dispatch(make_request_for_db(0, 1, i + 10)) - .unwrap(); - } - for i in 0..4u64 { - dispatcher - .dispatch(make_request_for_db(0, 2, i + 20)) - .unwrap(); - } - // After filling DB1's fair share, it should be suspended on core 0. - // (exact state depends on WFQ flush draining items to ring first) - // The test confirms per-DB pressure is being tracked without panic. - let _ = dispatcher.db_pressure_on_core(0, 1); - let _ = dispatcher.db_pressure_on_core(0, 2); - } - - // --- Dead-core request loss (GitHub #265) --- - // - // When a Data Plane core's consumer/producer is dropped (the core thread - // died), `Dispatcher` must synthesize an error `Response` for every - // request it knows is outstanding on that core, rather than dropping the - // request silently and leaking the caller's waiter + `tenant_inflight` - // slot forever. Dropping one element of the `data_sides` vector handed - // back by `Dispatcher::new`/`with_resolver` simulates that core thread - // dying, matching how `dispatch_routes_to_correct_core` and - // `response_roundtrip` above obtain the data-plane side of the channel. - - #[test] - fn dead_core_synthesizes_error_response_for_lost_request() { - let (mut dispatcher, mut data_sides) = Dispatcher::new(3, 64); - - // Core 2's thread has died: both halves of its data-plane side are gone. - let dead_core = 2; - drop(data_sides.remove(dead_core)); - - let request = make_request_for_db(0, 1, 7); - let request_id = request.request_id.as_u64(); - - // `dispatch_to_core` still reports success: the request was already - // moved into the doomed `try_push` inside `flush_wfq` before the - // failure is observed, which is exactly the defect being covered. - dispatcher.dispatch_to_core(dead_core, request).unwrap(); - - let responses = dispatcher.poll_responses(); - assert_eq!(responses.len(), 1, "expected one synthesized response"); - let resp = &responses[0]; - assert_eq!(resp.request_id.as_u64(), request_id); - assert_eq!(resp.status, Status::Error); - match resp.error_code.as_deref() { - Some(ErrorCode::Internal { detail }) => { - assert!( - detail.contains(&dead_core.to_string()), - "error detail should name the dead core, got: {detail}" - ); - } - other => panic!("expected ErrorCode::Internal naming the core, got: {other:?}"), - } - } - - #[test] - fn dead_core_synthesized_response_resets_tenant_inflight() { - // The ratchet: `tenant_inflight` is incremented on dispatch and must - // return to its pre-dispatch value once the synthesized response for - // the lost request is drained through `poll_responses` — otherwise it - // climbs forever and eventually starves the tenant on healthy cores. - let (mut dispatcher, mut data_sides) = Dispatcher::new(2, 64); - let dead_core = 0; - drop(data_sides.remove(dead_core)); - - let request = make_request_for_db(0, 1, 1); - let tenant_id = request.tenant_id.as_u64(); - - let before = dispatcher - .tenant_inflight - .get(&tenant_id) - .copied() - .unwrap_or(0); - - dispatcher.dispatch_to_core(dead_core, request).unwrap(); - assert_eq!( - dispatcher.tenant_inflight.get(&tenant_id).copied(), - Some(before + 1), - "dispatch must still increment tenant_inflight even though the core is dead" - ); - - let responses = dispatcher.poll_responses(); - assert_eq!(responses.len(), 1); - assert_eq!( - dispatcher - .tenant_inflight - .get(&tenant_id) - .copied() - .unwrap_or(0), - before, - "tenant_inflight must return to its pre-dispatch value, not ratchet upward" - ); - assert!(!dispatcher.request_tenant.contains_key(&1)); - } - - #[test] - fn dead_core_does_not_affect_live_core() { - let (mut dispatcher, mut data_sides) = Dispatcher::new(2, 64); - let dead_core = 0; - let live_core = 1; - drop(data_sides.remove(dead_core)); - // Removing index 0 shifted core 1's data side down to index 0. - let live_data_side = &mut data_sides[0]; - - let dead_request = make_request_for_db(0, 1, 1); - let live_request = make_request_for_db(0, 2, 2); - let live_request_id = live_request.request_id.as_u64(); - - dispatcher - .dispatch_to_core(dead_core, dead_request) - .unwrap(); - dispatcher - .dispatch_to_core(live_core, live_request) - .unwrap(); - - // The live core answers normally, through the real ring buffer. - let _req = live_data_side.request_rx.try_pop().unwrap(); - live_data_side - .response_tx - .try_push(BridgeResponse { - inner: envelope::Response { - request_id: RequestId::new(live_request_id), - status: Status::Ok, - attempt: 1, - partial: false, - payload: Payload::empty(), - watermark_lsn: Lsn::ZERO, - error_code: None, - read_set_valid: None, - read_version_lsn: crate::types::Lsn::ZERO, - write_set: Vec::new(), - }, - }) - .unwrap(); - - let responses = dispatcher.poll_responses(); - assert_eq!( - responses.len(), - 2, - "one synthesized error from the dead core, one real Ok from the live core" - ); - - let live_resp = responses - .iter() - .find(|r| r.request_id.as_u64() == live_request_id) - .expect("live core's real response must be present"); - assert_eq!(live_resp.status, Status::Ok); - assert!(live_resp.error_code.is_none()); - - let dead_resp = responses - .iter() - .find(|r| r.request_id.as_u64() != live_request_id) - .expect("dead core's synthesized response must be present"); - assert_eq!(dead_resp.status, Status::Error); - assert!(dead_resp.error_code.is_some()); - } - - #[test] - fn dead_core_fails_requests_still_queued_in_wfq() { - // Fill the physical ring to capacity while the core is alive, so a - // request dispatched afterward parks in the WFQ without ever - // attempting a push (flush_wfq's utilization check breaks before it - // reaches the doomed try_push). Then kill the core and confirm the - // WFQ-queued request is failed too, not left sitting in the queue - // forever. - let (mut dispatcher, mut data_sides) = Dispatcher::new(1, 4); - - for i in 0..4u64 { - dispatcher - .dispatch_to_core(0, make_request_for_db(0, i + 1, i + 1)) - .unwrap(); - } - assert_eq!(data_sides[0].request_rx.len(), 4); - - // Core 0's thread dies with 4 unanswered requests sitting in its ring. - drop(data_sides.remove(0)); - - // This request cannot reach the (full, dead) physical ring — it stays - // parked in the WFQ. - let parked_request_id = 99u64; - dispatcher - .dispatch_to_core(0, make_request_for_db(0, 99, parked_request_id)) - .unwrap(); - - let responses = dispatcher.poll_responses(); - let ids: std::collections::HashSet = - responses.iter().map(|r| r.request_id.as_u64()).collect(); - - // The 4 previously-dispatched-but-unanswered requests, plus the one - // still parked in the WFQ, must all be failed. - assert_eq!( - responses.len(), - 5, - "expected all 5 outstanding requests failed" - ); - for id in 1..=4u64 { - assert!( - ids.contains(&id), - "request {id} in the dead ring must be failed" - ); - } - assert!( - ids.contains(&parked_request_id), - "request parked in the WFQ must be failed, not left queued" - ); - for r in &responses { - assert_eq!(r.status, Status::Error); - assert!(r.error_code.is_some()); - } - } -} diff --git a/nodedb/src/bridge/dispatch/drain.rs b/nodedb/src/bridge/dispatch/drain.rs index 37a89d9c4..064ba8155 100644 --- a/nodedb/src/bridge/dispatch/drain.rs +++ b/nodedb/src/bridge/dispatch/drain.rs @@ -27,6 +27,7 @@ use crate::bridge::envelope::{ErrorCode, Payload, Status}; use crate::types::{Lsn, RequestId}; use super::dispatcher::Dispatcher; +use super::enqueue::release_inflight_slot; /// Work one core still owes the Control Plane. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -62,8 +63,9 @@ impl Dispatcher { /// Idempotent: a second call re-flushes and changes nothing else. pub fn begin_data_plane_drain(&mut self) { self.data_plane_draining = true; + let outcome_floor = self.outcome_floor.floor(); for channel in self.cores.iter_mut() { - channel.flush_wfq(); + channel.flush_wfq(outcome_floor); } } @@ -128,6 +130,7 @@ impl Dispatcher { /// core, and outstanding work reached one that did not answer in time. pub fn abandon_data_plane_work(&mut self) -> Vec { let mut abandoned = Vec::new(); + let mut freed = false; for (core_id, channel) in self.cores.iter_mut().enumerate() { let mut ids: Vec = channel .wfq @@ -150,11 +153,10 @@ impl Dispatcher { "data plane drain deadline expired — failing the requests this core still holds" ); for rid in ids { - if let Some(tid) = self.request_tenant.remove(&rid) - && let Some(count) = self.tenant_inflight.get_mut(&tid) - { - *count = count.saturating_sub(1); - } + // The node is shutting down, and this core publishes nothing more. + self.dispatched_lsns.settle(rid); + freed |= + release_inflight_slot(&mut self.request_tenant, &mut self.tenant_inflight, rid); abandoned.push(envelope::Response { request_id: RequestId::new(rid), status: Status::Error, @@ -174,6 +176,9 @@ impl Dispatcher { }); } } + if freed { + self.capacity_freed.notify_waiters(); + } abandoned } } @@ -181,42 +186,7 @@ impl Dispatcher { #[cfg(test)] mod tests { use super::*; - use crate::bridge::envelope::{Admission, ExemptReason, PhysicalPlan, Priority, Request}; - use crate::types::{DatabaseId, ReadConsistency, TenantId, TraceId, VShardId}; - use nodedb_physical::physical_plan::DocumentOp; - use nodedb_types::QualifiedCollection; - use std::time::{Duration, Instant}; - - fn make_request(id: u64, vshard: u32) -> Request { - Request { - request_id: RequestId::new(id), - tenant_id: TenantId::new(1), - database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::new(vshard), - plan: PhysicalPlan::Document(DocumentOp::PointGet { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), - document_id: "d".into(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - rls_filters: Vec::new(), - system_time: nodedb_types::SystemTimeScope::Current, - valid_at_ms: None, - }), - deadline: Instant::now() + Duration::from_secs(5), - priority: Priority::Normal, - trace_id: TraceId::ZERO, - consistency: ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission: Admission::Exempt(ExemptReason::Read), - } - } + use crate::bridge::dispatch::test_requests::make_request_for_db; #[test] fn a_fresh_dispatcher_accepts_work_and_reports_it_pending() { @@ -225,7 +195,7 @@ mod tests { assert!(dispatcher.data_plane_pending().is_empty()); dispatcher - .dispatch(make_request(1, 0)) + .dispatch(make_request_for_db(0, 0, 1)) .expect("a running dispatcher accepts work"); let pending = dispatcher.data_plane_pending(); assert_eq!(pending.len(), 1, "one core holds the request"); @@ -238,7 +208,7 @@ mod tests { dispatcher.begin_data_plane_drain(); let err = dispatcher - .dispatch(make_request(1, 0)) + .dispatch(make_request_for_db(0, 0, 1)) .expect_err("a draining dispatcher must refuse new work"); assert!( matches!(err, crate::Error::Dispatch { .. }), @@ -256,7 +226,7 @@ mod tests { dispatcher.begin_data_plane_drain(); let err = dispatcher - .dispatch_to_core(0, make_request(1, 0)) + .dispatch_to_core(0, make_request_for_db(0, 0, 1)) .expect_err("the direct-to-core path uses the same gate"); assert!(matches!(err, crate::Error::Dispatch { .. })); } @@ -268,7 +238,7 @@ mod tests { dispatcher.begin_data_plane_drain(); assert!(dispatcher.is_data_plane_draining()); assert!( - dispatcher.dispatch(make_request(1, 0)).is_err(), + dispatcher.dispatch(make_request_for_db(0, 0, 1)).is_err(), "a second drain start changes nothing" ); } @@ -277,7 +247,7 @@ mod tests { fn abandoned_work_is_answered_and_cleared() { let (mut dispatcher, _data_sides) = Dispatcher::new(1, 64); dispatcher - .dispatch(make_request(7, 0)) + .dispatch(make_request_for_db(0, 0, 7)) .expect("accept before the drain"); dispatcher.begin_data_plane_drain(); diff --git a/nodedb/src/bridge/dispatch/enqueue.rs b/nodedb/src/bridge/dispatch/enqueue.rs new file mode 100644 index 000000000..ada1ef612 --- /dev/null +++ b/nodedb/src/bridge/dispatch/enqueue.rs @@ -0,0 +1,368 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Admission, weighted-fair enqueue, and per-tenant in-flight accounting for +//! the bridge [`Dispatcher`]. +//! +//! A capacity limit refuses with [`crate::Error::DispatchCapacity`] and hands +//! the request back. Every other refusal is terminal. + +use std::collections::HashMap; + +use tracing::warn; + +use crate::DispatchCapacityScope; +use crate::bridge::admission_chokepoint::{assert_write_admitted, reject_uninjected_write}; +use crate::bridge::envelope; +use crate::types::Lsn; + +use super::dispatcher::Dispatcher; +use super::refusal::DispatchRefusal; + +impl Dispatcher { + /// Dispatch a request to the correct Data Plane core. + /// + /// Enqueues into the per-core weighted-fair queue keyed by `DatabaseId`, + /// then flushes WFQ → physical ring. A capacity limit refuses with + /// [`crate::Error::DispatchCapacity`]. Every other refusal is terminal. + pub fn dispatch(&mut self, request: envelope::Request) -> crate::Result<()> { + self.try_dispatch(request).map_err(|refusal| refusal.error) + } + + /// Dispatch like [`Self::dispatch`], handing the request back on refusal. + /// + /// A caller that must retry a capacity refusal re-sends the returned + /// request. The dispatcher tracks nothing for a refused request. + pub fn try_dispatch(&mut self, request: envelope::Request) -> Result<(), Box> { + if let Err(error) = reject_uninjected_write(&request) { + return Err(DispatchRefusal::boxed(error, request)); + } + assert_write_admitted(&request); + if let Err(error) = self.reject_if_draining() { + return Err(DispatchRefusal::boxed(error, request)); + } + let tenant_id = request.tenant_id.as_u64(); + let req_id = request.request_id.as_u64(); + let database_id = request.database_id.as_u64(); + let wal_lsn = request.wal_lsn; + + // Per-tenant fairness: refuse while the tenant holds its in-flight cap. + if self.max_per_tenant_inflight > 0 { + let inflight = self.tenant_inflight.get(&tenant_id).copied().unwrap_or(0); + if inflight >= self.max_per_tenant_inflight { + let scope = DispatchCapacityScope::TenantInflight { + tenant_id: request.tenant_id, + inflight, + cap: self.max_per_tenant_inflight, + }; + return Err(DispatchRefusal::boxed( + crate::Error::DispatchCapacity { scope }, + request, + )); + } + } + + let Some(core_id) = self.router.resolve(request.vshard_id) else { + let error = crate::Error::Dispatch { + detail: format!("no core for vshard {}", request.vshard_id), + }; + return Err(DispatchRefusal::boxed(error, request)); + }; + + let channel = &mut self.cores[core_id]; + + // Refresh priority for this DB in the WFQ. + let cls = self.priority_resolver.priority_for(database_id); + channel.wfq.set_priority(database_id, cls); + + // Check per-DB suspended state (≥95% of fair share). + if channel.wfq.is_suspended_for(database_id) { + let scope = DispatchCapacityScope::DatabaseSuspended { + database_id: request.database_id, + core_id, + }; + return Err(DispatchRefusal::boxed( + crate::Error::DispatchCapacity { scope }, + request, + )); + } + + // Enqueue into the WFQ. A full queue hands the request back. + if let Err(request) = channel.wfq.try_enqueue(database_id, request) { + let scope = DispatchCapacityScope::QueueFull { + core_id, + capacity: self.per_core_capacity, + }; + return Err(DispatchRefusal::boxed( + crate::Error::DispatchCapacity { scope }, + request, + )); + } + + self.commit_enqueued(core_id, database_id, tenant_id, req_id, wal_lsn); + Ok(()) + } + + /// Dispatch a request directly to a specific core by index. + /// + /// Bypasses vShard routing. Used by the checkpoint manager to send + /// checkpoint requests to every core regardless of vShard assignment. + pub fn dispatch_to_core( + &mut self, + core_id: usize, + request: envelope::Request, + ) -> crate::Result<()> { + reject_uninjected_write(&request)?; + assert_write_admitted(&request); + self.reject_if_draining()?; + if core_id >= self.cores.len() { + return Err(crate::Error::Dispatch { + detail: format!("core {core_id} out of range (have {})", self.cores.len()), + }); + } + + let tenant_id = request.tenant_id.as_u64(); + let req_id = request.request_id.as_u64(); + let database_id = request.database_id.as_u64(); + let wal_lsn = request.wal_lsn; + let channel = &mut self.cores[core_id]; + + let cls = self.priority_resolver.priority_for(database_id); + channel.wfq.set_priority(database_id, cls); + + channel.wfq.try_enqueue(database_id, request).map_err(|_| { + crate::Error::DispatchCapacity { + scope: DispatchCapacityScope::QueueFull { + core_id, + capacity: self.per_core_capacity, + }, + } + })?; + + self.commit_enqueued(core_id, database_id, tenant_id, req_id, wal_lsn); + Ok(()) + } + + /// Recalculate the per-tenant in-flight limit based on active tenants. + pub fn recalculate_tenant_limits(&mut self) { + let active = self.tenant_inflight.len().max(1) as u32; + let total_capacity: u32 = self.cores.len() as u32 * self.per_core_capacity; + self.max_per_tenant_inflight = (total_capacity / active).max(2); + self.tenant_inflight.retain(|_, count| *count > 0); + } + + /// Bookkeeping once a request sits in `core_id`'s weighted-fair queue: + /// hold the outcome floor below its WAL LSN, flush it toward the ring, + /// record pressure, track it as outstanding and in flight for its tenant, + /// and wake the core. + fn commit_enqueued( + &mut self, + core_id: usize, + database_id: u64, + tenant_id: u64, + req_id: u64, + wal_lsn: Option, + ) { + // The hold starts before the flush below, so no push carries a floor + // at or above this request's LSN while it is unanswered. + if let Some(lsn) = wal_lsn { + self.dispatched_lsns.track(&self.outcome_floor, req_id, lsn); + } + let outcome_floor = self.outcome_floor.floor(); + let channel = &mut self.cores[core_id]; + + // Update per-DB pressure. + channel.update_db_pressure(database_id); + + // Flush WFQ → physical ring. + channel.flush_wfq(outcome_floor); + + // Update global backpressure based on ring utilization. + let util = channel.request_tx.utilization(); + if let Some(new_state) = channel.backpressure.update(util) { + warn!( + core_id, + utilization = util, + state = ?new_state, + "backpressure transition" + ); + } + + // Track the request as outstanding on this core, so a later core death + // can fail it instead of stranding the caller's waiter. + channel.outstanding.insert(req_id); + + // Track per-tenant in-flight + request→tenant mapping for response routing. + *self.tenant_inflight.entry(tenant_id).or_insert(0) += 1; + self.request_tenant.insert(req_id, tenant_id); + + // Wake the Data Plane core via eventfd. + if let Some(ref notifier) = self.cores[core_id].wake_notifier { + notifier.notify(); + } + } +} + +/// Release the in-flight slot `request_id` holds for its tenant. +/// +/// Returns `true` when a slot was freed. A free function over the two maps, +/// so a caller can release while it holds a borrow of one core's channel. +pub(super) fn release_inflight_slot( + request_tenant: &mut HashMap, + tenant_inflight: &mut HashMap, + request_id: u64, +) -> bool { + let Some(tenant_id) = request_tenant.remove(&request_id) else { + return false; + }; + match tenant_inflight.get_mut(&tenant_id) { + Some(count) if *count > 0 => { + *count -= 1; + true + } + _ => false, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::dispatch::BridgeResponse; + use crate::bridge::dispatch::test_requests::{make_request, make_request_for_db}; + use crate::bridge::envelope::*; + use crate::types::*; + + #[test] + fn dispatch_routes_to_correct_core() { + let (mut dispatcher, data_sides) = Dispatcher::new(4, 64); + + dispatcher.dispatch(make_request(0)).unwrap(); + dispatcher.dispatch(make_request(1)).unwrap(); + dispatcher.dispatch(make_request(4)).unwrap(); // Wraps to core 0. + + assert_eq!(data_sides[0].request_rx.len(), 2); + assert_eq!(data_sides[1].request_rx.len(), 1); + assert_eq!(data_sides[2].request_rx.len(), 0); + } + + #[test] + fn tenant_at_inflight_cap_is_refused_with_tenant_scope() { + // One core with capacity 4 caps each tenant at 4 in-flight requests. + let (mut dispatcher, _data_sides) = Dispatcher::new(1, 4); + + for i in 0..4u64 { + dispatcher + .dispatch(make_request_for_db(0, i + 1, i + 1)) + .unwrap(); + } + + let refusal = dispatcher + .try_dispatch(make_request_for_db(0, 99, 99)) + .expect_err("the fifth request exceeds the tenant cap"); + let DispatchRefusal { error, request } = *refusal; + match error { + crate::Error::DispatchCapacity { + scope: + DispatchCapacityScope::TenantInflight { + tenant_id, + inflight, + cap, + }, + } => { + assert_eq!(tenant_id, TenantId::new(1)); + assert_eq!(inflight, 4); + assert_eq!(cap, 4); + } + other => panic!("expected a tenant-cap refusal, got: {other}"), + } + assert_eq!( + request.request_id, + RequestId::new(99), + "the refused request is handed back unsent" + ); + } + + #[test] + fn full_weighted_fair_queue_is_refused_with_queue_full_scope() { + // Distinct tenants and databases keep the tenant cap and the per-DB + // suspension out of play, so only the queue total can refuse. + let (mut dispatcher, _data_sides) = Dispatcher::new(1, 4); + + for i in 1..=64u64 { + let mut request = make_request_for_db(0, i, i); + request.tenant_id = TenantId::new(i); + match dispatcher.dispatch(request) { + Ok(()) => continue, + Err(crate::Error::DispatchCapacity { + scope: DispatchCapacityScope::QueueFull { core_id, capacity }, + }) => { + assert_eq!(core_id, 0); + assert_eq!(capacity, 4); + return; + } + Err(other) => panic!("expected a queue-full refusal, got: {other}"), + } + } + panic!("the weighted-fair queue never filled"); + } + + #[test] + fn dispatch_to_core_tracks_request_lifecycle() { + let (mut dispatcher, mut data_sides) = Dispatcher::new(2, 64); + let request = make_request(0); + let tenant_id = request.tenant_id.as_u64(); + let request_id = request.request_id.as_u64(); + + dispatcher.dispatch_to_core(1, request).unwrap(); + + assert_eq!(dispatcher.tenant_inflight.get(&tenant_id), Some(&1)); + assert_eq!(dispatcher.request_tenant.get(&request_id), Some(&tenant_id)); + assert_eq!(data_sides[1].request_rx.len(), 1); + + let _req = data_sides[1].request_rx.try_pop().unwrap(); + data_sides[1] + .response_tx + .try_push(BridgeResponse { + inner: envelope::Response { + request_id: RequestId::new(request_id), + status: Status::Ok, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: None, + read_set_valid: None, + read_version_lsn: crate::types::Lsn::ZERO, + write_set: Vec::new(), + }, + }) + .unwrap(); + + let responses = dispatcher.poll_responses(); + assert_eq!(responses.len(), 1); + assert_eq!(dispatcher.tenant_inflight.get(&tenant_id), Some(&0)); + assert!(!dispatcher.request_tenant.contains_key(&request_id)); + } + + #[test] + fn per_db_pressure_reported() { + let (mut dispatcher, _) = Dispatcher::new(1, 8); + // Fill fair share for DB 1 using 4 of 8 slots. + // With one DB initially, fair share = 8. With two DBs = 4 each. + // First enqueue DB1 + DB2, so fair_share = 4. + for i in 0..4u64 { + dispatcher + .dispatch(make_request_for_db(0, 1, i + 10)) + .unwrap(); + } + for i in 0..4u64 { + dispatcher + .dispatch(make_request_for_db(0, 2, i + 20)) + .unwrap(); + } + // After filling DB1's fair share, it should be suspended on core 0. + // (exact state depends on WFQ flush draining items to ring first) + // The test confirms per-DB pressure is being tracked without panic. + let _ = dispatcher.db_pressure_on_core(0, 1); + let _ = dispatcher.db_pressure_on_core(0, 2); + } +} diff --git a/nodedb/src/bridge/dispatch/mod.rs b/nodedb/src/bridge/dispatch/mod.rs index c261c0805..2bd5862d2 100644 --- a/nodedb/src/bridge/dispatch/mod.rs +++ b/nodedb/src/bridge/dispatch/mod.rs @@ -1,11 +1,22 @@ // SPDX-License-Identifier: BUSL-1.1 +mod closed_lsns; mod core_channel; +mod dispatched_lsns; mod dispatcher; mod drain; +mod enqueue; +mod outcome_floor; +mod refusal; +mod response_poll; +#[cfg(test)] +mod test_requests; pub use core_channel::{CoreChannel, CoreChannelDataSide}; pub use dispatcher::{ - BridgeRequest, BridgeResponse, DatabasePriorityResolver, DefaultPriorityResolver, Dispatcher, + BridgeRequest, BridgeResponse, DATA_PLANE_QUEUE_CAPACITY, DatabasePriorityResolver, + DefaultPriorityResolver, Dispatcher, }; pub use drain::CorePending; +pub use outcome_floor::{OutcomeFloor, ResendRefusal, StuckFloor, WriteWindow}; +pub use refusal::DispatchRefusal; diff --git a/nodedb/src/bridge/dispatch/outcome_floor.rs b/nodedb/src/bridge/dispatch/outcome_floor.rs new file mode 100644 index 000000000..b712bf673 --- /dev/null +++ b/nodedb/src/bridge/dispatch/outcome_floor.rs @@ -0,0 +1,702 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The outcome floor: the highest WAL LSN at or below which every record sent +//! to a Data Plane core has a final outcome. +//! +//! A record's outcome is final when the core applied it, or refused it and its +//! `WriteAborted` marker is durable. A watermark that says "restart may skip +//! every record at or below me" is sound only at or below this floor: a record +//! above it can still be on its way to a core, or still be applying. +//! +//! ## Windows +//! +//! A write opens a [`WriteWindow`] before it mints its LSN, notes each LSN it +//! mints, and settles the window once its outcome is final. The dispatcher +//! also opens a window for every accepted request that carries a WAL LSN, and +//! settles it when the core's final response arrives. A record minted outside +//! any window must never reach a core. +//! +//! Each open window has a horizon, a lower bound on every LSN it holds back: +//! +//! - A window opened before its mint takes `max_noted + 1`. WAL LSNs strictly +//! increase, so an LSN minted after the open exceeds every LSN noted before +//! it. +//! - A dispatcher window takes the request's LSN. +//! +//! ## The floor +//! +//! F is the smallest open horizon minus one, or the highest noted LSN when no +//! window is open. F never decreases: each computed value is raised to the +//! last published one. +//! +//! F never passes an open mint window: +//! +//! 1. Every computed value is at most `max_noted`, so every published F is at +//! most `max_noted`. +//! 2. A mint window opens with horizon `max_noted + 1`, above every F published +//! before it. +//! 3. While it stays open, every computed value is at most its horizon minus +//! one, so the published F stays below its horizon too. +//! +//! A dispatcher window whose LSN is at or below the published F cannot hold F +//! back. Its record was minted outside a window, and the floor passed it +//! before it reached the dispatcher. The open logs that record. +//! +//! ## Owned records +//! +//! A window owns every LSN it records with [`WriteWindow::own`]. A record +//! sent to a core again through [`OutcomeFloor::open_existing`] must have no +//! owner and no final outcome: an owner carries its record to an outcome, and +//! a closed owner already did. Both refuse the resend, as the floor does. +//! +//! ## Closing a window +//! +//! [`WriteWindow::settle`] states that the outcome is final. +//! [`WriteWindow::hold`] states that the record has no final outcome in this +//! process: restart replay must reach it, so the window stays open until the +//! process exits. A window dropped without either is a leak. It stays open +//! too, and the drop counts it, logs an error, and files a report. + +use std::collections::BTreeMap; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::{Arc, Mutex, MutexGuard}; +use std::time::{Duration, Instant}; + +use tracing::{error, warn}; + +use super::closed_lsns::ClosedLsns; +use crate::types::Lsn; + +/// The node's registry of open write windows. +#[derive(Debug, Default)] +pub struct OutcomeFloor { + windows: Mutex, + /// Windows dropped without a settle or a hold. + leaked: AtomicU64, + /// Woken each time a window closes, so a waiter re-reads the floor. + closed: tokio::sync::Notify, +} + +#[derive(Debug)] +struct OpenWindow { + horizon: u64, + opened_at: Instant, + /// Held until restart: the floor stays below it by design. + held: bool, + /// LSNs this window owns. + lsns: Vec, +} + +#[derive(Debug, Default)] +struct Windows { + next_ticket: u64, + /// Each open window, by ticket. Tickets increase, so the first entry is + /// the oldest open window. + open: BTreeMap, + /// Number of open windows at each horizon. + horizons: BTreeMap, + /// Highest LSN any window noted. + max_noted: u64, + /// Highest floor handed out. + published: u64, + /// Number of held windows. + held: usize, + /// Open windows owning each LSN. + owners: BTreeMap, + /// LSNs above the published floor whose last owner closed. Bounded by + /// the number of live owned LSNs, whatever the floor does. + closed: ClosedLsns, +} + +impl Windows { + fn open(&mut self, horizon: u64) -> u64 { + let ticket = self.next_ticket; + self.next_ticket += 1; + self.open.insert( + ticket, + OpenWindow { + horizon, + opened_at: Instant::now(), + held: false, + lsns: Vec::new(), + }, + ); + *self.horizons.entry(horizon).or_insert(0) += 1; + ticket + } + + fn close(&mut self, ticket: u64) { + let Some(window) = self.open.remove(&ticket) else { + return; + }; + if let Some(count) = self.horizons.get_mut(&window.horizon) { + *count -= 1; + if *count == 0 { + self.horizons.remove(&window.horizon); + } + } + let mut released = Vec::new(); + for lsn in window.lsns { + if let Some(count) = self.owners.get_mut(&lsn) { + *count -= 1; + if *count == 0 { + self.owners.remove(&lsn); + released.push(lsn); + } + } + } + for lsn in released { + self.closed.insert(lsn, &self.owners); + } + } + + /// Record that window `ticket` owns `lsn`. + fn own(&mut self, ticket: u64, lsn: u64) { + self.note(lsn); + let Some(window) = self.open.get_mut(&ticket) else { + return; + }; + if !window.lsns.contains(&lsn) { + window.lsns.push(lsn); + *self.owners.entry(lsn).or_insert(0) += 1; + } + } + + /// Why `lsn` is claimed: a live window owns it, or its last owner + /// closed. `None` when it is free. + fn claim(&self, lsn: u64) -> Option { + if self.owners.contains_key(&lsn) { + Some(ResendRefusal::Owned) + } else if self.closed.contains(lsn) { + Some(ResendRefusal::Closed) + } else { + None + } + } + + /// Mark a window held. Returns its horizon and age. + fn mark_held(&mut self, ticket: u64) -> Option<(u64, Duration)> { + let window = self.open.get_mut(&ticket)?; + if !window.held { + window.held = true; + self.held += 1; + } + Some((window.horizon, window.opened_at.elapsed())) + } + + /// The oldest window that is not held. + fn oldest_unheld(&self) -> Option<&OpenWindow> { + self.open.values().find(|window| !window.held) + } + + fn note(&mut self, lsn: u64) { + self.max_noted = self.max_noted.max(lsn); + } + + fn floor(&mut self) -> u64 { + let computed = match self.horizons.first_key_value() { + Some((&horizon, _)) => horizon.saturating_sub(1).min(self.max_noted), + None => self.max_noted, + }; + self.published = self.published.max(computed); + // A closed LSN at or below the floor is refused by the floor itself. + self.closed.prune_through(self.published); + self.published + } +} + +/// Why [`OutcomeFloor::open_existing`] refused to send a record again. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ResendRefusal { + /// The floor passed the record: its outcome is final. + BelowFloor, + /// A live or held window owns the record and carries it to its outcome. + Owned, + /// The record's last owner closed: its outcome is final. + Closed, +} + +/// The oldest window that holds the floor, and how long it has held it. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct StuckFloor { + /// The current floor. + pub floor: Lsn, + /// The oldest open window's horizon: the floor stays below it. + pub horizon: Lsn, + /// How long the oldest open window has been open. + pub open_for: Duration, + /// Number of open windows that are not held. + pub open_windows: usize, +} + +impl OutcomeFloor { + /// An empty registry. Its floor starts at zero. + pub fn new() -> Arc { + Arc::new(Self::default()) + } + + fn lock(&self) -> MutexGuard<'_, Windows> { + self.windows.lock().unwrap_or_else(|p| p.into_inner()) + } + + /// Open a window before the write mints its LSN. + pub fn open_write(self: &Arc) -> WriteWindow { + let ticket = { + let mut windows = self.lock(); + let horizon = windows.max_noted.saturating_add(1); + windows.open(horizon) + }; + WriteWindow::new(Arc::clone(self), ticket) + } + + /// Open a window for a request that carries `lsn` as it enters the + /// dispatcher. + pub fn open_dispatched(self: &Arc, lsn: Lsn) -> WriteWindow { + let ticket = { + let mut windows = self.lock(); + if lsn.as_u64() <= windows.published { + warn!( + lsn = lsn.as_u64(), + floor = windows.published, + "a record reached the dispatcher after the outcome floor passed it; \ + it was minted outside a write window" + ); + } + let ticket = windows.open(lsn.as_u64()); + windows.own(ticket, lsn.as_u64()); + ticket + }; + WriteWindow::new(Arc::clone(self), ticket) + } + + /// Open a window for an existing record at `lsn` that is sent to a core + /// again. Refused, with the reason, when the record must not be sent: + /// + /// - the floor passed `lsn`, so its outcome is final; + /// - a live or held window owns it, and carries it to its outcome; + /// - a window that owned it closed, so its outcome is final. + pub fn open_existing(self: &Arc, lsn: Lsn) -> Result { + let ticket = { + let mut windows = self.lock(); + if lsn.as_u64() <= windows.floor() { + return Err(ResendRefusal::BelowFloor); + } + if let Some(refusal) = windows.claim(lsn.as_u64()) { + return Err(refusal); + } + let ticket = windows.open(lsn.as_u64()); + windows.own(ticket, lsn.as_u64()); + ticket + }; + Ok(WriteWindow::new(Arc::clone(self), ticket)) + } + + /// The current floor. Never lower than a value returned before. + pub fn floor(&self) -> Lsn { + Lsn::new(self.lock().floor()) + } + + /// The highest LSN any window noted: every record a write window minted + /// so far is at or below it. + pub fn max_noted(&self) -> Lsn { + Lsn::new(self.lock().max_noted) + } + + /// Windows dropped without a settle or a hold since the process started. + pub fn leaked_windows(&self) -> u64 { + self.leaked.load(Ordering::Relaxed) + } + + /// Windows held until restart. The floor stays below each by design. + pub fn held_windows(&self) -> usize { + self.lock().held + } + + /// The oldest open window that is not held, when it has held the floor + /// for longer than `bound`. A held window is expected to stay open, so + /// it never makes the floor stuck. + pub fn stuck(&self, bound: Duration) -> Option { + let mut windows = self.lock(); + let floor = windows.floor(); + let open_windows = windows.open.len() - windows.held; + let oldest = windows.oldest_unheld()?; + let open_for = oldest.opened_at.elapsed(); + (open_for > bound).then_some(StuckFloor { + floor: Lsn::new(floor), + horizon: Lsn::new(oldest.horizon), + open_for, + open_windows, + }) + } + + /// How long the oldest open window that is not held has been open, or + /// zero when none is. + pub fn oldest_open_for(&self) -> Duration { + self.lock() + .oldest_unheld() + .map_or(Duration::ZERO, |oldest| oldest.opened_at.elapsed()) + } + + fn close(&self, ticket: u64) { + self.lock().close(ticket); + self.closed.notify_waiters(); + } + + /// Wait until the floor reaches `target`: every record minted at or below + /// it has a final outcome. Returns `false` when `deadline` passes first, + /// for example behind a window held until restart. + pub async fn await_floor(&self, target: Lsn, deadline: tokio::time::Instant) -> bool { + loop { + let notified = self.closed.notified(); + tokio::pin!(notified); + notified.as_mut().enable(); + if self.floor() >= target { + return true; + } + if tokio::time::timeout_at(deadline, notified).await.is_err() { + return self.floor() >= target; + } + } + } + + /// Count a leaked window and report it. The window stays open. + fn leak(&self, ticket: u64) { + self.leaked.fetch_add(1, Ordering::Relaxed); + let (horizon, open_for) = self + .lock() + .open + .get(&ticket) + .map_or((0, Duration::ZERO), |window| { + (window.horizon, window.opened_at.elapsed()) + }); + error!( + ticket, + horizon, + open_for_ms = u64::try_from(open_for.as_millis()).unwrap_or(u64::MAX), + "a write window was dropped before its outcome was final; the outcome \ + floor stays below it until restart" + ); + crate::diag::write_window_leaked(ticket, horizon, open_for); + } +} + +/// One write's hold on the outcome floor. Settle it once the write's outcome +/// is final, or hold it when the record has no final outcome in this process. +/// Dropped without either, it leaks: it holds the floor until the process +/// exits, and the drop reports it. +#[derive(Debug)] +pub struct WriteWindow { + owner: Arc, + ticket: u64, + closed: bool, +} + +impl WriteWindow { + fn new(owner: Arc, ticket: u64) -> Self { + Self { + owner, + ticket, + closed: false, + } + } + + /// Record an LSN this window minted. Call it before the window settles. + pub fn note_minted(&self, lsn: Lsn) { + self.owner.lock().note(lsn.as_u64()); + } + + /// Record that this window owns the record at `lsn`. Call it when the + /// record is appended. + pub fn own(&self, lsn: Lsn) { + self.owner.lock().own(self.ticket, lsn.as_u64()); + } + + /// Close the window: the write's outcome is final. + pub fn settle(mut self) { + self.closed = true; + self.owner.close(self.ticket); + } + + /// Keep the window open until the process exits: the record has no final + /// outcome here, and restart replay must reach it. Files a report naming + /// the caller. + #[track_caller] + pub fn hold(mut self) { + self.closed = true; + let site = std::panic::Location::caller(); + let (horizon, open_for) = self + .owner + .lock() + .mark_held(self.ticket) + .unwrap_or((0, Duration::ZERO)); + warn!( + ticket = self.ticket, + horizon, + site = %site, + "a write window is held until restart; the outcome floor stays below it" + ); + crate::diag::write_window_held(site, self.ticket, horizon, open_for); + } +} + +impl Drop for WriteWindow { + fn drop(&mut self) { + if !self.closed { + self.owner.leak(self.ticket); + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::atomic::{AtomicU64, Ordering}; + + use super::*; + + #[test] + fn an_empty_registry_has_floor_zero() { + assert_eq!(OutcomeFloor::new().floor(), Lsn::ZERO); + } + + #[test] + fn the_floor_never_passes_an_open_window() { + let floor = OutcomeFloor::new(); + let early = floor.open_write(); + early.note_minted(Lsn::new(10)); + let late = floor.open_write(); + late.note_minted(Lsn::new(11)); + late.settle(); + assert!( + floor.floor() < Lsn::new(10), + "the open window at 10 holds F" + ); + early.settle(); + assert_eq!(floor.floor(), Lsn::new(11)); + } + + #[test] + fn the_floor_advances_when_windows_settle() { + let floor = OutcomeFloor::new(); + let first = floor.open_write(); + first.note_minted(Lsn::new(5)); + first.settle(); + assert_eq!(floor.floor(), Lsn::new(5)); + let second = floor.open_write(); + second.note_minted(Lsn::new(9)); + assert_eq!(floor.floor(), Lsn::new(5)); + second.settle(); + assert_eq!(floor.floor(), Lsn::new(9)); + } + + #[test] + fn a_window_opened_before_its_mint_holds_the_floor_below_the_mint() { + let floor = OutcomeFloor::new(); + let done = floor.open_write(); + done.note_minted(Lsn::new(3)); + done.settle(); + let pending = floor.open_write(); + assert_eq!(floor.floor(), Lsn::new(3)); + pending.note_minted(Lsn::new(4)); + assert_eq!(floor.floor(), Lsn::new(3)); + pending.settle(); + } + + #[test] + fn a_dropped_window_holds_the_floor() { + let floor = OutcomeFloor::new(); + let before = floor.open_write(); + before.note_minted(Lsn::new(6)); + before.settle(); + let lost = floor.open_write(); + lost.note_minted(Lsn::new(7)); + drop(lost); + let later = floor.open_write(); + later.note_minted(Lsn::new(8)); + later.settle(); + assert_eq!(floor.floor(), Lsn::new(6)); + } + + #[test] + fn a_dropped_window_counts_as_a_leak() { + let floor = OutcomeFloor::new(); + assert_eq!(floor.leaked_windows(), 0); + drop(floor.open_write()); + assert_eq!(floor.leaked_windows(), 1); + floor.open_write().settle(); + floor.open_write().hold(); + assert_eq!( + floor.leaked_windows(), + 1, + "a settled or held window is not a leak" + ); + } + + #[test] + fn a_held_window_holds_the_floor() { + let floor = OutcomeFloor::new(); + let before = floor.open_write(); + before.note_minted(Lsn::new(4)); + before.settle(); + let held = floor.open_write(); + held.note_minted(Lsn::new(5)); + held.hold(); + let later = floor.open_write(); + later.note_minted(Lsn::new(6)); + later.settle(); + assert_eq!(floor.floor(), Lsn::new(4)); + } + + #[test] + fn a_window_open_past_the_bound_reports_a_stuck_floor() { + let floor = OutcomeFloor::new(); + assert!(floor.stuck(Duration::ZERO).is_none(), "no open window"); + let window = floor.open_write(); + window.note_minted(Lsn::new(3)); + std::thread::sleep(Duration::from_millis(2)); + let stuck = floor + .stuck(Duration::ZERO) + .expect("the window is older than a zero bound"); + assert_eq!(stuck.horizon, Lsn::new(1)); + assert_eq!(stuck.floor, Lsn::ZERO); + assert_eq!(stuck.open_windows, 1); + assert!(floor.oldest_open_for() > Duration::ZERO); + assert!( + floor.stuck(Duration::from_secs(3600)).is_none(), + "a young window is inside the bound" + ); + window.settle(); + assert!(floor.stuck(Duration::ZERO).is_none()); + assert_eq!(floor.oldest_open_for(), Duration::ZERO); + } + + /// A held window stays open by design. It never makes the floor stuck, + /// and the held count reports it. + #[test] + fn a_held_window_is_counted_and_never_stuck() { + let floor = OutcomeFloor::new(); + let held = floor.open_write(); + held.note_minted(Lsn::new(3)); + held.hold(); + std::thread::sleep(Duration::from_millis(2)); + assert_eq!(floor.held_windows(), 1); + assert!(floor.stuck(Duration::ZERO).is_none()); + assert_eq!(floor.oldest_open_for(), Duration::ZERO); + + let open = floor.open_write(); + std::thread::sleep(Duration::from_millis(2)); + let stuck = floor + .stuck(Duration::ZERO) + .expect("the window that is not held is older than a zero bound"); + assert_eq!(stuck.open_windows, 1); + open.settle(); + assert!(floor.stuck(Duration::ZERO).is_none()); + } + + #[test] + fn an_existing_record_the_floor_passed_opens_no_window() { + let floor = OutcomeFloor::new(); + let window = floor.open_write(); + window.note_minted(Lsn::new(8)); + window.settle(); + assert_eq!( + floor.open_existing(Lsn::new(8)).err(), + Some(ResendRefusal::BelowFloor) + ); + let resent = floor + .open_existing(Lsn::new(9)) + .expect("the floor has not passed 9"); + assert_eq!(floor.floor(), Lsn::new(8)); + resent.settle(); + assert_eq!(floor.floor(), Lsn::new(9)); + } + + /// A held window keeps the floor down for the rest of the process. The + /// closed LSNs above it stay a bounded number of ranges however many + /// records close, and each stays refused. + #[test] + fn closed_lsns_stay_bounded_while_a_window_is_held() { + let floor = OutcomeFloor::new(); + floor.open_dispatched(Lsn::new(10)).hold(); + for lsn in 11..5_011u64 { + floor.open_dispatched(Lsn::new(lsn)).settle(); + } + assert_eq!( + floor.floor(), + Lsn::new(9), + "the held window keeps the floor down" + ); + assert!( + floor.lock().closed.range_count() <= 2, + "closed LSNs must stay bounded while a window is held" + ); + assert_eq!( + floor.open_existing(Lsn::new(2_500)).err(), + Some(ResendRefusal::Closed) + ); + assert_eq!( + floor.open_existing(Lsn::new(10)).err(), + Some(ResendRefusal::Owned), + "the held window still owns its record" + ); + } + + #[test] + fn a_dispatched_window_holds_the_floor_below_its_lsn() { + let floor = OutcomeFloor::new(); + let dispatched = floor.open_dispatched(Lsn::new(20)); + assert_eq!(floor.floor(), Lsn::new(19)); + dispatched.settle(); + assert_eq!(floor.floor(), Lsn::new(20)); + } + + #[test] + fn a_dispatched_window_below_the_floor_does_not_lower_it() { + let floor = OutcomeFloor::new(); + let window = floor.open_write(); + window.note_minted(Lsn::new(30)); + window.settle(); + assert_eq!(floor.floor(), Lsn::new(30)); + let late = floor.open_dispatched(Lsn::new(12)); + assert_eq!(floor.floor(), Lsn::new(30), "the floor never decreases"); + late.settle(); + } + + /// Writers mint from a shared counter the way the WAL does, each inside a + /// window. A reader samples the floor while they run. No sampled floor may + /// reach an LSN whose window was still open when the sample was taken. + #[test] + fn concurrent_writers_never_see_the_floor_pass_their_open_window() { + let floor = OutcomeFloor::new(); + let wal = Arc::new(AtomicU64::new(1)); + let writers: Vec<_> = (0..4) + .map(|_| { + let floor = Arc::clone(&floor); + let wal = Arc::clone(&wal); + std::thread::spawn(move || { + for _ in 0..500 { + let window = floor.open_write(); + let lsn = Lsn::new(wal.fetch_add(1, Ordering::SeqCst)); + window.note_minted(lsn); + let seen = floor.floor(); + assert!(seen < lsn, "floor {seen:?} passed open lsn {lsn:?}"); + window.settle(); + } + }) + }) + .collect(); + let mut last = Lsn::ZERO; + for _ in 0..2000 { + let seen = floor.floor(); + assert!( + seen >= last, + "the floor went back from {last:?} to {seen:?}" + ); + last = seen; + } + for writer in writers { + writer.join().expect("writer thread"); + } + let minted = wal.load(Ordering::SeqCst) - 1; + assert_eq!(floor.floor(), Lsn::new(minted), "every window settled"); + } +} diff --git a/nodedb/src/bridge/dispatch/refusal.rs b/nodedb/src/bridge/dispatch/refusal.rs new file mode 100644 index 000000000..5259d129d --- /dev/null +++ b/nodedb/src/bridge/dispatch/refusal.rs @@ -0,0 +1,24 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A refused dispatch that hands the unsent request back to the caller. + +use crate::bridge::envelope; + +/// A request the dispatcher refused, returned unsent with the reason. +/// +/// A caller that retries a capacity refusal re-dispatches `request` as is, +/// with no clone and no rebuild. +#[derive(Debug)] +pub struct DispatchRefusal { + /// Why the dispatcher refused the request. + pub error: crate::Error, + /// The refused request. The dispatcher tracked nothing for it. + pub request: envelope::Request, +} + +impl DispatchRefusal { + /// Box a refusal of `request` for `error`. + pub(super) fn boxed(error: crate::Error, request: envelope::Request) -> Box { + Box::new(Self { error, request }) + } +} diff --git a/nodedb/src/bridge/dispatch/response_poll.rs b/nodedb/src/bridge/dispatch/response_poll.rs new file mode 100644 index 000000000..a999c5007 --- /dev/null +++ b/nodedb/src/bridge/dispatch/response_poll.rs @@ -0,0 +1,485 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Response polling for the bridge [`Dispatcher`], including the error +//! responses it synthesizes for a core that died. + +use std::collections::HashSet; + +use crate::bridge::envelope; +use crate::bridge::envelope::{ErrorCode, Payload, Status}; +use crate::types::{Lsn, RequestId}; + +use super::dispatcher::Dispatcher; +use super::enqueue::release_inflight_slot; + +impl Dispatcher { + /// Poll responses from all Data Plane cores. + /// + /// A core whose channel has been observed dead contributes a synthesized + /// error `Response` for every request still outstanding on it: the one a + /// failed `try_push` consumed, everything still staged in its WFQ, and + /// everything dispatched earlier that it never answered. Those travel back + /// with the real responses so the single completion loop in the caller + /// finishes each waiter, and the loop below releases each request's + /// `tenant_inflight` slot exactly as a real response would — without which + /// one dead core ratchets the tenant's in-flight count until the tenant is + /// rejected on healthy cores too. + /// + /// Fires [`Dispatcher::capacity_freed`] once when the poll released at + /// least one in-flight slot. + pub fn poll_responses(&mut self) -> Vec { + let mut responses = Vec::new(); + let mut freed = false; + for (core_id, channel) in self.cores.iter_mut().enumerate() { + let mut batch = Vec::new(); + let (_drained, producer_gone) = channel.response_rx.drain_into(&mut batch, 64); + for br in batch { + let rid = br.inner.request_id.as_u64(); + // A streaming scan answers with many partials before its final + // response. The request is still executing on the core until + // that final one arrives, so releasing it here would let the + // shutdown drain call a live scan finished and would drop the + // tenant's in-flight slot mid-stream. + if !br.inner.partial { + channel.outstanding.remove(&rid); + self.dispatched_lsns.settle(rid); + freed |= release_inflight_slot( + &mut self.request_tenant, + &mut self.tenant_inflight, + rid, + ); + } + responses.push(br.inner); + } + + if !(producer_gone || channel.request_tx.is_disconnected()) { + // Opportunistically flush WFQ after draining responses to fill headroom. + channel.flush_wfq(self.outcome_floor.floor()); + continue; + } + + // The core is gone. Collect every request it can no longer answer: + // items still staged in the WFQ first (dispatch order), then the + // rest of the outstanding set. A staged item is also in + // `outstanding`, so `seen` keeps each id to a single response. + let mut seen = HashSet::new(); + let mut lost = Vec::new(); + for staged in channel.wfq.drain() { + let rid = staged.request_id.as_u64(); + if seen.insert(rid) { + lost.push(rid); + } + } + for rid in channel.outstanding.drain() { + if seen.insert(rid) { + lost.push(rid); + } + } + + // Idempotence: both sources are emptied here — `wfq.drain` leaves + // the staging queue empty and `outstanding.drain` clears the set — + // and `flush_wfq` refuses to stage anything new onto a + // disconnected producer. A later poll therefore finds both empty + // and emits nothing, so a permanently dead core costs one pass + // over two empty containers rather than a repeating failure storm. + for rid in lost { + // A dead core never publishes a watermark again. + self.dispatched_lsns.settle(rid); + freed |= + release_inflight_slot(&mut self.request_tenant, &mut self.tenant_inflight, rid); + responses.push(envelope::Response { + request_id: RequestId::new(rid), + status: Status::Error, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: Some(Box::new(ErrorCode::Internal { + detail: format!( + "core-{core_id} is gone; the request can never be executed" + ), + })), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + }); + } + } + if freed { + self.capacity_freed.notify_waiters(); + } + responses + } +} + +#[cfg(test)] +mod tests { + use std::time::Duration; + + use super::*; + use crate::bridge::dispatch::BridgeResponse; + use crate::bridge::dispatch::test_requests::{make_request, make_request_for_db}; + use crate::types::*; + + #[test] + fn response_roundtrip() { + let (mut dispatcher, mut data_sides) = Dispatcher::new(2, 64); + + dispatcher.dispatch(make_request(0)).unwrap(); + + let _req = data_sides[0].request_rx.try_pop().unwrap(); + data_sides[0] + .response_tx + .try_push(BridgeResponse { + inner: envelope::Response { + request_id: RequestId::new(1), + status: Status::Ok, + attempt: 1, + partial: false, + payload: Payload::from_vec(b"result".to_vec()), + watermark_lsn: Lsn::new(42), + error_code: None, + read_set_valid: None, + read_version_lsn: crate::types::Lsn::ZERO, + write_set: Vec::new(), + }, + }) + .unwrap(); + + let responses = dispatcher.poll_responses(); + assert_eq!(responses.len(), 1); + assert_eq!(responses[0].status, Status::Ok); + assert_eq!(&*responses[0].payload, b"result"); + } + + /// A routed final response frees its tenant's slot and fires the + /// capacity-freed signal. + #[tokio::test] + async fn final_response_fires_capacity_freed() { + let (mut dispatcher, mut data_sides) = Dispatcher::new(1, 64); + dispatcher.dispatch(make_request(0)).unwrap(); + let _req = data_sides[0].request_rx.try_pop().unwrap(); + data_sides[0] + .response_tx + .try_push(BridgeResponse { + inner: envelope::Response { + request_id: RequestId::new(1), + status: Status::Ok, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: None, + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + }, + }) + .unwrap(); + + let signal = dispatcher.capacity_freed(); + let notified = signal.notified(); + tokio::pin!(notified); + notified.as_mut().enable(); + assert_eq!(dispatcher.poll_responses().len(), 1); + + assert!( + tokio::time::timeout(Duration::from_millis(100), notified) + .await + .is_ok(), + "a freed in-flight slot must fire the capacity-freed signal" + ); + } + + // --- Dead-core request loss --- + // + // When a Data Plane core's consumer/producer is dropped (the core thread + // died), `Dispatcher` must synthesize an error `Response` for every + // request it knows is outstanding on that core, rather than dropping the + // request silently and leaking the caller's waiter + `tenant_inflight` + // slot forever. Dropping one element of the `data_sides` vector handed + // back by `Dispatcher::new`/`with_resolver` simulates that core thread + // dying, matching how `dispatch_routes_to_correct_core` and + // `response_roundtrip` above obtain the data-plane side of the channel. + + #[test] + fn dead_core_synthesizes_error_response_for_lost_request() { + let (mut dispatcher, mut data_sides) = Dispatcher::new(3, 64); + + // Core 2's thread has died: both halves of its data-plane side are gone. + let dead_core = 2; + drop(data_sides.remove(dead_core)); + + let request = make_request_for_db(0, 1, 7); + let request_id = request.request_id.as_u64(); + + // `dispatch_to_core` still reports success: the request was already + // moved into the doomed `try_push` inside `flush_wfq` before the + // failure is observed, which is exactly the defect being covered. + dispatcher.dispatch_to_core(dead_core, request).unwrap(); + + let responses = dispatcher.poll_responses(); + assert_eq!(responses.len(), 1, "expected one synthesized response"); + let resp = &responses[0]; + assert_eq!(resp.request_id.as_u64(), request_id); + assert_eq!(resp.status, Status::Error); + match resp.error_code.as_deref() { + Some(ErrorCode::Internal { detail }) => { + assert!( + detail.contains(&dead_core.to_string()), + "error detail should name the dead core, got: {detail}" + ); + } + other => panic!("expected ErrorCode::Internal naming the core, got: {other:?}"), + } + } + + #[test] + fn dead_core_synthesized_response_resets_tenant_inflight() { + // The ratchet: `tenant_inflight` is incremented on dispatch and must + // return to its pre-dispatch value once the synthesized response for + // the lost request is drained through `poll_responses` — otherwise it + // climbs forever and eventually starves the tenant on healthy cores. + let (mut dispatcher, mut data_sides) = Dispatcher::new(2, 64); + let dead_core = 0; + drop(data_sides.remove(dead_core)); + + let request = make_request_for_db(0, 1, 1); + let tenant_id = request.tenant_id.as_u64(); + + let before = dispatcher + .tenant_inflight + .get(&tenant_id) + .copied() + .unwrap_or(0); + + dispatcher.dispatch_to_core(dead_core, request).unwrap(); + assert_eq!( + dispatcher.tenant_inflight.get(&tenant_id).copied(), + Some(before + 1), + "dispatch must still increment tenant_inflight even though the core is dead" + ); + + let responses = dispatcher.poll_responses(); + assert_eq!(responses.len(), 1); + assert_eq!( + dispatcher + .tenant_inflight + .get(&tenant_id) + .copied() + .unwrap_or(0), + before, + "tenant_inflight must return to its pre-dispatch value, not ratchet upward" + ); + assert!(!dispatcher.request_tenant.contains_key(&1)); + } + + #[test] + fn dead_core_does_not_affect_live_core() { + let (mut dispatcher, mut data_sides) = Dispatcher::new(2, 64); + let dead_core = 0; + let live_core = 1; + drop(data_sides.remove(dead_core)); + // Removing index 0 shifted core 1's data side down to index 0. + let live_data_side = &mut data_sides[0]; + + let dead_request = make_request_for_db(0, 1, 1); + let live_request = make_request_for_db(0, 2, 2); + let live_request_id = live_request.request_id.as_u64(); + + dispatcher + .dispatch_to_core(dead_core, dead_request) + .unwrap(); + dispatcher + .dispatch_to_core(live_core, live_request) + .unwrap(); + + // The live core answers normally, through the real ring buffer. + let _req = live_data_side.request_rx.try_pop().unwrap(); + live_data_side + .response_tx + .try_push(BridgeResponse { + inner: envelope::Response { + request_id: RequestId::new(live_request_id), + status: Status::Ok, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: None, + read_set_valid: None, + read_version_lsn: crate::types::Lsn::ZERO, + write_set: Vec::new(), + }, + }) + .unwrap(); + + let responses = dispatcher.poll_responses(); + assert_eq!( + responses.len(), + 2, + "one synthesized error from the dead core, one real Ok from the live core" + ); + + let live_resp = responses + .iter() + .find(|r| r.request_id.as_u64() == live_request_id) + .expect("live core's real response must be present"); + assert_eq!(live_resp.status, Status::Ok); + assert!(live_resp.error_code.is_none()); + + let dead_resp = responses + .iter() + .find(|r| r.request_id.as_u64() != live_request_id) + .expect("dead core's synthesized response must be present"); + assert_eq!(dead_resp.status, Status::Error); + assert!(dead_resp.error_code.is_some()); + } + + #[test] + fn dead_core_fails_requests_still_queued_in_wfq() { + // Fill the physical ring to capacity while the core is alive, so a + // request dispatched afterward parks in the WFQ without ever + // attempting a push (flush_wfq's utilization check breaks before it + // reaches the doomed try_push). Then kill the core and confirm the + // WFQ-queued request is failed too, not left sitting in the queue + // forever. + let (mut dispatcher, mut data_sides) = Dispatcher::new(1, 4); + + for i in 0..4u64 { + dispatcher + .dispatch_to_core(0, make_request_for_db(0, i + 1, i + 1)) + .unwrap(); + } + assert_eq!(data_sides[0].request_rx.len(), 4); + + // Core 0's thread dies with 4 unanswered requests sitting in its ring. + drop(data_sides.remove(0)); + + // This request cannot reach the (full, dead) physical ring — it stays + // parked in the WFQ. + let parked_request_id = 99u64; + dispatcher + .dispatch_to_core(0, make_request_for_db(0, 99, parked_request_id)) + .unwrap(); + + let responses = dispatcher.poll_responses(); + let ids: std::collections::HashSet = + responses.iter().map(|r| r.request_id.as_u64()).collect(); + + // The 4 previously-dispatched-but-unanswered requests, plus the one + // still parked in the WFQ, must all be failed. + assert_eq!( + responses.len(), + 5, + "expected all 5 outstanding requests failed" + ); + for id in 1..=4u64 { + assert!( + ids.contains(&id), + "request {id} in the dead ring must be failed" + ); + } + assert!( + ids.contains(&parked_request_id), + "request parked in the WFQ must be failed, not left queued" + ); + for r in &responses { + assert_eq!(r.status, Status::Error); + assert!(r.error_code.is_some()); + } + } + + // --- Outcome floor --- + + fn ok_response(request_id: u64) -> BridgeResponse { + BridgeResponse { + inner: envelope::Response { + request_id: RequestId::new(request_id), + status: Status::Ok, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: None, + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + }, + } + } + + #[test] + fn a_dispatched_lsn_holds_the_outcome_floor_until_its_response() { + let (mut dispatcher, mut data_sides) = Dispatcher::new(1, 64); + let floor = dispatcher.outcome_floor(); + let mut request = make_request_for_db(0, 0, 5); + request.wal_lsn = Some(Lsn::new(40)); + dispatcher.dispatch(request).unwrap(); + assert_eq!(dispatcher.dispatched_lsns.len(), 1); + assert_eq!(floor.floor(), Lsn::new(39)); + + let pushed = data_sides[0].request_rx.try_pop().unwrap(); + assert!( + pushed.outcome_floor < Lsn::new(40), + "the request's own push carries a floor below its lsn" + ); + + data_sides[0].response_tx.try_push(ok_response(5)).unwrap(); + assert_eq!(dispatcher.poll_responses().len(), 1); + assert_eq!(dispatcher.dispatched_lsns.len(), 0); + assert_eq!(floor.floor(), Lsn::new(40)); + + dispatcher.dispatch(make_request_for_db(0, 0, 6)).unwrap(); + let next = data_sides[0].request_rx.try_pop().unwrap(); + assert_eq!( + next.outcome_floor, + Lsn::new(40), + "a push after the response carries the advanced floor" + ); + } + + #[test] + fn a_partial_response_keeps_the_outcome_floor_held() { + let (mut dispatcher, mut data_sides) = Dispatcher::new(1, 64); + let floor = dispatcher.outcome_floor(); + let mut request = make_request_for_db(0, 0, 8); + request.wal_lsn = Some(Lsn::new(12)); + dispatcher.dispatch(request).unwrap(); + let _req = data_sides[0].request_rx.try_pop().unwrap(); + + let mut partial = ok_response(8); + partial.inner.partial = true; + data_sides[0].response_tx.try_push(partial).unwrap(); + dispatcher.poll_responses(); + assert_eq!(floor.floor(), Lsn::new(11)); + + data_sides[0].response_tx.try_push(ok_response(8)).unwrap(); + dispatcher.poll_responses(); + assert_eq!(floor.floor(), Lsn::new(12)); + } + + #[test] + fn a_dead_core_releases_the_outcome_floor_it_held() { + let (mut dispatcher, mut data_sides) = Dispatcher::new(1, 64); + let floor = dispatcher.outcome_floor(); + let mut request = make_request_for_db(0, 0, 3); + request.wal_lsn = Some(Lsn::new(25)); + dispatcher.dispatch(request).unwrap(); + assert_eq!(floor.floor(), Lsn::new(24)); + + drop(data_sides.remove(0)); + let responses = dispatcher.poll_responses(); + assert_eq!(responses.len(), 1); + assert_eq!(dispatcher.dispatched_lsns.len(), 0); + assert_eq!(floor.floor(), Lsn::new(25)); + } + + #[test] + fn a_request_without_an_lsn_holds_nothing() { + let (mut dispatcher, _data_sides) = Dispatcher::new(1, 64); + dispatcher.dispatch(make_request_for_db(0, 0, 4)).unwrap(); + assert_eq!(dispatcher.dispatched_lsns.len(), 0); + assert_eq!(dispatcher.outcome_floor().floor(), Lsn::ZERO); + } +} diff --git a/nodedb/src/bridge/dispatch/test_requests.rs b/nodedb/src/bridge/dispatch/test_requests.rs new file mode 100644 index 000000000..442aae21a --- /dev/null +++ b/nodedb/src/bridge/dispatch/test_requests.rs @@ -0,0 +1,75 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Request builders shared by the dispatcher's unit tests. + +use std::time::{Duration, Instant}; + +use nodedb_physical::physical_plan::DocumentOp; + +use crate::bridge::envelope; +use crate::bridge::envelope::*; +use crate::types::*; + +/// A read request with id 1 for tenant 1 in the default database on `vshard`. +pub(super) fn make_request(vshard: u32) -> envelope::Request { + envelope::Request { + request_id: RequestId::new(1), + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(vshard), + plan: PhysicalPlan::Document(DocumentOp::PointGet { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "users"), + document_id: "u1".into(), + surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: Vec::new(), + rls_filters: Vec::new(), + system_time: nodedb_types::SystemTimeScope::Current, + valid_at_ms: None, + }), + deadline: Instant::now() + Duration::from_secs(5), + priority: Priority::Normal, + trace_id: TraceId::ZERO, + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: crate::event::EventSource::User, + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + admission: Admission::Exempt(ExemptReason::Read), + } +} + +/// A read request with id `req_id` for tenant 1 in database `db` on `vshard`. +pub(super) fn make_request_for_db(vshard: u32, db: u64, req_id: u64) -> envelope::Request { + envelope::Request { + request_id: RequestId::new(req_id), + tenant_id: TenantId::new(1), + database_id: DatabaseId::new(db), + vshard_id: VShardId::new(vshard), + plan: PhysicalPlan::Document(DocumentOp::PointGet { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::new(db), "c"), + document_id: "d".into(), + surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: Vec::new(), + rls_filters: Vec::new(), + system_time: nodedb_types::SystemTimeScope::Current, + valid_at_ms: None, + }), + deadline: Instant::now() + Duration::from_secs(5), + priority: Priority::Normal, + trace_id: TraceId::ZERO, + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: crate::event::EventSource::User, + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + admission: Admission::Exempt(ExemptReason::Read), + } +} diff --git a/nodedb/src/bridge/envelope/error_code.rs b/nodedb/src/bridge/envelope/error_code.rs index b0a115121..33f7ff573 100644 --- a/nodedb/src/bridge/envelope/error_code.rs +++ b/nodedb/src/bridge/envelope/error_code.rs @@ -25,6 +25,20 @@ pub enum ErrorCode { /// a retry channel can tell a retry apart from a permanent refusal instead /// of collapsing both into a terminal rejection. RetryableRefusal { reason: String }, + /// A sync frame the validator refused for good. Nothing applied. The + /// stream's high-water mark advanced to `provenance`, so the frame is + /// never admitted again, and `applied_seq` is the mark after the refusal. + SyncRejected { + violation: nodedb_types::sync::violation::ViolationType, + applied_seq: u64, + provenance: nodedb_types::sync::wire::SyncProvenance, + }, + /// A sync frame the idempotency gate held back. Nothing applied, and + /// the stream's mark did not move. `applied_seq` is that mark. + SyncNotApplied { + hold: super::SyncHold, + applied_seq: u64, + }, /// Document/collection not found. NotFound, /// Authorization failure. @@ -79,8 +93,12 @@ pub enum ErrorCode { TypeGuardViolation { collection: String, detail: String }, /// Value type does not match expected type for operation (e.g. INCR on a string). TypeMismatch { collection: String, detail: String }, - /// Arithmetic overflow (e.g. i64::MAX + 1 on INCR). - OverflowError { collection: String }, + /// A KV counter atomic read a stored value it cannot parse as a + /// number, or computed a result out of range. + CounterFault { + collection: String, + fault: super::CounterFault, + }, /// Insufficient balance for transfer (source lacks required amount). InsufficientBalance { collection: String, detail: String }, /// Rate limit exceeded for a rate gate / cooldown. @@ -132,6 +150,55 @@ pub enum ErrorCode { /// special-cases `NotFound`) and reaches the client as SQLSTATE `22012` /// rather than the generic `XX000` every `Internal` maps to. DivisionByZero, + /// Expression evaluation called a function no evaluator implements. + /// Surfaces as SQLSTATE `42883` (`undefined_function`), never as a + /// silent `NULL`. + UndefinedFunction { name: String }, + /// A function received an argument it cannot compute on: vectors of + /// different dimensions, an argument of the wrong type, a malformed + /// JSONPath. Surfaces as SQLSTATE `22000` (`data_exception`). + DataException { detail: String }, + /// The bridge dispatcher refused the request at a capacity limit, so + /// nothing was enqueued or applied. Transient: the same request succeeds + /// once capacity frees. `reason` names the limit and its counts. + DispatchCapacity { reason: String }, + /// The request's deadline passed before the core started it, so nothing + /// ran. Distinct from [`Self::DeadlineExceeded`], which a core also + /// answers for a task it stopped part way. Surfaces as the same + /// query-cancelled error. + ExpiredBeforeExecution, + /// The request itself is malformed: an FTS query with no positive term, + /// a value the target cannot hold. The same verdict the Control Plane + /// gives `crate::Error::BadRequest`: SQLSTATE `42601` (syntax_error). + BadRequest { detail: String }, + /// The whole transaction aborted before any read-set was validated, and + /// the client retries it. SQLSTATE `40000` (transaction_rollback). + TransactionRollback { detail: String }, + /// The statement cannot run in the current transaction state, such as + /// inside an explicit transaction block. SQLSTATE `25001` + /// (active_sql_transaction). + ActiveSqlTransaction { detail: String }, + /// A DROP refused because other objects depend on its target. `object` + /// names the target. SQLSTATE `2BP01` (dependent_objects_still_exist). + DependentObjectsExist { object: String, detail: String }, +} + +/// An expression evaluation failure, as the Data Plane reports it. +/// +/// Exhaustive, so a new evaluator error picks its own code rather than +/// defaulting to one. +impl From for ErrorCode { + fn from(e: nodedb_query::EvalError) -> Self { + match e { + nodedb_query::EvalError::DivisionByZero => Self::DivisionByZero, + nodedb_query::EvalError::UnknownFunction { name } => Self::UndefinedFunction { name }, + e @ (nodedb_query::EvalError::VectorDimensionMismatch { .. } + | nodedb_query::EvalError::ArgumentType { .. } + | nodedb_query::EvalError::InvalidJsonPath { .. }) => Self::DataException { + detail: e.to_string(), + }, + } + } } impl From for ErrorCode { @@ -145,11 +212,14 @@ impl From for ErrorCode { Self::RejectedPrevalidation { reason } } crate::Error::RetryableRefusal { reason } => Self::RetryableRefusal { reason }, - crate::Error::CollectionNotFound { .. } | crate::Error::DocumentNotFound { .. } => { - Self::NotFound - } + crate::Error::CollectionNotFound { .. } + | crate::Error::CollectionDeactivated { .. } + | crate::Error::DocumentNotFound { .. } => Self::NotFound, crate::Error::RejectedAuthz { resource, .. } => Self::RejectedAuthz { resource }, - crate::Error::ConflictRetry { .. } => Self::ConflictRetry, + // The Control Plane gives all three `40001` (serialization_failure). + crate::Error::ConflictRetry { .. } + | crate::Error::CalvinSerializationConflict + | crate::Error::SourceFrozen { .. } => Self::ConflictRetry, crate::Error::FanOutExceeded { .. } => Self::FanOutExceeded, crate::Error::MemoryExhausted { .. } => Self::ResourcesExhausted, crate::Error::Backpressure { .. } => Self::ResourcesExhausted, @@ -206,7 +276,6 @@ impl From for ErrorCode { crate::Error::TypeMismatch { collection, detail, .. } => Self::TypeMismatch { collection, detail }, - crate::Error::OverflowError { collection, .. } => Self::OverflowError { collection }, crate::Error::InsufficientBalance { collection, detail, .. } => Self::InsufficientBalance { collection, detail }, @@ -218,10 +287,26 @@ impl From for ErrorCode { gate, retry_after_ms, }, + // A capacity refusal enqueued nothing, and the same request + // succeeds once capacity frees. + capacity @ crate::Error::DispatchCapacity { .. } => Self::DispatchCapacity { + reason: capacity.to_string(), + }, crate::Error::TxnOverlayMemoryExceeded { limit } => { Self::TxnOverlayMemoryExceeded { limit } } crate::Error::DivisionByZero => Self::DivisionByZero, + crate::Error::UndefinedFunction { name } => Self::UndefinedFunction { name }, + crate::Error::DataException { detail } => Self::DataException { detail }, + // `42601` (syntax_error), as the Control Plane gives both. + crate::Error::BadRequest { detail } | crate::Error::PlanError { detail } => { + Self::BadRequest { detail } + } + // `0A000` (feature_not_supported), as the Control Plane gives both. + crate::Error::FeatureNotSupported { detail } => Self::Unsupported { detail }, + unsupported @ crate::Error::CrossCollectionNotColocated { .. } => Self::Unsupported { + detail: unsupported.to_string(), + }, crate::Error::UndefinedColumn { column } => Self::UndefinedColumn { column }, // Same condition an undefined column reports at plan time, raised // here by the strict encoder for a transport the planner never @@ -230,8 +315,142 @@ impl From for ErrorCode { // Already a Data-Plane verdict: hand back the same code rather // than re-wrapping it as `Internal` and losing its SQLSTATE. crate::Error::DataPlane(code) => code, - other => Self::Internal { - detail: other.to_string(), + // Class `22`, the class the Control Plane gives both. + e @ (crate::Error::OffsetRegression { .. } + | crate::Error::BackupTenantMismatch { .. } + | crate::Error::InvalidLimitValue { .. }) => Self::DataException { + detail: e.to_string(), + }, + // `40000`, as the Control Plane gives it. + e @ crate::Error::CalvinParticipantError => Self::TransactionRollback { + detail: e.to_string(), + }, + // `40001`: the client retries the statement. + e @ crate::Error::RetryableSchemaChanged { .. } => Self::RetryableRefusal { + reason: e.to_string(), + }, + // `25001`, as the Control Plane gives all three. + e @ (crate::Error::CrdtApplyForbiddenInTransaction + | crate::Error::NotInTransactionBlock { .. } + | crate::Error::CrossShardInExplicitTransaction) => Self::ActiveSqlTransaction { + detail: e.to_string(), + }, + // `2BP01`, as the Control Plane gives both. The detail is the + // public message the Control Plane renders. + crate::Error::DependentObjectsExist { + root_kind, + root_name, + dependent_count, + dependents, + .. + } => { + let (object, detail) = crate::error_classify::dependent_objects_text( + root_kind, + &root_name, + dependent_count, + &dependents, + ); + Self::DependentObjectsExist { object, detail } + } + crate::Error::RoleInUse { role, dependents } => { + let object = format!("role \"{role}\""); + let detail = crate::Error::RoleInUse { role, dependents }.to_string(); + Self::DependentObjectsExist { object, detail } + } + crate::Error::CrdtAdmissionRetriesExhausted { .. } => Self::ConflictRetry, + // Retryable refusals whose class (`55P03`) no Data-Plane code has. + // The retry contract survives: nothing was applied. + e @ (crate::Error::NoLeader { .. } + | crate::Error::GroupQuorumUnavailable { .. } + | crate::Error::GroupMarksUnavailable { .. } + | crate::Error::AuthorizationStateBehind { .. } + | crate::Error::StaleReadNotLeader { .. }) => Self::RetryableRefusal { + reason: e.to_string(), + }, + // Class `57`: the client retries once the leader settles. + e @ crate::Error::NotLeader { .. } => Self::DispatchCapacity { + reason: e.to_string(), + }, + crate::Error::CrdtAdmissionTimeout { .. } => Self::DeadlineExceeded, + e @ crate::Error::VShardAdmissionCapacityExceeded { .. } => Self::RateExceeded { + gate: e.to_string(), + retry_after_ms: 0, + }, + // Class `53`: a configured resource ceiling. + crate::Error::QuotaOvercommit { .. } + | crate::Error::TenantVectorDimExceeded { .. } + | crate::Error::TenantGraphDepthExceeded { .. } => Self::ResourcesExhausted, + // Class `28` has no Data-Plane code. The nearest is the access + // refusal, which keeps it a client error the client cannot retry. + e @ (crate::Error::BackupKeyMismatch | crate::Error::SessionTokenExpired) => { + Self::RejectedAuthz { + resource: e.to_string(), + } + } + // Client errors of class `42`, and client errors whose class + // (`25006`, `55`) no Data-Plane code has. `BadRequest` is the + // class their public code has. + e @ (crate::Error::CrdtAdmissionInvalidPlan { .. } + | crate::Error::CrdtAdmissionCallerFence + | crate::Error::CrdtApplyRequiresAdmission + | crate::Error::CloneWriteRequiresMaterialize { .. } + | crate::Error::ObjectNotInPrerequisiteState { .. } + | crate::Error::MirrorReadOnly { .. } + | crate::Error::UndefinedObject { .. } + | crate::Error::AmbiguousColumn { .. } + | crate::Error::ExecutionLimitExceeded { .. } + | crate::Error::LimitExceeded { .. } + | crate::Error::Promql(_) + | crate::Error::SequencerUnavailable + | crate::Error::SessionCapExceeded { .. } + | crate::Error::SessionIdleTimeout + | crate::Error::SessionKilledByAdmin + | crate::Error::SessionUserDropped + | crate::Error::OidcProviderTenantUnbound + | crate::Error::OidcProviderTenantUnavailable { .. } + | crate::Error::ExternalRoleUndefined { .. } + | crate::Error::OidcNoDefaultDatabase { .. } + | crate::Error::RoleInheritanceCycle { .. } + | crate::Error::RoleInheritanceDepthExceeded { .. }) => Self::BadRequest { + detail: e.to_string(), + }, + // Retry exhaustion takes the code of its cause. + crate::Error::OllpExhausted { cause, .. } => match cause { + crate::OllpExhaustedCause::PredicateDrift => Self::ConflictRetry, + crate::OllpExhaustedCause::PreAdmission(inner) => Self::from(*inner), + crate::OllpExhaustedCause::AdmissionRefused { detail } => Self::RateExceeded { + gate: detail, + retry_after_ms: 0, + }, + }, + // Server-side faults and system defects. `Shaping`, + // `RemoteTyped` and `Ddl` carry a public numeric code that has no + // Data-Plane twin, and none is raised on the Data Plane. + e @ (crate::Error::MaterializedSumResolutionMissing { .. } + | crate::Error::RetryableLeaderChange { .. } + | crate::Error::MetadataLeaderUnavailable + | crate::Error::Wal(_) + | crate::Error::Dispatch { .. } + | crate::Error::Storage { .. } + | crate::Error::ColdStorage { .. } + | crate::Error::Serialization { .. } + | crate::Error::Codec { .. } + | crate::Error::SegmentCorrupted { .. } + | crate::Error::Crdt(_) + | crate::Error::Io(_) + | crate::Error::Config { .. } + | crate::Error::Encryption { .. } + | crate::Error::Bridge { .. } + | crate::Error::VersionCompat { .. } + | crate::Error::Internal { .. } + | crate::Error::Shaping(_) + | crate::Error::RemoteTyped { .. } + | crate::Error::Ddl(_) + | crate::Error::DescriptorVersionAnomaly { .. } + | crate::Error::CollectionPurgeRowMissing { .. } + | crate::Error::CatalogIntegrityViolation { .. } + | crate::Error::CascadeCycle { .. }) => Self::Internal { + detail: e.to_string(), }, } } diff --git a/nodedb/src/bridge/envelope/mod.rs b/nodedb/src/bridge/envelope/mod.rs index f8b35d6ac..f09bb0a47 100644 --- a/nodedb/src/bridge/envelope/mod.rs +++ b/nodedb/src/bridge/envelope/mod.rs @@ -7,10 +7,13 @@ pub mod payload; pub mod request; pub mod response; pub mod status; +pub mod sync_hold; pub use error_code::ErrorCode; +pub use nodedb_physical::kv_atomic::CounterFault; pub use nodedb_physical::physical_plan::PhysicalPlan; pub use payload::Payload; pub use request::{Admission, ExemptReason, Request}; pub use response::{Response, WriteSetEntry}; pub use status::{Priority, Status}; +pub use sync_hold::SyncHold; diff --git a/nodedb/src/bridge/envelope/request.rs b/nodedb/src/bridge/envelope/request.rs index 791d1d887..cfc21210e 100644 --- a/nodedb/src/bridge/envelope/request.rs +++ b/nodedb/src/bridge/envelope/request.rs @@ -34,7 +34,9 @@ pub struct Request { /// Opaque plan digest identifying the physical operation to execute. pub plan: PhysicalPlan, - /// Absolute deadline. Data Plane MUST stop at next safe point after expiry. + /// Absolute deadline. The Data Plane reads it only through + /// [`Request::execution_deadline`], and stops at the next safe point after + /// that deadline passes. Already-ordered work has no execution deadline. pub deadline: Instant, /// Request priority for scheduling on the Data Plane. @@ -119,6 +121,25 @@ pub struct Request { pub admission: Admission, } +impl Request { + /// The deadline the Data Plane enforces while it runs this request. + /// + /// Returns `None` for [`ExemptReason::AlreadyOrdered`] work: Calvin + /// applies, replicated applies, replay, clone, and checkpoint. Their order + /// is already fixed and other replicas apply the same work. Such work has + /// no refusal outcome, so every replica must run it to completion. A + /// replica that drops it on a deadline diverges from the others. + /// + /// Returns `Some(self.deadline)` for every other request. Every Data-Plane + /// deadline check reads this method, never the raw field. + pub fn execution_deadline(&self) -> Option { + match self.admission { + Admission::Exempt(ExemptReason::AlreadyOrdered) => None, + Admission::Admitted | Admission::Exempt(ExemptReason::Read) => Some(self.deadline), + } + } +} + /// Write-admission marker carried by every [`Request`]. /// /// A write-class plan becomes [`Admission::Admitted`] only by passing the @@ -214,6 +235,26 @@ mod tests { assert_ne!(req.trace_id, TraceId::ZERO); } + #[test] + fn already_ordered_request_has_no_execution_deadline() { + let req = Request { + admission: Admission::Exempt(ExemptReason::AlreadyOrdered), + ..sample_request() + }; + assert_eq!(req.execution_deadline(), None); + } + + #[test] + fn admitted_and_read_requests_keep_their_deadline() { + for admission in [Admission::Admitted, Admission::Exempt(ExemptReason::Read)] { + let req = Request { + admission, + ..sample_request() + }; + assert_eq!(req.execution_deadline(), Some(req.deadline)); + } + } + #[test] fn cancel_plan() { let req = Request { diff --git a/nodedb/src/bridge/envelope/sync_hold.rs b/nodedb/src/bridge/envelope/sync_hold.rs new file mode 100644 index 000000000..7673f6007 --- /dev/null +++ b/nodedb/src/bridge/envelope/sync_hold.rs @@ -0,0 +1,42 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Why the sync idempotency gate held a frame back without applying it. + +use nodedb_types::sync::wire::AckStatus; + +/// A gate verdict that applies nothing and is not a refusal. +/// +/// The sender acts on each one differently, so each crosses the bridge by +/// name. None of them asks the sender to compensate. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum SyncHold { + /// The frame's sequence is at or below the stream's mark: it applied + /// under an earlier delivery. + Duplicate, + /// The frame's producer epoch is below the producer's floor. + Fenced, + /// The frame skipped sequences. `expected` is the next one the stream + /// admits. + Gap { expected: u64 }, +} + +impl SyncHold { + /// The ack status a sender receives for this hold. + pub fn ack_status(self) -> AckStatus { + match self { + Self::Duplicate => AckStatus::Duplicate, + Self::Fenced => AckStatus::Fenced, + Self::Gap { expected } => AckStatus::Gap { expected }, + } + } +} + +impl std::fmt::Display for SyncHold { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Duplicate => write!(f, "duplicate"), + Self::Fenced => write!(f, "fenced producer epoch"), + Self::Gap { expected } => write!(f, "sequence gap, expected {expected}"), + } + } +} diff --git a/nodedb/src/control/array_catalog/persist.rs b/nodedb/src/control/array_catalog/persist.rs index fe04357e0..4c5689505 100644 --- a/nodedb/src/control/array_catalog/persist.rs +++ b/nodedb/src/control/array_catalog/persist.rs @@ -140,9 +140,8 @@ mod tests { let array = entry("cells"); persist(&cat, &array).unwrap(); cat.put_surrogate( - array.array_id.database_id, + nodedb_types::CollectionKey::from_bare(array.array_id.database_id, &array.name), array.array_id.tenant_id, - &array.name, b"coord:1", nodedb_types::Surrogate::new(42), ) @@ -162,9 +161,8 @@ mod tests { ); assert_eq!( cat.get_surrogate_for_pk( - array.array_id.database_id, + nodedb_types::CollectionKey::from_bare(array.array_id.database_id, &array.name), array.array_id.tenant_id, - &array.name, b"coord:1" ) .unwrap(), @@ -183,9 +181,8 @@ mod tests { ); assert!( cat.get_surrogate_for_pk( - array.array_id.database_id, + nodedb_types::CollectionKey::from_bare(array.array_id.database_id, &array.name), array.array_id.tenant_id, - &array.name, b"coord:1" ) .unwrap() diff --git a/nodedb/src/control/array_sync/inbound_propose.rs b/nodedb/src/control/array_sync/inbound_propose.rs index 6c0bddcd4..1b741faac 100644 --- a/nodedb/src/control/array_sync/inbound_propose.rs +++ b/nodedb/src/control/array_sync/inbound_propose.rs @@ -74,7 +74,11 @@ impl OriginArrayInbound { // Use the async proposer (with transparent leader forwarding + apply // wait) when available. It returns the apply payload directly. if let Some(async_proposer) = self.shared().async_raft_proposer().map(|a| a.as_ref()) { - return match async_proposer(vshard_id, idempotency_key, data).await { + let deadline = tokio::time::Instant::now() + + std::time::Duration::from_secs( + self.shared().tuning.network.default_deadline_secs, + ); + return match async_proposer(vshard_id, idempotency_key, data, deadline).await { Ok((_payload, _committed_version)) => Ok(()), Err(e) => { warn!(array = %array, error = %e, "array_inbound: raft propose+apply failed"); diff --git a/nodedb/src/control/array_sync/raft_apply/cell.rs b/nodedb/src/control/array_sync/raft_apply/cell.rs index 83040de39..cb9e0f1a4 100644 --- a/nodedb/src/control/array_sync/raft_apply/cell.rs +++ b/nodedb/src/control/array_sync/raft_apply/cell.rs @@ -63,10 +63,12 @@ pub(crate) async fn apply_array_cell_write( target: ArrayCellTarget, plan: PhysicalPlan, ) -> bool { + let commit_hlc = pos.carried_commit_hlc(); let AppliedPosition { group_id, log_index, applied_key, + .. } = pos; let ArrayCellTarget { tenant_id, @@ -117,6 +119,8 @@ pub(crate) async fn apply_array_cell_write( // exactly as the generic committed-write branch does. event_source: crate::event::EventSource::User, resolved_now_ms, + apply_key: applied_key, + commit_hlc, op_label: "array cell write", }, ) @@ -128,7 +132,11 @@ pub(crate) async fn apply_array_cell_write( "apply_array_cell_write: apply failed" ); } - let applied_ok = result.is_ok(); + // A final refusal is the entry's outcome: its marker carries the key. + let applied_ok = result.is_ok() + || result + .as_ref() + .is_err_and(crate::control::server::dispatch_utils::error_is_final_refusal); tracker.complete(group_id, log_index, applied_key, result); applied_ok } diff --git a/nodedb/src/control/array_sync/raft_apply/common.rs b/nodedb/src/control/array_sync/raft_apply/common.rs index 5580ffc77..1b945ab24 100644 --- a/nodedb/src/control/array_sync/raft_apply/common.rs +++ b/nodedb/src/control/array_sync/raft_apply/common.rs @@ -17,19 +17,30 @@ use crate::control::server::dispatch_utils::{ ChangeFeedOwner, SubmitWrite, WalDurability, WriteOrdering, submit_write, }; use crate::control::state::SharedState; -use crate::types::{DatabaseId, ReadConsistency, TenantId, TraceId, VShardId}; +use crate::types::{DatabaseId, ReadConsistency, RequestId, TenantId, TraceId, VShardId}; /// Identifies a committed Raft entry within the apply loop. /// -/// Groups the three fields that always travel together: the Raft group, the -/// log index within that group, and the idempotency key extracted from the -/// `ReplicatedEntry` header. All three are forwarded together to -/// `ProposeTracker::complete` after each apply. +/// Groups the fields that always travel together: the Raft group, the log +/// index within that group, and the idempotency key extracted from the +/// `ReplicatedEntry` header, all forwarded to `ProposeTracker::complete` after +/// each apply, plus the entry's commit HLC. #[derive(Debug, Clone, Copy)] pub(crate) struct AppliedPosition { pub group_id: u64, pub log_index: u64, pub applied_key: u64, + /// HLC wall time, in nanoseconds, the proposer stamped on the entry. `0` + /// when the entry carries none. It is the write's commit instant on the + /// tenant's observed write high-water, however late this replica applies. + pub commit_hlc: u64, +} + +impl AppliedPosition { + /// The entry's commit HLC for the write funnel, `None` when it carries none. + pub(crate) fn carried_commit_hlc(&self) -> Option { + (self.commit_hlc != 0).then_some(self.commit_hlc) + } } /// One committed array write, ready for the Control-Plane write funnel. @@ -43,6 +54,11 @@ pub(super) struct ArrayWriteSubmit { /// the entry carries none. Passed through as the redo record's /// `now_override` so this replica records the value its peers recorded. pub resolved_now_ms: Option, + /// The idempotency key of the committed entry, carried by the redo + /// record's header. + pub apply_key: u64, + /// The committed entry's commit HLC, `None` when it carries none. + pub commit_hlc: Option, /// Contextual label for the error surfaced to the propose waiter. pub op_label: &'static str, } @@ -58,7 +74,9 @@ pub(super) struct ArrayWriteSubmit { /// /// An error-status response is surfaced as a typed error: a committed entry that /// failed to apply must reach the propose waiter as a failure, not an empty -/// success, and must NOT advance the floor. +/// success, and must NOT advance the floor. A coded refusal is +/// `crate::Error::DataPlane`, so the waiter classifies it and a final refusal +/// records its marker. The funnel's own errors keep their variant. pub(super) async fn submit_array_write( state: &Arc, params: ArrayWriteSubmit, @@ -70,6 +88,8 @@ pub(super) async fn submit_array_write( plan, event_source, resolved_now_ms, + apply_key, + commit_hlc, op_label, } = params; @@ -92,6 +112,8 @@ pub(super) async fn submit_array_write( // durability path than this record's replay. durability: WalDurability::AppendHere { now_override: resolved_now_ms, + apply_key, + commit_hlc, }, // Raft committed this entry at a fixed log index and every replica // applies it in that order; re-entering the write-admission gate @@ -105,19 +127,11 @@ pub(super) async fn submit_array_write( change_feed: ChangeFeedOwner::Unowned, }, ) - .await - .map_err(|e| crate::Error::Internal { - detail: format!("{op_label}: {e}"), - })?; + .await?; let response = outcome.response; if response.status != Status::Ok { - let detail = response - .error_code - .as_ref() - .map(|c| format!("{op_label} error: {c:?}")) - .unwrap_or_else(|| format!("{op_label} returned error status")); - return Err(crate::Error::Internal { detail }); + return Err(apply_refusal(op_label, &response)); } // The response carries the write-version this replica stamped alongside the // payload; an array plan names no user collection, so it is `Lsn::ZERO` here @@ -221,15 +235,16 @@ pub(super) async fn ensure_array_open( Err(poisoned) => poisoned.into_inner().dispatch(open_request), }; - if let Err(e) = dispatch_result { - return Err(crate::Error::Internal { - detail: format!("ensure_array_open: dispatch failed: {e}"), - }); - } + // A dispatch refusal, such as a capacity limit, keeps its own class. + dispatch_result?; - await_data_plane(async move { open_rx.recv().await.ok_or(()) }, "OpenArray") - .await - .map(|_| ()) + await_data_plane( + async move { open_rx.recv().await.ok_or(()) }, + open_request_id, + "OpenArray", + ) + .await + .map(|_| ()) } /// Build a `Request` for an array apply/open with default deadline / priority. @@ -268,27 +283,124 @@ pub(super) fn build_array_request( } } -/// Await a Data Plane response, mapping timeout / channel-closed / error-status -/// into `crate::Error::Internal` with a contextual `op_label`. +/// How long [`await_data_plane`] waits for the Data Plane's response. +const DATA_PLANE_AWAIT_TIMEOUT: Duration = Duration::from_secs(30); + +/// Await the Data Plane response to request `request_id`. An error status +/// becomes [`apply_refusal`]. A timeout is `crate::Error::DeadlineExceeded` +/// (`57014`). A closed channel is `crate::Error::Internal` with `op_label`. pub(super) async fn await_data_plane( rx: impl std::future::Future>, + request_id: RequestId, op_label: &str, ) -> ProposeResult { - match tokio::time::timeout(Duration::from_secs(30), rx).await { + match tokio::time::timeout(DATA_PLANE_AWAIT_TIMEOUT, rx).await { Ok(Ok(resp)) if resp.status == Status::Ok => Ok(AppliedWrite::from_response(&resp)), - Ok(Ok(resp)) => { - let detail = resp - .error_code - .as_ref() - .map(|c| format!("{op_label} error: {c:?}")) - .unwrap_or_else(|| format!("{op_label} returned error status")); - Err(crate::Error::Internal { detail }) - } + Ok(Ok(resp)) => Err(apply_refusal(op_label, &resp)), Ok(Err(_)) => Err(crate::Error::Internal { detail: format!("{op_label}: response channel closed"), }), - Err(_) => Err(crate::Error::Internal { - detail: format!("{op_label}: deadline exceeded"), - }), + Err(_) => Err(crate::Error::DeadlineExceeded { request_id }), + } +} + +/// The typed error for a Data-Plane response with a non-`Ok` status. +/// +/// A coded refusal is `crate::Error::DataPlane` with its own code. Only a +/// refusal with no code is `crate::Error::Internal`. +pub(super) fn apply_refusal(op_label: &str, response: &Response) -> crate::Error { + match response.error_code.as_deref() { + Some(code) => crate::Error::DataPlane(code.clone()), + None => crate::Error::Internal { + detail: format!("{op_label} returned error status"), + }, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::{ErrorCode, Payload}; + use crate::types::Lsn; + + fn refusal(code: Option) -> Response { + Response { + request_id: RequestId::new(1), + status: Status::Error, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: code.map(Box::new), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + /// A coded refusal of a committed array write keeps its code, so the + /// final-refusal check sees it. + #[test] + fn a_coded_refusal_keeps_its_code() { + let code = ErrorCode::RejectedPrevalidation { + reason: "cell out of bounds".into(), + }; + let error = apply_refusal("array cell write", &refusal(Some(code.clone()))); + match &error { + crate::Error::DataPlane(kept) => assert_eq!(kept, &code), + other => panic!("expected the typed refusal, got {other:?}"), + } + assert!(crate::control::server::dispatch_utils::error_is_final_refusal(&error)); + } + + #[test] + fn a_refusal_with_no_code_is_internal() { + match apply_refusal("OpenArray", &refusal(None)) { + crate::Error::Internal { detail } => assert!(detail.starts_with("OpenArray")), + other => panic!("expected an internal error, got {other:?}"), + } + } + + /// The Data-Plane response await keeps the code as well. + #[tokio::test] + async fn awaiting_a_coded_refusal_keeps_its_code() { + let code = ErrorCode::Unsupported { + detail: "not on this engine".into(), + }; + let response = refusal(Some(code.clone())); + let result = await_data_plane( + async move { Ok::<_, ()>(response) }, + RequestId::new(1), + "OpenArray", + ) + .await; + match result { + Err(crate::Error::DataPlane(kept)) => assert_eq!(kept, code), + other => panic!("expected the typed refusal, got {other:?}"), + } + } + + /// A local timeout is the typed deadline error, `57014` on pgwire. + #[tokio::test(start_paused = true)] + async fn a_local_timeout_is_a_typed_deadline() { + let result = await_data_plane( + std::future::pending::>(), + RequestId::new(7), + "OpenArray", + ) + .await; + let error = match result { + Err(error) => error, + Ok(_) => panic!("a pending response must time out"), + }; + assert!( + matches!( + &error, + crate::Error::DeadlineExceeded { request_id } if *request_id == RequestId::new(7) + ), + "expected a typed deadline, got {error:?}" + ); + let (_, state, _) = crate::control::server::pgwire::types::error_to_sqlstate(&error); + assert_eq!(state, nodedb_types::error::sqlstate::QUERY_CANCELED.0); } } diff --git a/nodedb/src/control/array_sync/raft_apply/op.rs b/nodedb/src/control/array_sync/raft_apply/op.rs index 2dd480b6f..500f0380c 100644 --- a/nodedb/src/control/array_sync/raft_apply/op.rs +++ b/nodedb/src/control/array_sync/raft_apply/op.rs @@ -35,10 +35,12 @@ pub(crate) async fn apply_array_op( database_id, array, } = target; + let commit_hlc = pos.carried_commit_hlc(); let AppliedPosition { group_id, log_index, applied_key, + .. } = pos; use nodedb_array::sync::op_codec; @@ -201,6 +203,8 @@ pub(crate) async fn apply_array_op( // A sync op carries no proposer-resolved instant; only TTL-bearing // KV writes resolve one, and no array op is such a write. resolved_now_ms: None, + apply_key: applied_key, + commit_hlc, op_label: "array op", }, ) @@ -225,8 +229,12 @@ pub(crate) async fn apply_array_op( group_id, index = log_index, array = %op.header.array, error = %e, "apply_array_op: apply failed" ); + // A final refusal is the entry's outcome: its marker carries the + // key, so a redelivered copy is never applied. + let refused_finally = + crate::control::server::dispatch_utils::error_is_final_refusal(&e); tracker.complete(group_id, log_index, applied_key, Err(e)); - false + refused_finally } } } diff --git a/nodedb/src/control/array_sync/raft_apply/schema.rs b/nodedb/src/control/array_sync/raft_apply/schema.rs index 615a1bc1a..f5db9f3f8 100644 --- a/nodedb/src/control/array_sync/raft_apply/schema.rs +++ b/nodedb/src/control/array_sync/raft_apply/schema.rs @@ -45,6 +45,7 @@ pub(crate) fn apply_array_schema( group_id, log_index, applied_key, + .. } = pos; use nodedb_array::sync::hlc::Hlc; diff --git a/nodedb/src/control/backup/cut.rs b/nodedb/src/control/backup/cut.rs new file mode 100644 index 000000000..897ae8863 --- /dev/null +++ b/nodedb/src/control/backup/cut.rs @@ -0,0 +1,254 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A backup's consistent cut. +//! +//! The cut picks the envelope watermark `W`, then waits until every user +//! write committed below `W` has a final outcome on this node, and only then +//! lets the backup snapshot. Afterwards every write is one of two kinds: +//! +//! - committed below `W`: applied before the snapshot, so the backup holds it; +//! - committed at or above `W`: its mark is above `W`, so a restore of this +//! backup refuses it. +//! +//! Three waits make the cut: +//! +//! - **Raft data groups.** A replicated write carries its proposer's commit +//! stamp and applies in log order. The cut proposes a +//! [`ReplicatedWrite::CutBarrier`] carrying `W` into every data group this +//! node hosts and waits for this node's apply of it. Every entry before the +//! barrier applied first. Every entry after it records a commit HLC above +//! `W`, however early its proposer stamped it. +//! - **Calvin transactions.** A Calvin transaction commits at its place in +//! the sequencer log. The cut proposes a `CutMarker` carrying `W` into the +//! sequencer log and waits until every Calvin scheduler on this node passed +//! it: every transaction delivered before the marker installed or dropped. +//! Every transaction delivered after it records a commit HLC above `W`. +//! - **Local write windows.** A write the funnel appends here stamps itself +//! after its record is minted inside an outcome-floor window. The cut reads +//! the highest LSN any window minted, after it picks `W`, and waits for the +//! outcome floor to reach it. A record minted later stamps itself above `W`. +//! On a server with no Raft groups a write stamps itself before it mints, so +//! its mark is durable first. The cut first waits for every such stamp at or +//! below `W` to mint, then reads the highest LSN. +//! +//! Two stamps can share a wall time. Once it picks `W`, the cut moves this +//! node's clock past `W`, so every later stamp here reads above it. +//! +//! The waits bind this node's replicas. A remote source node takes the same +//! cut at `W` itself before it snapshots: the snapshot plan sent to it carries +//! `W`, and its Control Plane runs [`cut_at`] first. + +use std::collections::BTreeMap; +use std::sync::Arc; +use std::time::Duration; + +use nodedb_cluster::METADATA_GROUP_ID; +use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use nodedb_cluster::calvin::SequencerEntry; +use nodedb_types::Hlc; + +use crate::Error; +use crate::control::security::auth_fence::cluster::{group_of_vshard, hosts_group, routed_groups}; +use crate::control::state::SharedState; +use crate::control::wal_replication::{ReplicatedEntry, ReplicatedWrite, propose_replicated_entry}; +use crate::types::{DatabaseId, VShardId}; + +/// Pick the envelope watermark and wait until every write committed below it +/// has a final outcome on this node. Returns the watermark. +pub(super) async fn consistent_cut(state: &Arc, tenant_id: u64) -> Result { + let watermark = state.hlc_clock.now().wall_ns; + cut_at(state, tenant_id, watermark).await?; + Ok(watermark) +} + +/// Take the consistent cut at `watermark` on this node: wait until every +/// write committed below it has a final outcome here. A remote source node +/// runs it for the watermark the backup's coordinator picked. +pub(crate) async fn cut_at( + state: &Arc, + tenant_id: u64, + watermark: u64, +) -> Result<(), Error> { + state + .hlc_clock + .update(Hlc::new(watermark.saturating_add(1), 0)); + let timeout = Duration::from_secs(state.tuning.network.default_deadline_secs); + let deadline = tokio::time::Instant::now() + timeout; + // A local write stamped at or below the watermark has not always minted + // its record yet. Every stamp taken from here on reads above it. + if !state + .tenant_marks + .await_local_stamps_minted(watermark, deadline) + .await + { + return Err(Error::Internal { + detail: format!( + "backup: local writes stamped at or below watermark {watermark} did not mint \ + their WAL records within {}s, so the backup cannot take a consistent cut. \ + Retry the backup", + timeout.as_secs(), + ), + }); + } + // Read after the watermark: a record minted after this read stamps its + // write above the watermark. + let target = state.outcome_floor.max_noted(); + + let (groups, calvin) = tokio::join!( + cut_data_groups(state, tenant_id, watermark), + cut_calvin(state, watermark, deadline), + ); + groups?; + calvin?; + + if !state.outcome_floor.await_floor(target, deadline).await { + return Err(Error::Internal { + detail: format!( + "backup: writes minted at or below WAL LSN {} had no final outcome within \ + {}s, so the backup cannot take a consistent cut (outcome floor at {}). \ + Retry the backup. A write window held until restart keeps the floor \ + below it: restart the node if the floor does not move", + target.as_u64(), + timeout.as_secs(), + state.outcome_floor.floor().as_u64(), + ), + }); + } + Ok(()) +} + +/// Propose a cut barrier carrying `watermark` into every data group this node +/// hosts, and wait for this node's apply of each. +async fn cut_data_groups( + state: &Arc, + tenant_id: u64, + watermark: u64, +) -> Result<(), Error> { + let Some(proposer) = state.async_raft_proposer() else { + return Ok(()); + }; + let barriers = futures::future::join_all(barrier_vshards(state).into_iter().map( + |(group_id, vshard_id)| { + // The barrier orders every entry of its group, whatever database + // the entry writes, so one barrier per group cuts every database. + // The entry's database id only frames it. + let entry = ReplicatedEntry::new( + tenant_id, + DatabaseId::DEFAULT.as_u64(), + vshard_id, + ReplicatedWrite::CutBarrier { hlc: watermark }, + ); + async move { + propose_replicated_entry(state, proposer, entry) + .await + .map_err(|error| (group_id, error)) + } + }, + )) + .await; + for barrier in barriers { + if let Err((group_id, error)) = barrier { + // A group this node left while the barrier waited holds no + // replica here to cut: the source node that snapshots it takes + // its own cut. + if !hosts_group(state, group_id) { + tracing::info!( + group_id, + %error, + "backup: this node left the group before its cut barrier applied here; \ + the group needs no cut on this node" + ); + continue; + } + return Err(Error::Internal { + detail: format!( + "backup: the consistent-cut barrier of raft group {group_id} did not \ + apply on this node: {error}. Retry the backup" + ), + }); + } + } + Ok(()) +} + +/// How long the cut waits for its Calvin marker before it proposes the +/// marker again. A leader change can drop a proposed marker; a second copy is +/// harmless, since a scheduler passes each marker once its earlier +/// transactions finished. +const CUT_MARKER_RETRY: Duration = Duration::from_secs(1); + +/// Propose a Calvin cut marker carrying `watermark`, and wait until every +/// Calvin scheduler on this node passed it, or `deadline`. +pub(crate) async fn cut_calvin( + state: &Arc, + watermark: u64, + deadline: tokio::time::Instant, +) -> Result<(), Error> { + let cuts = &state.calvin.cuts; + if cuts.is_empty() { + // No Calvin scheduler runs here: this node applies no Calvin write. + return Ok(()); + } + let proposer = state + .calvin + .sequencer_proposer + .get() + .ok_or_else(|| Error::Internal { + detail: "backup: Calvin schedulers run on this node, but no sequencer proposer \ + is set, so the consistent cut cannot place its marker. Retry the backup \ + once the cluster finished starting" + .into(), + })?; + let marker = zerompk::to_msgpack_vec(&SequencerEntry::CutMarker { hlc: watermark }).map_err( + |error| Error::Internal { + detail: format!("backup: encode the Calvin cut marker: {error}"), + }, + )?; + let mut last_refusal = None; + loop { + if let Err(error) = proposer.propose(marker.clone()) { + last_refusal = Some(error.to_string()); + } + let attempt_deadline = deadline.min(tokio::time::Instant::now() + CUT_MARKER_RETRY); + let lagging = cuts.await_passed(watermark, attempt_deadline).await; + if lagging.is_empty() { + return Ok(()); + } + if tokio::time::Instant::now() >= deadline { + return Err(Error::Internal { + detail: format!( + "backup: the Calvin schedulers of vShards {lagging:?} did not pass the \ + consistent-cut marker in time, so Calvin transactions sequenced before \ + the backup may not be installed (last marker refusal: {}). Retry the \ + backup", + last_refusal.as_deref().unwrap_or("none") + ), + }); + } + } +} + +/// One vShard per data group this node hosts: the barrier of a group is +/// proposed through any vShard the group homes. +fn barrier_vshards(state: &SharedState) -> BTreeMap { + let hosted: Vec = routed_groups(state) + .into_iter() + .filter(|group_id| { + *group_id != METADATA_GROUP_ID + && *group_id != SEQUENCER_GROUP_ID + && hosts_group(state, *group_id) + }) + .collect(); + let mut vshards = BTreeMap::new(); + for vshard_id in 0..VShardId::COUNT { + if vshards.len() == hosted.len() { + break; + } + if let Ok(group_id) = group_of_vshard(state, vshard_id) + && hosted.contains(&group_id) + { + vshards.entry(group_id).or_insert(vshard_id); + } + } + vshards +} diff --git a/nodedb/src/control/backup/metadata.rs b/nodedb/src/control/backup/metadata.rs new file mode 100644 index 000000000..1d821085f --- /dev/null +++ b/nodedb/src/control/backup/metadata.rs @@ -0,0 +1,199 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The tenant's databases and the metadata sections of its backup. +//! +//! A tenant's collections can live in any database. The backup covers every +//! database the tenant has a collection in. The metadata sections record, per +//! database: +//! +//! - the database itself: its descriptor, its quota and the tenant's quota in +//! it (`SECTION_ORIGIN_DATABASES`); +//! - the catalog row of each of the tenant's collections in it +//! (`SECTION_ORIGIN_CATALOG_ROWS`); +//! - the PK-to-surrogate binds of those collections +//! (`SECTION_ORIGIN_SURROGATE_PK`); +//! - the WAL tombstones of the tenant's purged collections in it +//! (`SECTION_ORIGIN_SOURCE_TOMBSTONES`). +//! +//! Every entry names its database by the source id. Restore maps each source +//! id to a destination database by name. + +use std::collections::{BTreeMap, BTreeSet}; + +use nodedb_types::backup_envelope::{ + DatabaseBlob, EnvelopeWriter, SECTION_ORIGIN_CATALOG_ROWS, SECTION_ORIGIN_DATABASES, + SECTION_ORIGIN_SOURCE_TOMBSTONES, SECTION_ORIGIN_SURROGATE_PK, SourceTombstoneEntry, + StoredCollectionBlob, SurrogateBindBlob, +}; + +use crate::Error; +use crate::control::security::catalog::{DatabaseDescriptor, StoredCollection, SystemCatalog}; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId}; + +/// One database the tenant has collections in. +pub struct TenantDatabase { + pub descriptor: DatabaseDescriptor, + /// Every collection of the tenant in this database, soft-deleted ones + /// included: UNDROP works after a restore of a backup taken during the + /// retention window. + pub collections: Vec, +} + +impl TenantDatabase { + pub fn id(&self) -> DatabaseId { + self.descriptor.id + } +} + +/// Every database `tenant_id` has a collection in, in database-id order. +/// +/// A collection whose database has no catalog entry fails the backup: the +/// restore could not recreate that database, and its rows would be lost. +pub fn tenant_databases(state: &SharedState, tenant_id: u64) -> Result, Error> { + let catalog = state.credentials.catalog(); + let mut by_database: BTreeMap> = BTreeMap::new(); + for coll in catalog.load_all_collections_across_databases()? { + if coll.tenant_id == tenant_id { + by_database + .entry(coll.database_id.as_u64()) + .or_default() + .push(coll); + } + } + let mut databases = Vec::with_capacity(by_database.len()); + for (raw_id, collections) in by_database { + let descriptor = catalog + .get_database(DatabaseId::new(raw_id))? + .ok_or_else(|| Error::Internal { + detail: format!( + "backup: tenant {tenant_id} has collections in database {raw_id}, \ + but the catalog has no entry for that database. Restore the \ + database entry, then retry the backup" + ), + })?; + databases.push(TenantDatabase { + descriptor, + collections, + }); + } + Ok(databases) +} + +/// Push the metadata sections of `databases`. A catalog read or encode error +/// fails the backup: an envelope without these sections restores rows into +/// no database, rows a point lookup cannot find, or a purged collection. +pub fn push_metadata_sections( + state: &SharedState, + tenant_id: u64, + databases: &[TenantDatabase], + writer: &mut EnvelopeWriter, +) -> Result<(), Error> { + let catalog = state.credentials.catalog(); + let tenant = TenantId::new(tenant_id); + + let mut blobs = Vec::with_capacity(databases.len()); + for database in databases { + blobs.push(DatabaseBlob { + database_id: database.id().as_u64(), + name: database.descriptor.name.clone(), + descriptor: encode_section_part("database descriptor", &database.descriptor)?, + database_quota: catalog.get_database_quota(database.id())?, + tenant_quota: catalog.get_tenant_quota(database.id(), tenant)?, + }); + } + push_nonempty(writer, SECTION_ORIGIN_DATABASES, "databases", &blobs)?; + + let mut rows = Vec::new(); + for database in databases { + for coll in &database.collections { + rows.push(StoredCollectionBlob { + database_id: database.id().as_u64(), + name: coll.name.clone(), + bytes: encode_section_part("catalog row", coll)?, + }); + } + } + push_nonempty(writer, SECTION_ORIGIN_CATALOG_ROWS, "catalog rows", &rows)?; + + // PK→surrogate identity map. This is DATA-derived per-node state that the + // per-node engine sections do NOT carry (the Data-Plane snapshot handler + // has no catalog access). Without it a restored node has documents but + // cannot resolve PK point-lookups (`WHERE id=`). + let binds = surrogate_binds(catalog, tenant, databases)?; + push_nonempty(writer, SECTION_ORIGIN_SURROGATE_PK, "surrogate pk", &binds)?; + + let backed_up: BTreeSet = databases.iter().map(|d| d.id().as_u64()).collect(); + let mut tombs = Vec::new(); + for (database_id, tid, name, purge_lsn) in catalog.load_wal_tombstones()?.iter() { + if tid == tenant_id && backed_up.contains(&database_id) { + tombs.push(SourceTombstoneEntry { + database_id, + collection: name.to_string(), + purge_lsn, + }); + } + } + push_nonempty( + writer, + SECTION_ORIGIN_SOURCE_TOMBSTONES, + "source tombstones", + &tombs, + ) +} + +/// Every PK→surrogate bind of the tenant's collections in `databases`. +fn surrogate_binds( + catalog: &SystemCatalog, + tenant: TenantId, + databases: &[TenantDatabase], +) -> Result, Error> { + let mut binds = Vec::new(); + for database in databases { + for coll in &database.collections { + let rows = catalog.scan_surrogates_for_collection( + nodedb_types::CollectionKey::from_bare(database.id(), &coll.name), + tenant, + )?; + for (pk, surrogate) in rows { + binds.push(SurrogateBindBlob { + database_id: database.id().as_u64(), + tenant_id: tenant.as_u64(), + collection: coll.name.clone(), + pk, + surrogate: surrogate.as_u32(), + }); + } + } + } + Ok(binds) +} + +/// Encode one part of a section. +pub(super) fn encode_section_part( + what: &str, + value: &T, +) -> Result, Error> { + zerompk::to_msgpack_vec(value).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("backup envelope ({what}): encode: {e}"), + }) +} + +/// Encode `entries` and push them as the section `origin`, unless empty. +fn push_nonempty( + writer: &mut EnvelopeWriter, + origin: u64, + what: &str, + entries: &[T], +) -> Result<(), Error> { + if entries.is_empty() { + return Ok(()); + } + let body = encode_section_part(what, &entries)?; + writer + .push_section(origin, body) + .map_err(|e| Error::Internal { + detail: format!("backup envelope ({what}): {e}"), + }) +} diff --git a/nodedb/src/control/backup/mod.rs b/nodedb/src/control/backup/mod.rs index f12b96bca..f0c34fd1e 100644 --- a/nodedb/src/control/backup/mod.rs +++ b/nodedb/src/control/backup/mod.rs @@ -1,6 +1,9 @@ // SPDX-License-Identifier: BUSL-1.1 +pub mod cut; pub mod detect; +pub mod metadata; +pub mod node_snapshot; pub mod orchestrator; pub mod restore; pub mod snapshot_keys; diff --git a/nodedb/src/control/backup/node_snapshot.rs b/nodedb/src/control/backup/node_snapshot.rs new file mode 100644 index 000000000..e1c3ea647 --- /dev/null +++ b/nodedb/src/control/backup/node_snapshot.rs @@ -0,0 +1,144 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Take one database's tenant snapshot on one source node. +//! +//! This node snapshots over the local SPSC bridge. A remote node receives the +//! plan in a `RaftRpc::ExecuteRequest`. Either way the request names the +//! database, and the Data Plane snapshot covers that database only. + +use std::sync::Arc; +use std::time::Duration; + +use nodedb_cluster::rpc_codec::{ExecuteRequest, ExecuteResponse, RaftRpc, TypedClusterError}; + +use crate::Error; +use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::exchange::snapshot_tenant_on_local_cores; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId, TraceId}; +use nodedb_physical::physical_plan::wire as plan_wire; + +/// Default per-node snapshot dispatch timeout. +const NODE_SNAPSHOT_TIMEOUT: Duration = Duration::from_secs(120); + +/// Whether `node_id` names this node. +pub(super) fn is_self(state: &SharedState, node_id: u64) -> bool { + node_id == state.node_id || node_id == 0 || state.cluster_transport.is_none() +} + +/// Snapshot `tenant_id` in `database_id` on every core of this node. +/// +/// This node already took the backup's cut, so the local snapshot carries no +/// cut request. +pub(super) async fn snapshot_self( + state: &Arc, + tenant_id: u64, + database_id: DatabaseId, +) -> Result, Error> { + snapshot_tenant_on_local_cores( + state, + TenantId::new(tenant_id), + database_id, + NODE_SNAPSHOT_TIMEOUT, + ) + .await +} + +/// Snapshot `tenant_id` in `database_id` on the remote node `node_id`. +pub(super) async fn snapshot_remote( + state: &Arc, + node_id: u64, + tenant_id: u64, + database_id: DatabaseId, + plan: &PhysicalPlan, +) -> Result, Error> { + let transport = state + .cluster_transport + .as_ref() + .ok_or_else(|| Error::Internal { + detail: format!("backup: cluster_transport unavailable but node {node_id} is remote"), + })?; + + let plan_bytes = plan_wire::encode(plan).map_err(|e| Error::Internal { + detail: format!("backup: plan encode failed: {e}"), + })?; + let req = RaftRpc::ExecuteRequest(ExecuteRequest { + plan_bytes, + tenant_id, + database_id: database_id.as_u64(), + deadline_remaining_ms: NODE_SNAPSHOT_TIMEOUT.as_millis() as u64, + trace_id: TraceId::generate().0, + descriptor_versions: Vec::new(), + // Backup snapshot dispatch is not session-transaction-scoped. + txn_id: None, + }); + + let resp = transport + .send_rpc(node_id, req) + .await + .map_err(|e| Error::Internal { + detail: format!("backup: snapshot RPC to node {node_id} failed: {e}"), + })?; + match resp { + RaftRpc::ExecuteResponse(ExecuteResponse { + success: true, + mut payloads, + .. + }) => { + // CreateTenantSnapshot returns exactly one payload. + if payloads.len() != 1 { + return Err(Error::Internal { + detail: format!( + "backup: expected 1 payload from node {node_id}, got {}", + payloads.len() + ), + }); + } + Ok(payloads.remove(0)) + } + RaftRpc::ExecuteResponse(ExecuteResponse { + error: Some(err), .. + }) => Err(map_typed_error(err, node_id)), + RaftRpc::ExecuteResponse(_) => Err(Error::Internal { + detail: format!("backup: empty error response from node {node_id}"), + }), + other => Err(Error::Internal { + detail: format!( + "backup: unexpected RPC response variant from node {node_id}: {other:?}" + ), + }), + } +} + +fn map_typed_error(err: TypedClusterError, node_id: u64) -> Error { + match err { + TypedClusterError::Internal { message, .. } => Error::Internal { + detail: format!("backup node {node_id}: {message}"), + }, + TypedClusterError::DeadlineExceeded { elapsed_ms } => Error::Internal { + detail: format!("backup node {node_id}: deadline exceeded after {elapsed_ms}ms"), + }, + TypedClusterError::NotLeader { .. } => Error::Internal { + detail: format!("backup node {node_id}: snapshot RPC routed to non-leader"), + }, + TypedClusterError::DescriptorMismatch { collection, .. } => Error::Internal { + detail: format!( + "backup node {node_id}: descriptor mismatch on collection {collection}" + ), + }, + // Keep the shard's verdict typed: a backup snapshot refused by the + // Data Plane must not read as a generic internal backup fault. + TypedClusterError::DataPlane { code } => Error::DataPlane(code.into()), + // A constraint verdict keeps its collection and kind, so the client + // reads the SQLSTATE the refusing shard meant. + TypedClusterError::RejectedConstraint { + collection, + constraint, + detail, + } => Error::RejectedConstraint { + collection, + constraint, + detail, + }, + } +} diff --git a/nodedb/src/control/backup/orchestrator.rs b/nodedb/src/control/backup/orchestrator.rs index 74535cefa..64962f836 100644 --- a/nodedb/src/control/backup/orchestrator.rs +++ b/nodedb/src/control/backup/orchestrator.rs @@ -7,33 +7,34 @@ //! (local SPSC for self, `RaftRpc::ExecuteRequest` for remotes), //! and packs the gathered per-node snapshots into a `BackupEnvelope`. //! +//! The backup covers every database the tenant has a collection in. Each +//! source node snapshots each of those databases after one consistent cut, +//! and each snapshot becomes one data section that names its database. +//! //! Single-node mode is the degenerate case: routing table absent -//! (or 1 node) → 1 section, origin = self. +//! (or 1 node) → one source, origin = self. use std::collections::{BTreeMap, HashSet}; use std::sync::Arc; -use std::time::Duration; use bytes::Bytes; use nodedb_cluster::routing::VSHARD_COUNT; -use nodedb_cluster::rpc_codec::{ExecuteRequest, ExecuteResponse, RaftRpc, TypedClusterError}; -use nodedb_types::backup_envelope::{EnvelopeMeta, EnvelopeWriter}; +use nodedb_types::backup_envelope::{DatabaseDataSection, EnvelopeMeta, EnvelopeWriter}; use crate::Error; use crate::bridge::envelope::PhysicalPlan; -use crate::control::server::shared::ddl::sync_dispatch; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TenantId, TraceId}; -use nodedb_physical::physical_plan::{MetaOp, wire as plan_wire}; +use crate::types::DatabaseId; +use nodedb_physical::physical_plan::MetaOp; -/// Default per-node snapshot dispatch timeout. -const NODE_SNAPSHOT_TIMEOUT: Duration = Duration::from_secs(120); +use super::metadata::{TenantDatabase, encode_section_part}; +use super::node_snapshot::{is_self, snapshot_remote, snapshot_self}; /// Build a complete tenant backup envelope by fanning out across the /// cluster, gathering each node's slice, and framing the result. /// /// Single-node and cluster paths converge here — a single-node server -/// produces a one-section envelope with origin = self. +/// produces one data section per database with origin = self. pub async fn backup_tenant(state: &Arc, tenant_id: u64) -> Result { // Assign every vshard to exactly ONE source node (the leader of its Raft // group, or — when no leader is elected yet — the lowest-id member), and @@ -43,36 +44,37 @@ pub async fn backup_tenant(state: &Arc, tenant_id: u64) -> Result1: keep only the - // vshards this node is the assigned source for; the other replicas' - // copies are dropped here so the restore merge sums disjoint sections. - let body = filter_node_snapshot(body, tenant_id, &source_vshards)?; - sections.push((node_id, body)); - } + // The envelope watermark is the consistent cut: every user write + // committed below it has applied before the snapshots below, and every + // write committed at or above it refuses a restore of this envelope. + let snapshot_watermark = super::cut::consistent_cut(state, tenant_id).await?; + + // Every database the tenant has a collection in. Read after the cut, so + // a collection created before the cut is in the list. + let databases = super::metadata::tenant_databases(state, tenant_id)?; + + // Each source node snapshots its databases in order. The nodes run + // concurrently. + let per_node = + futures::future::join_all(assignment.into_iter().map(|(node_id, source_vshards)| { + let databases = &databases; + async move { + snapshot_node( + state, + NodeSnapshot { + node_id, + tenant_id, + snapshot_watermark, + source_vshards: &source_vshards, + databases, + }, + ) + .await + } + })) + .await; - // Capture a cluster-wide logical instant for the envelope via the - // HLC. `hlc_clock.now()` advances past any previously observed - // local or remote HLC — the wall-ns component is the scalar - // watermark we stamp into the header. Restore compares this - // against the destination's `tenant_write_hlc` to detect stale - // envelopes. - let snapshot_watermark = state.hlc_clock.now().wall_ns; let meta = EnvelopeMeta { tenant_id, source_vshard_count: VSHARD_COUNT as u16, @@ -81,110 +83,25 @@ pub async fn backup_tenant(state: &Arc, tenant_id: u64) -> Result = Vec::new(); - for coll in all.iter().filter(|c| c.tenant_id == tenant_id) { - if let Ok(bytes) = zerompk::to_msgpack_vec(coll) { - blobs.push(nodedb_types::backup_envelope::StoredCollectionBlob { - name: coll.name.clone(), - bytes, - }); - } - } - if !blobs.is_empty() - && let Ok(body) = zerompk::to_msgpack_vec(&blobs) - { - writer - .push_section( - nodedb_types::backup_envelope::SECTION_ORIGIN_CATALOG_ROWS, - body, - ) - .map_err(|e| Error::Internal { - detail: format!("backup envelope (catalog rows): {e}"), - })?; - } - } - - // PK→surrogate identity map for the tenant's collections. This is - // DATA-derived per-node state that the per-node engine sections do NOT - // carry (the Data-Plane snapshot handler has no catalog access). Without - // it a restored node has documents but cannot resolve PK point-lookups - // (`WHERE id=`) — full scans work, point-lookups silently miss. The - // restore path rebinds these into the destination catalog. - if let Ok(all) = catalog.load_all_collections(DatabaseId::DEFAULT) { - let mut binds: Vec = Vec::new(); - for coll in all.iter().filter(|c| c.tenant_id == tenant_id) { - if let Ok(rows) = catalog.scan_surrogates_for_collection( - DatabaseId::DEFAULT, - TenantId::new(tenant_id), - &coll.name, - ) { - for (pk, surrogate) in rows { - binds.push(nodedb_types::backup_envelope::SurrogateBindBlob { - tenant_id, - collection: coll.name.clone(), - pk, - surrogate: surrogate.as_u32(), - }); - } - } - } - if !binds.is_empty() - && let Ok(body) = zerompk::to_msgpack_vec(&binds) - { - writer - .push_section( - nodedb_types::backup_envelope::SECTION_ORIGIN_SURROGATE_PK, - body, - ) - .map_err(|e| Error::Internal { - detail: format!("backup envelope (surrogate pk): {e}"), - })?; - } - } - - if let Ok(tset) = catalog.load_wal_tombstones() { - let mut tombs: Vec = Vec::new(); - for (database_id, tid, name, purge_lsn) in tset.iter() { - if database_id == DatabaseId::DEFAULT.as_u64() && tid == tenant_id { - tombs.push(nodedb_types::backup_envelope::SourceTombstoneEntry { - collection: name.to_string(), - purge_lsn, - }); - } - } - if !tombs.is_empty() - && let Ok(body) = zerompk::to_msgpack_vec(&tombs) - { - writer - .push_section( - nodedb_types::backup_envelope::SECTION_ORIGIN_SOURCE_TOMBSTONES, - body, - ) - .map_err(|e| Error::Internal { - detail: format!("backup envelope (source tombstones): {e}"), - })?; - } + for node_sections in per_node { + for (node_id, body) in node_sections? { + writer + .push_section(node_id, body) + .map_err(|e| Error::Internal { + detail: format!("backup envelope: {e}"), + })?; } } + // Metadata sections: databases, catalog rows, surrogate binds and + // source-side tombstones. These live in dedicated sections with sentinel + // origin_node_ids so the restore path can distinguish them from per-node + // engine data. Without these, a backup taken during a collection's + // retention window loses its soft-deleted row (UNDROP can't work after + // restore), and a restore whose source has already purged a collection + // can resurrect rows that were properly reaped. + super::metadata::push_metadata_sections(state, tenant_id, &databases, &mut writer)?; + // A backup KEK must be configured; plaintext backup envelopes are no // longer supported. let envelope_bytes = match &state.backup_kek { @@ -204,6 +121,53 @@ pub async fn backup_tenant(state: &Arc, tenant_id: u64) -> Result { + node_id: u64, + tenant_id: u64, + snapshot_watermark: u64, + /// The vShards this node is the assigned source for. + source_vshards: &'a HashSet, + databases: &'a [TenantDatabase], +} + +/// Snapshot every database of the tenant on one source node. Returns one +/// `(origin_node_id, section_body)` per database. +/// +/// This node took the cut already. A remote node takes the cut at the same +/// watermark before its first snapshot. Its later snapshots follow that cut. +async fn snapshot_node( + state: &Arc, + job: NodeSnapshot<'_>, +) -> Result)>, Error> { + let local = is_self(state, job.node_id); + let mut sections = Vec::with_capacity(job.databases.len()); + for (index, database) in job.databases.iter().enumerate() { + let database_id = database.id(); + let body = if local { + snapshot_self(state, job.tenant_id, database_id).await? + } else { + // A remote node takes the cut before its first snapshot. + let plan = PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { + tenant_id: job.tenant_id, + cut_watermark: (index == 0).then_some(job.snapshot_watermark), + }); + snapshot_remote(state, job.node_id, job.tenant_id, database_id, &plan).await? + }; + // Single-node / single-replica: the node owns every vshard it leads, + // so the filter retains everything (no-op). Under RF>1: keep only the + // vshards this node is the assigned source for; the other replicas' + // copies are dropped here so the restore merge sums disjoint sections. + let snapshot = filter_node_snapshot(body, job.tenant_id, database_id, job.source_vshards)?; + let section = DatabaseDataSection { + database_id: database_id.as_u64(), + snapshot, + }; + sections.push((job.node_id, encode_section_part("data section", §ion)?)); + } + Ok(sections) +} + /// Assign every vshard to exactly ONE source node, returning the per-source /// `(node_id, owned_vshard_set)` pairs to gather from. /// @@ -249,16 +213,18 @@ fn source_assignment(state: &SharedState) -> Vec<(u64, HashSet)> { by_node.into_iter().collect() } -/// Decode a gathered per-node `TenantDataSnapshot`, filter it in place to the -/// vshards this node is the assigned source for, and re-encode it. +/// Decode a gathered per-node `TenantDataSnapshot` of `database_id`, filter +/// it in place to the vshards this node is the assigned source for, and +/// re-encode it. /// /// The per-section vshard classification is shared with the Raft snapshot SEND /// builder via `snapshot_keys::retain_tenant_data_for_vshards`. The vshard-of -/// closure is the canonical routing function, matching both the snapshot -/// builder and the restore topology splitter. +/// closure routes each stored name in `database_id`, matching the snapshot +/// builder. fn filter_node_snapshot( body: Vec, tenant_id: u64, + database_id: DatabaseId, source_vshards: &HashSet, ) -> Result, Error> { let mut snap: crate::types::TenantDataSnapshot = @@ -269,132 +235,9 @@ fn filter_node_snapshot( &mut snap, tenant_id, source_vshards, - |collection| { - nodedb_cluster::routing::vshard_for_collection(DatabaseId::DEFAULT, collection) - }, + |collection| super::snapshot_keys::vshard_of_stored(database_id, collection), ); zerompk::to_msgpack_vec(&snap).map_err(|e| Error::Internal { detail: format!("backup: re-encode filtered snapshot: {e}"), }) } - -fn is_self(state: &SharedState, node_id: u64) -> bool { - node_id == state.node_id || node_id == 0 || state.cluster_transport.is_none() -} - -async fn snapshot_self( - state: &Arc, - tenant_id: u64, - plan: &PhysicalPlan, -) -> Result, Error> { - sync_dispatch::dispatch_system( - state, - sync_dispatch::SystemTask::new( - sync_dispatch::SystemReason::BackupRestore, - TenantId::new(tenant_id), - // TODO(A8-followup): backup/restore not yet multi-database. - DatabaseId::DEFAULT, - "__system", - plan.clone(), - ), - NODE_SNAPSHOT_TIMEOUT, - ) - .await -} - -async fn snapshot_remote( - state: &Arc, - node_id: u64, - tenant_id: u64, - plan: &PhysicalPlan, -) -> Result, Error> { - let transport = state - .cluster_transport - .as_ref() - .ok_or_else(|| Error::Internal { - detail: format!("backup: cluster_transport unavailable but node {node_id} is remote"), - })?; - - let plan_bytes = plan_wire::encode(plan).map_err(|e| Error::Internal { - detail: format!("backup: plan encode failed: {e}"), - })?; - let req = RaftRpc::ExecuteRequest(ExecuteRequest { - plan_bytes, - tenant_id, - database_id: DatabaseId::DEFAULT.as_u64(), - deadline_remaining_ms: NODE_SNAPSHOT_TIMEOUT.as_millis() as u64, - trace_id: TraceId::generate().0, - descriptor_versions: Vec::new(), - // Backup snapshot dispatch is not session-transaction-scoped. - txn_id: None, - }); - - let resp = transport - .send_rpc(node_id, req) - .await - .map_err(|e| Error::Internal { - detail: format!("backup: snapshot RPC to node {node_id} failed: {e}"), - })?; - match resp { - RaftRpc::ExecuteResponse(ExecuteResponse { - success: true, - mut payloads, - .. - }) => { - // CreateTenantSnapshot returns exactly one payload. - if payloads.len() != 1 { - return Err(Error::Internal { - detail: format!( - "backup: expected 1 payload from node {node_id}, got {}", - payloads.len() - ), - }); - } - Ok(payloads.remove(0)) - } - RaftRpc::ExecuteResponse(ExecuteResponse { - error: Some(err), .. - }) => Err(map_typed_error(err, node_id)), - RaftRpc::ExecuteResponse(_) => Err(Error::Internal { - detail: format!("backup: empty error response from node {node_id}"), - }), - other => Err(Error::Internal { - detail: format!( - "backup: unexpected RPC response variant from node {node_id}: {other:?}" - ), - }), - } -} - -fn map_typed_error(err: TypedClusterError, node_id: u64) -> Error { - match err { - TypedClusterError::Internal { message, .. } => Error::Internal { - detail: format!("backup node {node_id}: {message}"), - }, - TypedClusterError::DeadlineExceeded { elapsed_ms } => Error::Internal { - detail: format!("backup node {node_id}: deadline exceeded after {elapsed_ms}ms"), - }, - TypedClusterError::NotLeader { .. } => Error::Internal { - detail: format!("backup node {node_id}: snapshot RPC routed to non-leader"), - }, - TypedClusterError::DescriptorMismatch { collection, .. } => Error::Internal { - detail: format!( - "backup node {node_id}: descriptor mismatch on collection {collection}" - ), - }, - // Keep the shard's verdict typed: a backup snapshot refused by the - // Data Plane must not read as a generic internal backup fault. - TypedClusterError::DataPlane { code } => Error::DataPlane(code.into()), - // A constraint verdict keeps its collection and kind, so the client - // reads the SQLSTATE the refusing shard meant. - TypedClusterError::RejectedConstraint { - collection, - constraint, - detail, - } => Error::RejectedConstraint { - collection, - constraint, - detail, - }, - } -} diff --git a/nodedb/src/control/backup/restore/columnar_reissue.rs b/nodedb/src/control/backup/restore/columnar_reissue.rs index 549a4d3c4..323fb5105 100644 --- a/nodedb/src/control/backup/restore/columnar_reissue.rs +++ b/nodedb/src/control/backup/restore/columnar_reissue.rs @@ -9,7 +9,6 @@ //! on single-node — the same branch a normal write takes. use std::collections::HashMap; -use std::time::Duration; use nodedb_columnar::{ColumnarEngineSnapshot, MutationEngine, materialize_segment_live_rows}; use nodedb_types::RlsWriteCheck; @@ -18,10 +17,6 @@ use nodedb_types::value::Value; use crate::Error; use crate::bridge::envelope::PhysicalPlan; -use crate::control::server::shared::ddl::sync_dispatch; -use crate::control::server::wal_dispatch::wal_append_if_write; -use crate::control::state::SharedState; -use crate::types::{DatabaseId, TenantId, VShardId}; use nodedb_physical::physical_plan::{ColumnarInsertIntent, ColumnarOp}; /// Live rows of a decoded snapshot, ready to re-issue. @@ -161,58 +156,6 @@ pub fn build_columnar_insert_plan( })) } -/// Re-issue a restored columnar collection's rows durably. -/// -/// Branches identically to a normal write: -/// - Cluster: `to_replicated_entry` + `propose_replicated_entry`. -/// - Single-node: `wal_append_if_write` then `sync_dispatch::dispatch_system`. -pub async fn reissue_columnar_durably( - state: &SharedState, - tenant_id: TenantId, - database_id: DatabaseId, - collection: &str, - plan: PhysicalPlan, -) -> crate::Result<()> { - let vshard = VShardId::from_collection_in_database(database_id, collection); - - if let Some(proposer) = state.async_raft_proposer() { - let entry = crate::control::wal_replication::to_replicated_entry( - tenant_id, - database_id, - vshard, - &crate::control::wal_replication::ReplicableWrite::decide_for_replication(&plan)?, - )? - .ok_or_else(|| Error::Internal { - detail: format!( - "restore reissue: columnar plan for '{collection}' did not map to a \ - replicated write" - ), - })?; - crate::control::wal_replication::propose_replicated_entry(state, proposer, entry).await?; - return Ok(()); - } - - // Single-node: WAL first (durable for restart replay), then install live. - wal_append_if_write(&state.wal, tenant_id, vshard, database_id, &plan)?; - sync_dispatch::dispatch_system( - state, - sync_dispatch::SystemTask::new( - sync_dispatch::SystemReason::BackupRestore, - tenant_id, - database_id, - collection, - plan, - ), - REISSUE_TIMEOUT, - ) - .await?; - Ok(()) -} - -/// Per-collection re-issue dispatch timeout. Generous: a restored collection may -/// carry many flushed segments' worth of rows in one insert. -const REISSUE_TIMEOUT: Duration = Duration::from_secs(120); - /// Convert a row's positional `Value`s (schema column order) into a field-keyed /// `Value::Object`. Errors if the arity does not match the schema. fn row_values_to_object( diff --git a/nodedb/src/control/backup/restore/crdt_reissue.rs b/nodedb/src/control/backup/restore/crdt_reissue.rs index 9f99877e1..437742509 100644 --- a/nodedb/src/control/backup/restore/crdt_reissue.rs +++ b/nodedb/src/control/backup/restore/crdt_reissue.rs @@ -17,9 +17,11 @@ use crate::bridge::envelope::{PhysicalPlan, Status}; use crate::control::server::dispatch_utils::{AutocommitWrite, dispatch_autocommit_write}; use crate::control::state::SharedState; use crate::event::EventSource; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::CrdtOp; +use super::target::{DatabaseTarget, RestoredName}; + /// Per-import dispatch timeout. Generous: a collection's Loro snapshot may be /// large. const REISSUE_TIMEOUT: Duration = Duration::from_secs(120); @@ -27,20 +29,20 @@ const REISSUE_TIMEOUT: Duration = Duration::from_secs(120); /// Re-issue one collection's snapshot import to the data group owning its /// vshard. /// -/// Branches identically to a normal write (and to `reissue_timeseries_durably`): +/// Branches identically to a normal write (and to `durable::reissue_plan_durably`): /// - Cluster: `to_replicated_entry` + `propose_replicated_entry`. -/// - Single-node: `wal_append_if_write` then `sync_dispatch::dispatch_system`. +/// - Single-node: the autocommit funnel appends the redo and installs it. async fn reissue_crdt_collection( state: &SharedState, tenant_id: TenantId, database_id: DatabaseId, - collection: &str, + name: RestoredName, bytes: Vec, ) -> crate::Result<()> { - let vshard = VShardId::from_collection_in_database(database_id, collection); + let vshard = name.key(database_id).vshard(); let plan = PhysicalPlan::Crdt(CrdtOp::ImportSnapshot { tenant_id: tenant_id.as_u64(), - collection: nodedb_types::QualifiedCollection::from_stored(collection.to_string()), + collection: name.stored, bytes, }); @@ -53,7 +55,8 @@ async fn reissue_crdt_collection( )? .ok_or_else(|| Error::Internal { detail: "restore reissue: crdt import did not map to a replicated write".into(), - })?; + })? + .with_event_source(EventSource::Restore); crate::control::wal_replication::propose_replicated_entry(state, proposer, entry).await?; return Ok(()); } @@ -74,7 +77,7 @@ async fn reissue_crdt_collection( vshard_id: vshard, plan, trace_id: crate::types::TraceId::ZERO, - event_source: EventSource::CrdtSync, + event_source: EventSource::Restore, txn_id: None, }, ), @@ -101,26 +104,31 @@ async fn reissue_crdt_collection( .await } -/// Durably re-issue every restored CRDT collection snapshot. +/// Durably re-issue every restored CRDT collection snapshot of one database. /// -/// `crdt_state` entries are `(tenant_id, collection, snapshot_bytes)`; each is -/// routed to the single data group owning that collection's vshard. Returns the -/// number of imports issued. +/// `crdt_state` entries are `(database_id, tenant_id, collection, +/// snapshot_bytes)`, the collection named as the source Data Plane stored it. +/// Each is routed to the single data group owning its destination +/// collection's vshard. Returns the number of imports issued. pub(crate) async fn reissue_crdt_snapshots( state: &SharedState, + target: DatabaseTarget, crdt_state: Vec<(u64, u64, String, Vec)>, ) -> crate::Result { let mut imported = 0usize; for (database_id, tid, collection, bytes) in crdt_state { - reissue_crdt_collection( - state, - TenantId::new(tid), - DatabaseId::new(database_id), - &collection, - bytes, - ) - .await?; + if database_id != target.source.as_u64() { + return Err(Error::Internal { + detail: format!( + "invalid backup format: CRDT state of '{collection}' names database \ + {database_id}, but sits with database {}", + target.source.as_u64() + ), + }); + } + let name = target.resolve(&collection)?; + reissue_crdt_collection(state, TenantId::new(tid), target.dest, name, bytes).await?; imported += 1; } diff --git a/nodedb/src/control/backup/restore/databases.rs b/nodedb/src/control/backup/restore/databases.rs new file mode 100644 index 000000000..39e643b98 --- /dev/null +++ b/nodedb/src/control/backup/restore/databases.rs @@ -0,0 +1,229 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Map every backed-up database to its destination database. +//! +//! The backup names each database by its source id and its name. The +//! destination database is the one of the same name. A database the +//! destination does not have is created with a fresh id: its descriptor +//! settings, its quota and the tenant's quota in it come from the backup. +//! Every entry is proposed through the metadata Raft group, exactly like +//! `CREATE DATABASE` and `ALTER ... SET QUOTA`, so every node learns it. +//! +//! Database grants are not carried: a grant names a user id of the source +//! cluster, and a tenant backup carries no users. + +use std::collections::BTreeMap; + +use nodedb_types::QuotaRecord; +use nodedb_types::backup_envelope::{DatabaseBlob, Envelope, SECTION_ORIGIN_DATABASES}; + +use crate::Error; +use crate::control::catalog_entry::CatalogEntry; +use crate::control::catalog_entry::post_apply::quota as quota_apply; +use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::security::catalog::{DatabaseDescriptor, DatabaseStatus}; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId}; + +use super::target::DatabaseTarget; + +/// The destination database of every backed-up database, keyed by source id. +#[derive(Debug, Default)] +pub(super) struct DatabaseMap { + targets: BTreeMap, + /// Databases the restore created on this cluster. + created: usize, +} + +impl DatabaseMap { + /// Number of databases the restore created on this cluster. + pub fn created(&self) -> usize { + self.created + } + + /// The target of the source database `source`, when it has one. A dry + /// run leaves a database the destination lacks without a target. + pub fn get(&self, source: u64) -> Option { + self.targets.get(&source).copied() + } + + /// The target of the source database `source`. A section that names a + /// database the backup's database section does not list is malformed. + pub fn target(&self, source: u64) -> Result { + self.get(source).ok_or_else(|| Error::Internal { + detail: format!( + "invalid backup format: a section names database {source}, which the \ + backup's database section does not list" + ), + }) + } +} + +/// Decode the database section of `env`. An envelope with rows but no +/// database section fails the restore when a row section names a database. +pub(super) fn decode_databases(env: &Envelope) -> Result, Error> { + let mut databases = Vec::new(); + for section in &env.sections { + if section.origin_node_id == SECTION_ORIGIN_DATABASES { + let blobs: Vec = + zerompk::from_msgpack(§ion.body).map_err(|_| Error::Internal { + detail: "invalid backup format: database section is not decodable".into(), + })?; + databases.extend(blobs); + } + } + Ok(databases) +} + +/// Map every database of `blobs` to its destination. Unless `dry_run`, +/// create each database the destination lacks and restore the tenant's +/// quota in each database that has none. +pub(super) fn resolve_databases( + state: &SharedState, + tenant_id: u64, + blobs: &[DatabaseBlob], + dry_run: bool, +) -> Result { + let catalog = state.credentials.catalog(); + let mut map = DatabaseMap::default(); + for blob in blobs { + let dest = match catalog.get_database_id_by_name(&blob.name)? { + Some(id) => { + require_writable(state, id, &blob.name)?; + id + } + None if dry_run => continue, + None => { + map.created += 1; + create_database(state, blob)? + } + }; + if !dry_run && let Some(record) = &blob.tenant_quota { + restore_tenant_quota(state, dest, TenantId::new(tenant_id), record)?; + } + map.targets.insert( + blob.database_id, + DatabaseTarget { + source: DatabaseId::new(blob.database_id), + dest, + }, + ); + } + Ok(map) +} + +/// A restore writes rows, so the destination database must take writes. +fn require_writable(state: &SharedState, id: DatabaseId, name: &str) -> Result<(), Error> { + let status = state + .credentials + .catalog() + .get_database(id)? + .map(|descriptor| descriptor.status); + match status { + Some(DatabaseStatus::Active) => Ok(()), + Some(DatabaseStatus::Deactivated | DatabaseStatus::Cloning | DatabaseStatus::Mirroring) + | None => Err(Error::BadRequest { + detail: format!( + "restore refused: database '{name}' exists on this cluster but does not take \ + writes (status {status:?}). Make it active or remove it, then retry the restore" + ), + }), + } +} + +/// Create the database `blob` describes under a fresh id, with its settings +/// and its quota. Returns the new id. +fn create_database(state: &SharedState, blob: &DatabaseBlob) -> Result { + let source: DatabaseDescriptor = + zerompk::from_msgpack(&blob.descriptor).map_err(|_| Error::Internal { + detail: format!( + "invalid backup format: descriptor of database '{}' is not decodable", + blob.name + ), + })?; + let catalog = state.credentials.catalog(); + let id = state.database_registry.alloc_one(); + // The restored database is a new, independent database: it is active, + // and it is no clone or mirror of a database on this cluster. + let descriptor = DatabaseDescriptor { + id, + name: blob.name.clone(), + status: DatabaseStatus::Active, + created_at_lsn: state.wal.next_lsn().as_u64(), + parent_clone: None, + mirror_origin: None, + ..source + }; + let entry = CatalogEntry::PutDatabase(Box::new(descriptor.clone())); + if propose_catalog_entry(state, &entry)?.needs_local_apply() { + catalog.put_database(&descriptor)?; + } + // Persist the allocator high-water mark now, so a restart never hands + // the restored id to another database. + catalog.put_database_hwm(state.database_registry.current_hwm())?; + + if let Some(record) = &blob.database_quota { + restore_database_quota(state, id, record)?; + } + if let Some(m) = &state.system_metrics { + m.set_database_collections(&blob.name, 0); + m.set_database_tenants(&blob.name, 0); + m.set_database_memory_bytes(&blob.name, 0); + m.set_database_storage_bytes(&blob.name, 0); + } + state.audit_record_with_db( + crate::control::security::audit::AuditEvent::DatabaseCreated, + None, + Some(id), + "__restore", + &format!( + "RESTORE created database '{}' (source id {})", + blob.name, blob.database_id + ), + ); + Ok(id) +} + +/// Install the backed-up quota of a database the restore created. +fn restore_database_quota( + state: &SharedState, + id: DatabaseId, + record: &QuotaRecord, +) -> Result<(), Error> { + let catalog = state.credentials.catalog(); + catalog.check_database_quota(id, record, &state.quota_ceiling_snapshot())?; + let entry = CatalogEntry::PutDatabaseQuota { + db_id: id.as_u64(), + record: Box::new(record.clone()), + }; + if propose_catalog_entry(state, &entry)?.needs_local_apply() { + catalog.write_database_quota(id, record)?; + quota_apply::put_database(id, record, state); + } + Ok(()) +} + +/// Install the tenant's backed-up quota in `id`, unless the destination +/// already sets one there. +fn restore_tenant_quota( + state: &SharedState, + id: DatabaseId, + tenant: TenantId, + record: &QuotaRecord, +) -> Result<(), Error> { + let catalog = state.credentials.catalog(); + if catalog.get_tenant_quota(id, tenant)?.is_some() { + return Ok(()); + } + catalog.check_tenant_quota(id, tenant, record)?; + let entry = CatalogEntry::PutTenantQuota { + db_id: id.as_u64(), + tenant_id: tenant.as_u64(), + record: Box::new(record.clone()), + }; + if propose_catalog_entry(state, &entry)?.needs_local_apply() { + catalog.write_tenant_quota(id, tenant, record)?; + quota_apply::put_tenant(id, tenant, record, state); + } + Ok(()) +} diff --git a/nodedb/src/control/backup/restore/durable.rs b/nodedb/src/control/backup/restore/durable.rs new file mode 100644 index 000000000..8bf326861 --- /dev/null +++ b/nodedb/src/control/backup/restore/durable.rs @@ -0,0 +1,124 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The durable write every RESTORE re-issue goes through. +//! +//! A restored row installed straight into a Data-Plane map has no WAL record +//! and no Raft entry: it is lost on restart, and only the one node that +//! installed it holds it. A re-issued row is a normal write instead: every +//! replica of its group applies it, and each one's WAL makes it durable. + +use std::time::Duration; + +use crate::Error; +use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; +use crate::control::server::shared::ddl::sync_dispatch; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId, VShardId}; + +/// Dispatch timeout of one re-issued write. Generous: one restored +/// collection's rows may travel in a single write. +const REISSUE_TIMEOUT: Duration = Duration::from_secs(120); + +/// Write `plan`, restored into `collection`, durably. `collection` is the +/// bare catalog name in `database_id`. +/// +/// Branches identically to a normal write: +/// - Cluster: `to_replicated_entry` + `propose_replicated_entry`. +/// - Single-node: append the redo under an outcome-floor window, then +/// `sync_dispatch::dispatch_system`, which closes the window. +/// +/// Both branches give the write's events [`crate::event::EventSource::Restore`]. +pub async fn reissue_plan_durably( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection: &str, + plan: PhysicalPlan, +) -> crate::Result<()> { + let vshard = nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(); + + if let Some(proposer) = state.async_raft_proposer() { + let entry = crate::control::wal_replication::to_replicated_entry( + tenant_id, + database_id, + vshard, + &crate::control::wal_replication::ReplicableWrite::decide_for_replication(&plan)?, + )? + .ok_or_else(|| Error::Internal { + detail: format!( + "restore reissue: the plan restored into '{collection}' did not map to a \ + replicated write" + ), + })? + // Every replica applies the write as restored: AFTER triggers do not + // fire again for it. + .with_event_source(crate::event::EventSource::Restore); + let (_, write_version) = + crate::control::wal_replication::propose_replicated_entry(state, proposer, entry) + .await?; + tracing::debug!( + collection, + vshard_id = vshard.as_u32(), + write_version = write_version.as_u64(), + "restore: re-issued write applied on this node" + ); + return Ok(()); + } + + // Single-node: WAL first (durable for restart replay), then install live. + // The record's outcome-floor window opens before the append and closes + // from the install's outcome. + let owner = RecordOwner { + tenant_id, + database_id, + vshard_id: vshard, + }; + let minted = MintedRecords::open(&state.outcome_floor); + if let Err(error) = minted.append_plan( + &state.wal, + owner, + &plan, + sync_dispatch::SystemReason::BackupRestore.event_source(), + ) { + // Any record appended before the error never reaches a core. + minted.cancel(&state.wal, owner, 0).await?; + return Err(error); + } + sync_dispatch::dispatch_system( + state, + sync_dispatch::SystemTask::new( + sync_dispatch::SystemReason::BackupRestore, + tenant_id, + nodedb_types::CollectionKey::from_bare(database_id, collection), + plan, + ) + .with_minted(minted), + REISSUE_TIMEOUT, + ) + .await?; + Ok(()) +} + +/// Log one restore re-issue step: what it writes, where, and who proposes +/// it, so every step of a restore shows in the log. +pub(super) fn log_reissue_step( + state: &SharedState, + step: &'static str, + collection: &str, + vshard: VShardId, + rows: usize, +) { + let group_id = + crate::control::security::auth_fence::cluster::group_of_vshard(state, vshard.as_u32()).ok(); + tracing::info!( + step, + collection, + vshard_id = vshard.as_u32(), + group_id = ?group_id, + rows, + proposer_node = state.node_id, + replicated = state.async_raft_proposer().is_some(), + "restore: re-issuing rows" + ); +} diff --git a/nodedb/src/control/backup/restore/guard.rs b/nodedb/src/control/backup/restore/guard.rs new file mode 100644 index 000000000..d22c2febe --- /dev/null +++ b/nodedb/src/control/backup/restore/guard.rs @@ -0,0 +1,422 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! RESTORE's staleness guard: the newest committed write of a tenant. +//! +//! The answer comes from replicated state. Every data group's replicas derive +//! the same durable per-tenant marks from the entries they apply, and a +//! committed Calvin transaction records its mark in the data group that homes +//! its vShard before its COMMIT is acknowledged. The guard reads the marks of +//! every data group: +//! +//! - a group this node replicates, here, once this node applied every entry +//! the group committed before the read and every Calvin transaction +//! sequenced before it installed; +//! - any other group, from a current replica, which does the same before it +//! answers. +//! +//! A replica that left the group refuses with a typed `NotLeader`. The guard +//! then asks another replica, until one statement deadline shared by every +//! group. A group no replica answered for by then refuses the restore. +//! +//! A node that restarted, or that never applied the write, answers alike. +//! +//! A mark is kept per `(group, tenant)`, whatever database the write named. +//! Reading every data group's marks therefore covers the groups of every +//! database the tenant writes. + +use std::collections::{BTreeMap, BTreeSet}; +use std::sync::Arc; +use std::time::Duration; + +use nodedb_cluster::METADATA_GROUP_ID; +use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use nodedb_cluster::rpc_codec::{ExecuteRequest, ExecuteResponse, RaftRpc, TypedClusterError}; +use nodedb_physical::physical_plan::{ClusterEventOp, PhysicalPlan, wire as plan_wire}; + +use crate::Error; +use crate::control::security::auth_fence::cluster::{ + confirmed_read_index, hosts_group, routed_groups, wait_applied, +}; +use crate::control::state::SharedState; +use crate::control::state::tenant_marks::{GroupMark, LOCAL_MARK_GROUP}; +use crate::types::{DatabaseId, TraceId}; + +/// First wait before the guard asks again for the marks of groups whose +/// replica refused. Each round doubles it up to [`MAX_ASK_BACKOFF`]. +const FIRST_ASK_BACKOFF: Duration = Duration::from_millis(10); + +/// Longest wait between two rounds of asking. +const MAX_ASK_BACKOFF: Duration = Duration::from_millis(200); + +/// The newest committed write of a tenant, and where it was recorded. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(super) struct NewestWrite { + /// HLC wall time, in nanoseconds, of its commit. + pub hlc: u64, + /// The apply path that recorded it. + pub site: String, + /// The collection it named, when it named one. + pub collection: Option, +} + +/// One group's mark on the wire: `(group_id, commit_hlc, site_code, +/// collection)`. +type WireMark = (u64, u64, u8, String); + +/// The newest committed write of `tenant_id` across every data group, and +/// this node's own mark of writes no data group carries. +pub(super) async fn newest_committed_write( + state: &Arc, + tenant_id: u64, +) -> Result, Error> { + let mut newest = state.tenant_write_mark(tenant_id).map(|mark| NewestWrite { + hlc: mark.hlc, + site: mark.origin.site.to_owned(), + collection: mark.origin.collection, + }); + let mut consider = |mark: GroupMark| { + if newest.as_ref().is_none_or(|current| mark.hlc > current.hlc) { + newest = Some(NewestWrite { + hlc: mark.hlc, + site: mark.site.as_str().to_owned(), + collection: mark.collection, + }); + } + }; + // The durable mark of this node's writes while it ran with no Raft groups. + if let Some(mark) = state.tenant_marks.get(LOCAL_MARK_GROUP, tenant_id) { + consider(mark); + } + + // The statement deadline, shared by every group and every attempt. + let deadline = tokio::time::Instant::now() + + Duration::from_secs(state.tuning.network.default_deadline_secs); + let groups: Vec = routed_groups(state) + .into_iter() + .filter(|group| *group != METADATA_GROUP_ID && *group != SEQUENCER_GROUP_ID) + .collect(); + for (_, mark) in group_marks(state, tenant_id, groups, deadline).await? { + consider(mark); + } + Ok(newest) +} + +/// The marks of `tenant_id` in every group of `groups`, each from a current +/// replica of the group. +/// +/// Each round reads the groups this node replicates here, and asks one +/// replica of every other group. A replica that refuses a group, because it +/// does not replicate it, is not asked for that group again until every known +/// replica refused. Every round ends by `deadline`. A group still unanswered +/// then fails the call with [`Error::GroupMarksUnavailable`]. Any other error +/// fails it at once. +async fn group_marks( + state: &Arc, + tenant_id: u64, + groups: Vec, + deadline: tokio::time::Instant, +) -> Result, Error> { + let mut pending: BTreeMap = groups + .into_iter() + .map(|group| (group, GroupAsk::default())) + .collect(); + let mut marks = Vec::new(); + let mut backoff = FIRST_ASK_BACKOFF; + loop { + let (local, remote): (Vec, Vec) = pending + .keys() + .copied() + .partition(|group| hosts_group(state, *group)); + if !local.is_empty() { + marks.extend(local_tenant_marks(state, tenant_id, &local, deadline).await?); + for group in &local { + pending.remove(group); + } + } + let mut by_node: BTreeMap> = BTreeMap::new(); + for group_id in remote { + if let Some(ask) = pending.get_mut(&group_id) + && let Some(node_id) = ask.next_target(state, group_id) + { + by_node.entry(node_id).or_default().push(group_id); + } + } + let answers = futures::future::join_all(by_node.into_iter().map(|(node_id, groups)| { + let asked = groups.clone(); + async move { + let answer = remote_tenant_marks(state, node_id, tenant_id, groups, deadline).await; + (node_id, asked, answer) + } + })) + .await; + for (node_id, asked, answer) in answers { + match answer { + Ok(answered) => { + marks.extend(answered); + for group in &asked { + pending.remove(group); + } + } + Err(RemoteMarksError::NotReplica { group_id, hint }) => { + if let Some(ask) = pending.get_mut(&group_id) { + ask.refused(node_id, hint); + } + } + Err(RemoteMarksError::Failed(error)) => return Err(error), + } + } + let Some((&group_id, ask)) = pending.iter().next() else { + return Ok(marks); + }; + if tokio::time::Instant::now() + backoff >= deadline { + return Err(Error::GroupMarksUnavailable { + group_id, + refused_by: ask.refused_by.iter().copied().collect(), + }); + } + tokio::time::sleep(backoff).await; + backoff = (backoff * 2).min(MAX_ASK_BACKOFF); + } +} + +/// Which replicas of one group the guard asked, and which refused. +#[derive(Debug, Default)] +struct GroupAsk { + /// Nodes that refused the group since the last time every known replica + /// refused it. + refused: BTreeSet, + /// Every node that refused the group, for the deadline error. + refused_by: BTreeSet, + /// The replica a refusing node's routing table named. + hint: Option, +} + +impl GroupAsk { + /// The node to ask next: the last refusal's hint, then the group's leader, + /// voters and learners in this node's routing table, skipping this node + /// and every node that refused. When every candidate refused, the refused + /// set is cleared for the next round, since routing tables converge, and + /// this round asks no node. + fn next_target(&mut self, state: &SharedState, group_id: u64) -> Option { + let candidates = self.hint.into_iter().chain(group_replicas(state, group_id)); + let mut any = false; + for node in candidates { + if node == 0 || node == state.node_id { + continue; + } + any = true; + if !self.refused.contains(&node) { + return Some(node); + } + } + if any { + self.refused.clear(); + self.hint = None; + } + None + } + + fn refused(&mut self, node_id: u64, hint: Option) { + self.refused.insert(node_id); + self.refused_by.insert(node_id); + self.hint = hint.filter(|hint| !self.refused.contains(hint)); + } +} + +/// The replicas of `group_id` in this node's routing table: its leader when +/// known, then its voters, then its learners. +fn group_replicas(state: &SharedState, group_id: u64) -> Vec { + let Some(routing) = state.cluster_routing.as_ref() else { + return Vec::new(); + }; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + let Some(info) = routing.group_info(group_id) else { + return Vec::new(); + }; + std::iter::once(info.leader) + .chain(info.members.iter().copied()) + .chain(info.learners.iter().copied()) + .collect() +} + +/// This node's marks of `tenant_id` in `group_ids`, once this node applied +/// every entry the groups committed before the call and every Calvin +/// transaction sequenced before it installed here. Every wait ends by +/// `deadline`. +pub(crate) async fn local_tenant_marks( + state: &Arc, + tenant_id: u64, + group_ids: &[u64], + deadline: tokio::time::Instant, +) -> Result, Error> { + if !group_ids.is_empty() { + let marker = state.hlc_clock.now().wall_ns; + crate::control::backup::cut::cut_calvin(state, marker, deadline).await?; + } + let mut marks = Vec::new(); + for &group_id in group_ids { + let index = confirmed_read_index(state, group_id, remaining(deadline)?).await?; + wait_applied(state, group_id, index, remaining(deadline)?).await?; + if let Some(mark) = state.tenant_marks.get(group_id, tenant_id) { + marks.push((group_id, mark)); + } + } + Ok(marks) +} + +/// Encode marks for the wire. +pub(crate) fn encode_marks(marks: &[(u64, GroupMark)]) -> Result, Error> { + let wire: Vec = marks + .iter() + .map(|(group_id, mark)| { + ( + *group_id, + mark.hlc, + mark.site.code(), + mark.collection.clone().unwrap_or_default(), + ) + }) + .collect(); + zerompk::to_msgpack_vec(&wire).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("tenant write marks: encode: {e}"), + }) +} + +fn decode_marks(bytes: &[u8]) -> Result, Error> { + let wire: Vec = zerompk::from_msgpack(bytes).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("tenant write marks: decode: {e}"), + })?; + Ok(wire + .into_iter() + .map(|(group_id, hlc, site, collection)| { + ( + group_id, + GroupMark { + hlc, + site: crate::control::state::tenant_marks::MarkSite::from_code(site), + collection: (!collection.is_empty()).then_some(collection), + }, + ) + }) + .collect()) +} + +/// The time left before `deadline`, or the deadline error once it passed. +fn remaining(deadline: tokio::time::Instant) -> Result { + let left = deadline.saturating_duration_since(tokio::time::Instant::now()); + if left.is_zero() { + return Err(Error::DeadlineExceeded { + request_id: crate::types::RequestId::new(0), + }); + } + Ok(left) +} + +/// Why a replica gave no marks. +enum RemoteMarksError { + /// The node does not replicate `group_id`. `hint` is the replica its + /// routing table names. + NotReplica { group_id: u64, hint: Option }, + /// Any other error. It ends the guard. + Failed(Error), +} + +impl From for RemoteMarksError { + fn from(error: Error) -> Self { + Self::Failed(error) + } +} + +/// Ask `node_id` for its marks of `tenant_id` in `group_ids`, within what +/// remains of `deadline`. +async fn remote_tenant_marks( + state: &SharedState, + node_id: u64, + tenant_id: u64, + group_ids: Vec, + deadline: tokio::time::Instant, +) -> Result, RemoteMarksError> { + let transport = state + .cluster_transport + .as_ref() + .ok_or_else(|| Error::Internal { + detail: format!( + "restore: node {node_id} replicates a data group of the tenant, but this node \ + has no cluster transport to ask it for the group's newest write" + ), + })?; + let plan = PhysicalPlan::ClusterEvent(ClusterEventOp::TenantWriteMarks { + tenant_id, + group_ids, + }); + let plan_bytes = plan_wire::encode(&plan).map_err(|e| Error::Internal { + detail: format!("restore: encode the tenant write-mark request: {e}"), + })?; + let budget = remaining(deadline)?; + // A group's marks cover the tenant's writes in every database, so the + // request needs no database. Its database id only frames it. + let request = RaftRpc::ExecuteRequest(ExecuteRequest { + plan_bytes, + tenant_id, + database_id: DatabaseId::DEFAULT.as_u64(), + deadline_remaining_ms: u64::try_from(budget.as_millis()).unwrap_or(u64::MAX), + trace_id: TraceId::generate().0, + descriptor_versions: Vec::new(), + txn_id: None, + }); + let response = tokio::time::timeout_at(deadline, transport.send_rpc(node_id, request)) + .await + .map_err(|_| Error::DeadlineExceeded { + request_id: crate::types::RequestId::new(0), + })? + .map_err(|e| Error::Internal { + detail: format!("restore: tenant write-mark request to node {node_id} failed: {e}"), + })?; + match response { + RaftRpc::ExecuteResponse(ExecuteResponse { + success: true, + payloads, + .. + }) => match payloads.as_slice() { + [payload] => Ok(decode_marks(payload)?), + _ => Err(Error::Internal { + detail: format!( + "restore: node {node_id} answered the tenant write-mark request with {} \ + payloads, expected 1", + payloads.len() + ), + } + .into()), + }, + RaftRpc::ExecuteResponse(ExecuteResponse { + error: + Some(TypedClusterError::NotLeader { + group_id, + leader_node_id, + .. + }), + .. + }) => Err(RemoteMarksError::NotReplica { + group_id, + hint: leader_node_id, + }), + RaftRpc::ExecuteResponse(ExecuteResponse { + error: Some(error), .. + }) => Err(Error::from(error).into()), + RaftRpc::ExecuteResponse(ExecuteResponse { error: None, .. }) => Err(Error::Internal { + detail: format!( + "restore: node {node_id} failed the tenant write-mark request without an error" + ), + } + .into()), + other => Err(Error::Internal { + detail: format!( + "restore: unexpected reply to the tenant write-mark request from node \ + {node_id}: {other:?}" + ), + } + .into()), + } +} diff --git a/nodedb/src/control/backup/restore/kv_reissue.rs b/nodedb/src/control/backup/restore/kv_reissue.rs new file mode 100644 index 000000000..061962b06 --- /dev/null +++ b/nodedb/src/control/backup/restore/kv_reissue.rs @@ -0,0 +1,97 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Durable re-issue of restored KV rows. +//! +//! The per-node snapshot install puts a KV row straight into the target +//! node's memtable, with no WAL record and no Raft entry: only that node holds +//! it, and it is gone after a restart. RESTORE re-issues each row as a +//! `KvOp::Put` instead, so every replica of the collection's group applies it +//! and its WAL makes it durable. + +use nodedb_physical::physical_plan::KvOp; + +use crate::Error; +use crate::bridge::envelope::PhysicalPlan; +use crate::control::state::SharedState; +use crate::types::TenantId; + +use super::target::DatabaseTarget; + +/// One restored KV table's rows: `(key, value, expire_at_ms)`, the shape the +/// KV snapshot captures. `expire_at_ms` is `0` for a row with no TTL. +type KvRows = Vec<(Vec, Vec, u64)>; + +/// The TTL a restored row keeps at `now_ms`: `Some(0)` for no TTL, the time +/// left for a row that has not expired, `None` for a row already expired. +fn remaining_ttl_ms(expire_at_ms: u64, now_ms: u64) -> Option { + match expire_at_ms { + 0 => Some(0), + at if at > now_ms => Some(at - now_ms), + _ => None, + } +} + +/// Decode and durably re-issue every restored KV table of one database. +/// Each table key is `"{db}:{tid}:{collection}"`, the collection named as the +/// source KV engine stored it. Returns the number of rows re-issued. +pub(in crate::control::backup::restore) async fn reissue_kv_tables( + state: &SharedState, + tenant_id: u64, + target: DatabaseTarget, + tables: Vec<(String, Vec)>, +) -> Result { + let tenant = TenantId::new(tenant_id); + let mut reissued = 0usize; + for (table_key, bytes) in tables { + let name = target.resolve_scoped(&table_key, tenant_id)?; + let rows: KvRows = zerompk::from_msgpack(&bytes).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("restore reissue: deserialize KV table '{table_key}': {e}"), + })?; + super::durable::log_reissue_step( + state, + "kv", + &name.bare, + name.key(target.dest).vshard(), + rows.len(), + ); + let now_ms = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_millis() as u64) + .unwrap_or(0); + for (key, value, expire_at_ms) in rows { + let Some(ttl_ms) = remaining_ttl_ms(expire_at_ms, now_ms) else { + continue; + }; + let surrogate = state + .surrogate_assigner + .assign(name.key(target.dest), tenant, &key)?; + let plan = PhysicalPlan::Kv(KvOp::Put { + collection: name.stored.clone(), + key, + value, + ttl_ms, + surrogate, + returning: None, + rls_filters: Vec::new(), + provenance: None, + }); + super::durable::reissue_plan_durably(state, tenant, target.dest, &name.bare, plan) + .await?; + reissued += 1; + } + } + Ok(reissued) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn an_expired_row_is_not_restored() { + assert_eq!(remaining_ttl_ms(0, 500), Some(0)); + assert_eq!(remaining_ttl_ms(900, 500), Some(400)); + assert_eq!(remaining_ttl_ms(400, 500), None); + } +} diff --git a/nodedb/src/control/backup/restore/mod.rs b/nodedb/src/control/backup/restore/mod.rs index 103f79bbf..0ff463c4a 100644 --- a/nodedb/src/control/backup/restore/mod.rs +++ b/nodedb/src/control/backup/restore/mod.rs @@ -3,17 +3,23 @@ //! RESTORE TENANT — module root. //! //! Submodule wiring only. All restore orchestrator logic lives in -//! [`orchestrate`]; column/timeseries/vector re-issue helpers in their -//! respective submodules; supporting primitives in `remote`, `sections`, and -//! `topology`. +//! [`orchestrate`]; each engine's re-issue lives in its own submodule; the +//! section decoding lives in `sections`; the destination databases are +//! resolved in `databases`; `target` maps a source collection name to its +//! destination database. pub mod columnar_reissue; pub(crate) mod crdt_reissue; +mod databases; +mod durable; +pub(crate) mod guard; +mod kv_reissue; mod orchestrate; -mod remote; +mod quorum; +mod redo_reissue; mod sections; +mod target; pub mod timeseries_reissue; -mod topology; pub mod vector_reissue; pub use orchestrate::{RestoreStats, restore_tenant}; diff --git a/nodedb/src/control/backup/restore/orchestrate/database.rs b/nodedb/src/control/backup/restore/orchestrate/database.rs new file mode 100644 index 000000000..6e6b64bc8 --- /dev/null +++ b/nodedb/src/control/backup/restore/orchestrate/database.rs @@ -0,0 +1,115 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Re-issue one backed-up database's rows into its destination database. +//! +//! Every section re-issues as durable writes: Raft-replicated to every +//! replica of its group in cluster mode, WAL-appended then installed on a +//! single node. None is installed straight into a Data-Plane map, which +//! would hold it on one node only and lose it on restart. Every write names +//! the destination database, so each row lands on the vShard and core its +//! destination collection key homes to. + +use std::sync::Arc; + +use crate::Error; +use crate::control::state::SharedState; +use crate::types::TenantDataSnapshot; + +use super::super::target::DatabaseTarget; +use super::rebind; +use super::reissue; +use super::stats::RestoreStats; + +/// Re-issue every section of `snap`, the merged backup of the source +/// database `target.source`, into `target.dest`. Any failure is fatal — no +/// warn-and-continue. +pub(super) async fn reissue_database( + state: &Arc, + tenant_id: u64, + target: DatabaseTarget, + mut snap: TenantDataSnapshot, + stats: &mut RestoreStats, +) -> Result<(), Error> { + let columnar_snapshots = std::mem::take(&mut snap.columnar_engines); + let timeseries_memtables = std::mem::take(&mut snap.timeseries); + let flushed_ts_segments = std::mem::take(&mut snap.flushed_ts_segments); + let crdt_state = std::mem::take(&mut snap.crdt_state); + let kv_tables = std::mem::take(&mut snap.kv_tables); + let vector_snapshots = std::mem::take(&mut snap.vectors); + // Vector-index config re-issues as `VectorOp::SetParams` before the first + // vector `Insert`: the Data Plane creates a (collection, field) HNSW index + // on its first `Insert`, from whatever params it holds by then. + let vector_params_snapshots = std::mem::take(&mut snap.vector_params); + let index_config_snapshots = std::mem::take(&mut snap.index_configs); + + // The PK→surrogate identity map. It is bound on this node before any + // re-issue, so a re-issued row keeps the surrogate the backup stored it + // under unless this node already binds its key. + let surrogate_binds = std::mem::take(&mut snap.surrogate_pk); + rebind::rebind_surrogates(state, target, &surrogate_binds)?; + + // Document rows, their versions and graph edges re-issue as committed + // redo records through each collection's apply log: every replica binds + // the rows' identities, appends the record to its WAL, installs the rows + // and derives their secondary index entries. The backup's own index + // entries are therefore not installed. + let redo = super::super::redo_reissue::reissue_rows_and_edges( + state, + tenant_id, + target, + super::super::redo_reissue::RestoredRows { + documents: std::mem::take(&mut snap.documents), + documents_versioned: std::mem::take(&mut snap.documents_versioned), + edges: std::mem::take(&mut snap.edges), + binds: &surrogate_binds, + }, + ) + .await?; + stats.documents_reissued += redo.documents; + stats.edges_reissued += redo.edges; + stats.redo_records += redo.records; + + // Plain-columnar rows: each collection's live rows replay as one durable + // `ColumnarOp::Insert`. Collections with zero live rows are skipped. + stats.columnar_engines += + reissue::reissue_columnar_snapshots(state, tenant_id, target, columnar_snapshots).await?; + + // Timeseries rows: each collection's memtable rows plus every flushed + // partition's rows replay as one durable `TimeseriesOp::Ingest`. + stats.timeseries_reissued += reissue::reissue_timeseries_snapshots( + state, + tenant_id, + target, + timeseries_memtables, + flushed_ts_segments, + ) + .await?; + + // CRDT state: each collection's Loro snapshot is proposed to the data + // group owning that collection's vshard. Every replica applies the same + // idempotent Loro merge and converges deterministically. + stats.crdt_reissued += + super::super::crdt_reissue::reissue_crdt_snapshots(state, target, crdt_state).await?; + + // KV rows, one `KvOp::Put` per live row. + stats.kv_reissued += + super::super::kv_reissue::reissue_kv_tables(state, tenant_id, target, kv_tables).await?; + + // Vector-index configuration, as `VectorOp::SetParams`. MUST run before + // the vector-insert re-issue below — see the `vector_params_snapshots` + // drain comment above. + stats.vector_params_reissued += reissue::reissue_vector_params( + state, + tenant_id, + target, + vector_params_snapshots, + index_config_snapshots, + ) + .await?; + + // Vector rows, one `VectorOp::Insert` per restored vector. + stats.vectors_reissued += + reissue::reissue_vector_snapshots(state, tenant_id, target, vector_snapshots).await?; + + Ok(()) +} diff --git a/nodedb/src/control/backup/restore/orchestrate/mod.rs b/nodedb/src/control/backup/restore/orchestrate/mod.rs index c0c717022..63b9632ff 100644 --- a/nodedb/src/control/backup/restore/orchestrate/mod.rs +++ b/nodedb/src/control/backup/restore/orchestrate/mod.rs @@ -2,15 +2,15 @@ //! RESTORE TENANT orchestrator logic. //! -//! Validates a backup envelope, merges all sections into a single -//! `TenantDataSnapshot`, then splits the merged snapshot into per-node -//! sub-snapshots according to the *current* cluster topology and -//! dispatches `MetaOp::RestoreTenantSnapshot` to each owning node. +//! Validates a backup envelope, merges the sections of each backed-up +//! database into one `TenantDataSnapshot`, then re-issues every section as +//! durable, replicated writes into the destination database. //! -//! Durable re-issue of columnar/timeseries/vector rows lives in [`reissue`]; -//! post-install surrogate rebinding and tombstone warnings live in -//! [`rebind`]. +//! [`database`] re-issues one database. Durable re-issue of +//! columnar/timeseries/vector rows lives in [`reissue`]; surrogate rebinding +//! and tombstone warnings live in [`rebind`]. +mod database; mod rebind; mod reissue; mod restore; diff --git a/nodedb/src/control/backup/restore/orchestrate/rebind.rs b/nodedb/src/control/backup/restore/orchestrate/rebind.rs index d8027fa9e..31a2c8782 100644 --- a/nodedb/src/control/backup/restore/orchestrate/rebind.rs +++ b/nodedb/src/control/backup/restore/orchestrate/rebind.rs @@ -1,36 +1,53 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Post-install surrogate rebinding and tombstoned-collection warnings for +//! Surrogate rebinding and tombstoned-collection warnings for //! [`super::restore_tenant`]. +use std::collections::BTreeSet; use std::sync::Arc; -use nodedb_types::Surrogate; +use nodedb_types::{CollectionKey, Surrogate}; use crate::Error; +use crate::control::backup::snapshot_keys::{ + extract_db_scoped_collection, extract_db_tenant_scoped_collection, +}; use crate::control::state::SharedState; -use crate::types::{DatabaseId, SurrogateBindEntry, TenantDataSnapshot, TenantId}; +use crate::engine::graph::edge_store::parse_versioned_edge_key; +use crate::types::{SurrogateBindEntry, TenantDataSnapshot, TenantId}; -/// Rebind every PK→surrogate identity carried in the backup into the -/// destination catalog so restored rows resolve by PK point-lookup. +use super::super::target::DatabaseTarget; + +/// Bind every PK→surrogate identity the backup carries for one database on +/// this node, before any re-issue, so a re-issued row keeps the surrogate it +/// was stored under. Each bind names the bare collection, keyed in the +/// destination database. /// -/// No-op when the snapshot carried no bindings (e.g. an older backup created -/// before the surrogate-pk section existed) or when the node has no catalog. -/// Any catalog write failure is FATAL. +/// Binding is first-wins: a key this node already binds keeps its surrogate, +/// and the re-issue writes that row over it. Each bind also raises this +/// node's surrogate high-water mark past the backup's surrogate, so no later +/// allocation here reuses it. Every replica binds the identities a re-issued +/// write carries as it applies the write. Any bind error is fatal. pub(super) fn rebind_surrogates( state: &Arc, - binds: Vec, + target: DatabaseTarget, + binds: &[SurrogateBindEntry], ) -> Result<(), Error> { - if binds.is_empty() { - return Ok(()); - } - let catalog = state.credentials.catalog(); - let database_id = crate::types::DatabaseId::DEFAULT; - for e in &binds { - catalog.put_surrogate( - database_id, + for e in binds { + if e.database_id != target.source.as_u64() { + return Err(Error::Internal { + detail: format!( + "invalid backup format: a surrogate bind of '{}' names database {}, but \ + sits with database {}", + e.collection, + e.database_id, + target.source.as_u64() + ), + }); + } + state.surrogate_assigner.bind( + nodedb_types::CollectionKey::from_bare(target.dest, &e.collection), TenantId::new(e.tenant_id), - &e.collection, &e.pk, Surrogate::new(e.surrogate), )?; @@ -41,6 +58,7 @@ pub(super) fn rebind_surrogates( pub(super) fn warn_on_tombstoned_restores( state: &Arc, tenant_id: u64, + target: DatabaseTarget, merged: &TenantDataSnapshot, snapshot_watermark: u64, ) { @@ -52,25 +70,9 @@ pub(super) fn warn_on_tombstoned_restores( return; } - let mut names = std::collections::BTreeSet::new(); - let sections: [&[(String, Vec)]; 6] = [ - &merged.documents, - &merged.indexes, - &merged.vectors, - &merged.kv_tables, - &merged.timeseries, - &merged.edges, - ]; - for section in sections { - for (key, _) in section { - if let Some(name) = collection_from_key(key) { - names.insert(name.to_string()); - } - } - } - - for name in &names { - let Some(purge_lsn) = tombstones.purge_lsn(DatabaseId::DEFAULT.as_u64(), tenant_id, name) + for name in &restored_collection_names(tenant_id, target, merged) { + let Some(purge_lsn) = + tombstones.purge_lsn(CollectionKey::from_bare(target.dest, name), tenant_id) else { continue; }; @@ -79,6 +81,7 @@ pub(super) fn warn_on_tombstoned_restores( } tracing::warn!( tenant_id, + database_id = target.dest.as_u64(), collection = %name, purge_lsn, snapshot_watermark, @@ -89,39 +92,111 @@ pub(super) fn warn_on_tombstoned_restores( Some(TenantId::new(tenant_id)), "__restore", &format!( - "restore resurrected tombstoned collection '{name}' \ - (purge_lsn={purge_lsn}, snapshot_watermark={snapshot_watermark})" + "restore resurrected tombstoned collection '{name}' in database {} \ + (purge_lsn={purge_lsn}, snapshot_watermark={snapshot_watermark})", + target.dest.as_u64() ), ); } } -fn collection_from_key(key: &str) -> Option<&str> { - let tail = key.split_once(':')?.1; - tail.split([':', '\0']).next() +/// The bare catalog name of every collection the backup restores rows into +/// for one database, read from each section's key in that section's own +/// format. A name that does not resolve in the source database is skipped +/// here: the re-issue of its section refuses it. +fn restored_collection_names( + tenant_id: u64, + target: DatabaseTarget, + merged: &TenantDataSnapshot, +) -> BTreeSet { + let mut stored: Vec<&str> = Vec::new(); + let db_tenant_scoped: [&[(String, Vec)]; 5] = [ + &merged.documents, + &merged.documents_versioned, + &merged.indexes, + &merged.vectors, + &merged.timeseries, + ]; + for section in db_tenant_scoped { + for (key, _) in section { + stored.extend(extract_db_tenant_scoped_collection(key, tenant_id)); + } + } + for (key, _) in merged.kv_tables.iter().chain(&merged.columnar_engines) { + stored.extend(extract_db_scoped_collection(key, tenant_id)); + } + for (key, _) in &merged.edges { + if let Some((name, ..)) = parse_versioned_edge_key(key) { + stored.push(name); + } + } + stored + .into_iter() + .filter_map(|name| target.resolve(name).ok()) + .map(|name| name.bare) + .collect() } #[cfg(test)] -mod collection_key_tests { - use super::collection_from_key; +mod collection_name_tests { + use super::*; + use crate::types::DatabaseId; - #[test] - fn extracts_collection_with_colon_separator() { - assert_eq!(collection_from_key("1:users:doc-1"), Some("users")); - } + const DEFAULT_TARGET: DatabaseTarget = DatabaseTarget { + source: DatabaseId::DEFAULT, + dest: DatabaseId::DEFAULT, + }; #[test] - fn extracts_collection_with_null_separator() { - assert_eq!(collection_from_key("1:src\0label\0"), Some("src")); + fn every_section_names_its_collection() { + let snap = TenantDataSnapshot { + documents: vec![("0:7:users:0000002a".into(), vec![])], + documents_versioned: vec![( + "0:7:ledger:0000002a\x0000000000000000000001".into(), + vec![], + )], + vectors: vec![("0:7:embeddings".into(), vec![])], + kv_tables: vec![("0:7:sessions".into(), vec![])], + edges: vec![( + "follows\x00a\x00L\x00b\x0000000000000000000001".into(), + vec![], + )], + ..Default::default() + }; + let names: Vec = restored_collection_names(7, DEFAULT_TARGET, &snap) + .into_iter() + .collect(); + assert_eq!( + names, + vec!["embeddings", "follows", "ledger", "sessions", "users"] + ); } + /// A named database's sections name qualified collections. The warning + /// reads the bare names the destination catalog keys its tombstones by. #[test] - fn vector_and_kv_key_shapes() { - assert_eq!(collection_from_key("1:events"), Some("events")); + fn a_named_database_yields_bare_names() { + let target = DatabaseTarget { + source: DatabaseId::new(1025), + dest: DatabaseId::new(1030), + }; + let snap = TenantDataSnapshot { + documents: vec![("1025:7:1025/users:0000002a".into(), vec![])], + kv_tables: vec![("1025:7:1025/sessions".into(), vec![])], + ..Default::default() + }; + let names: Vec = restored_collection_names(7, target, &snap) + .into_iter() + .collect(); + assert_eq!(names, vec!["sessions", "users"]); } #[test] - fn no_tenant_prefix_returns_none() { - assert_eq!(collection_from_key("no_colon"), None); + fn another_tenants_key_names_nothing() { + let snap = TenantDataSnapshot { + documents: vec![("0:8:users:0000002a".into(), vec![])], + ..Default::default() + }; + assert!(restored_collection_names(7, DEFAULT_TARGET, &snap).is_empty()); } } diff --git a/nodedb/src/control/backup/restore/orchestrate/reissue.rs b/nodedb/src/control/backup/restore/orchestrate/reissue.rs index 013647bbb..6bb6f4b5e 100644 --- a/nodedb/src/control/backup/restore/orchestrate/reissue.rs +++ b/nodedb/src/control/backup/restore/orchestrate/reissue.rs @@ -1,9 +1,12 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Durable re-issue of columnar, timeseries, and vector rows drained from the -//! snapshot before the topology split (see [`super::restore_tenant`]'s -//! doc comments on why these engines bypass the per-node snapshot -//! install path). +//! Durable re-issue of columnar, timeseries, and vector rows drained from one +//! database's merged backup snapshot (see [`super::restore_tenant`]). +//! +//! Every section key names a collection as the source Data Plane stored it. +//! Each re-issue resolves it through the [`DatabaseTarget`]: the plan names +//! the destination-qualified collection, and the write routes by the bare +//! name in the destination database. use std::sync::Arc; @@ -14,7 +17,7 @@ use crate::control::state::SharedState; use crate::engine::vector::index_config::IndexConfig; use crate::types::TenantId; -use crate::control::backup::snapshot_keys::extract_db_scoped_collection; +use super::super::target::DatabaseTarget; /// Decode and durably re-issue every restored timeseries collection. /// @@ -27,6 +30,7 @@ use crate::control::backup::snapshot_keys::extract_db_scoped_collection; pub(super) async fn reissue_timeseries_snapshots( state: &Arc, tenant_id: u64, + target: DatabaseTarget, memtables: Vec<(String, Vec)>, flushed: Vec, ) -> Result { @@ -34,7 +38,6 @@ pub(super) async fn reissue_timeseries_snapshots( // the same key). Absent when at-rest encryption is not configured, in which // case segments are plaintext and decode with `kek = None`. let kek = state.wal.encryption_key().cloned(); - let database_id = crate::types::DatabaseId::DEFAULT; // Index memtable bytes and flushed blobs by their `{db}:{tid}:{collection}` // key so each collection is decoded + re-issued exactly once. @@ -58,18 +61,13 @@ pub(super) async fn reissue_timeseries_snapshots( let empty_flushed = crate::types::TsFlushedCollectionBlob::default(); let mut reissued = 0usize; for key in keys_in_order { - let Some(collection) = extract_db_scoped_collection(&key, tenant_id) else { - return Err(Error::Internal { - detail: format!("restore reissue: malformed timeseries snapshot key '{key}'"), - }); - }; - let collection = collection.to_owned(); + let name = target.resolve_scoped(&key, tenant_id)?; let memtable_bytes = memtable_by_key.remove(&key); let flushed_blob = flushed_by_key.get(&key).unwrap_or(&empty_flushed); let rows = super::super::timeseries_reissue::decode_timeseries_live_rows( - &collection, + &name.bare, memtable_bytes.as_deref(), flushed_blob, kek.as_ref(), @@ -78,13 +76,15 @@ pub(super) async fn reissue_timeseries_snapshots( continue; } - let plan = - super::super::timeseries_reissue::build_timeseries_ingest_plan(&collection, rows)?; - super::super::timeseries_reissue::reissue_timeseries_durably( + let plan = super::super::timeseries_reissue::build_timeseries_ingest_plan( + name.stored.as_str(), + rows, + )?; + super::super::durable::reissue_plan_durably( state, TenantId::new(tenant_id), - database_id, - &collection, + target.dest, + &name.bare, plan, ) .await?; @@ -101,6 +101,7 @@ pub(super) async fn reissue_timeseries_snapshots( pub(super) async fn reissue_columnar_snapshots( state: &Arc, tenant_id: u64, + target: DatabaseTarget, entries: Vec<(String, Vec)>, ) -> Result { // Columnar segment KEK == the WAL encryption key (segments are written via @@ -108,27 +109,22 @@ pub(super) async fn reissue_columnar_snapshots( // key). Absent when at-rest encryption is not configured, in which case // segments are plaintext NDBS and decode with `kek = None`. let kek = state.wal.encryption_key().cloned(); - let database_id = crate::types::DatabaseId::DEFAULT; let mut reissued = 0usize; for (key, bytes) in entries { - let Some(collection) = extract_db_scoped_collection(&key, tenant_id) else { - return Err(Error::Internal { - detail: format!("restore reissue: malformed columnar snapshot key '{key}'"), - }); - }; - let collection = collection.to_owned(); + let name = target.resolve_scoped(&key, tenant_id)?; let snap: nodedb_columnar::ColumnarEngineSnapshot = zerompk::from_msgpack(&bytes).map_err(|e| Error::Serialization { format: "msgpack".into(), detail: format!( - "restore reissue: deserialize ColumnarEngineSnapshot for '{collection}': {e}" + "restore reissue: deserialize ColumnarEngineSnapshot for '{}': {e}", + name.bare ), })?; let decoded = super::super::columnar_reissue::decode_snapshot_live_rows( - &collection, + &name.bare, snap, kek.as_ref(), )?; @@ -136,13 +132,15 @@ pub(super) async fn reissue_columnar_snapshots( continue; } - let plan = - super::super::columnar_reissue::build_columnar_insert_plan(&collection, decoded)?; - super::super::columnar_reissue::reissue_columnar_durably( + let plan = super::super::columnar_reissue::build_columnar_insert_plan( + name.stored.as_str(), + decoded, + )?; + super::super::durable::reissue_plan_durably( state, TenantId::new(tenant_id), - database_id, - &collection, + target.dest, + &name.bare, plan, ) .await?; @@ -168,27 +166,23 @@ pub(super) async fn reissue_columnar_snapshots( pub(super) async fn reissue_vector_snapshots( state: &Arc, tenant_id: u64, + target: DatabaseTarget, entries: Vec<(String, Vec)>, ) -> Result { - let database_id = crate::types::DatabaseId::DEFAULT; - let mut reissued = 0usize; for (key, bytes) in entries { - let Some(coll_key) = extract_db_scoped_collection(&key, tenant_id) else { - return Err(Error::Internal { - detail: format!("restore reissue: malformed vector snapshot key '{key}'"), - }); - }; + let coll_key = target.scoped_rest(&key, tenant_id)?; let (collection, field_name) = super::super::vector_reissue::split_vector_coll_key(coll_key); - let collection = collection.to_owned(); + let name = target.resolve(collection)?; let field_name = field_name.to_owned(); let vectors: Vec<(u32, Vec, Option)> = zerompk::from_msgpack(&bytes) .map_err(|e| Error::Serialization { format: "msgpack".into(), detail: format!( - "restore reissue: deserialize vector snapshot for '{collection}': {e}" + "restore reissue: deserialize vector snapshot for '{}': {e}", + name.bare ), })?; if vectors.is_empty() { @@ -198,16 +192,16 @@ pub(super) async fn reissue_vector_snapshots( for (_node_id, vector, surrogate) in vectors { let surrogate = surrogate.unwrap_or(Surrogate::ZERO); let plan = super::super::vector_reissue::build_vector_insert_plan( - &collection, + name.stored.as_str(), &field_name, vector, surrogate, ); - super::super::vector_reissue::reissue_vector_durably( + super::super::durable::reissue_plan_durably( state, TenantId::new(tenant_id), - database_id, - &collection, + target.dest, + &name.bare, plan, ) .await?; @@ -240,22 +234,16 @@ pub(super) async fn reissue_vector_snapshots( pub(super) async fn reissue_vector_params( state: &Arc, tenant_id: u64, + target: DatabaseTarget, params: Vec<(String, Vec)>, index_configs: Vec<(String, Vec)>, ) -> Result { - let database_id = crate::types::DatabaseId::DEFAULT; - let mut resolved: std::collections::HashMap = std::collections::HashMap::new(); let mut keys_in_order: Vec = Vec::new(); for (key, bytes) in index_configs { - let Some(coll_key) = extract_db_scoped_collection(&key, tenant_id) else { - return Err(Error::Internal { - detail: format!("restore reissue: malformed index_configs snapshot key '{key}'"), - }); - }; - let coll_key = coll_key.to_owned(); + let coll_key = target.scoped_rest(&key, tenant_id)?.to_owned(); let cfg: IndexConfig = zerompk::from_msgpack(&bytes).map_err(|e| Error::Serialization { format: "msgpack".into(), detail: format!("restore reissue: deserialize IndexConfig for '{coll_key}': {e}"), @@ -265,12 +253,7 @@ pub(super) async fn reissue_vector_params( } for (key, bytes) in params { - let Some(coll_key) = extract_db_scoped_collection(&key, tenant_id) else { - return Err(Error::Internal { - detail: format!("restore reissue: malformed vector_params snapshot key '{key}'"), - }); - }; - let coll_key = coll_key.to_owned(); + let coll_key = target.scoped_rest(&key, tenant_id)?.to_owned(); if resolved.contains_key(&coll_key) { // Superseded by a full IndexConfig entry for the same (collection, // field) — the two sections always describe the same DDL state, @@ -300,19 +283,18 @@ pub(super) async fn reissue_vector_params( }; let (collection, field_name) = super::super::vector_reissue::split_vector_coll_key(&coll_key); - let collection = collection.to_owned(); - let field_name = field_name.to_owned(); + let name = target.resolve(collection)?; let plan = super::super::vector_reissue::build_vector_set_params_plan( - &collection, - &field_name, + name.stored.as_str(), + field_name, &config, ); - super::super::vector_reissue::reissue_vector_durably( + super::super::durable::reissue_plan_durably( state, TenantId::new(tenant_id), - database_id, - &collection, + target.dest, + &name.bare, plan, ) .await?; diff --git a/nodedb/src/control/backup/restore/orchestrate/restore.rs b/nodedb/src/control/backup/restore/orchestrate/restore.rs index 0391e3f87..9aa51fda5 100644 --- a/nodedb/src/control/backup/restore/orchestrate/restore.rs +++ b/nodedb/src/control/backup/restore/orchestrate/restore.rs @@ -1,8 +1,9 @@ // SPDX-License-Identifier: BUSL-1.1 -//! `restore_tenant`: validates a backup envelope, merges all sections into -//! a single `TenantDataSnapshot`, splits it by current cluster topology, and -//! dispatches `MetaOp::RestoreTenantSnapshot` to each owning node. +//! `restore_tenant`: validates a backup envelope, maps every backed-up +//! database to its destination, merges the sections of each database into +//! one `TenantDataSnapshot`, and re-issues every section as durable, +//! replicated writes into its destination database. use std::sync::Arc; @@ -11,18 +12,12 @@ use nodedb_types::backup_envelope::{ }; use crate::Error; -use crate::bridge::envelope::PhysicalPlan; use crate::control::server::shared::ddl::neutral::collection::dispatch_register_from_stored; -use crate::control::server::shared::ddl::sync_dispatch; use crate::control::state::SharedState; -use crate::types::TenantId; -use nodedb_physical::physical_plan::MetaOp; -use super::super::remote::{NODE_RESTORE_TIMEOUT, dispatch_remote}; +use super::super::databases::{decode_databases, resolve_databases}; use super::super::sections::{apply_metadata_sections, merge_sections}; -use super::super::topology::{SplitOutput, is_self, split_by_current_topology}; use super::rebind; -use super::reissue; use super::stats::RestoreStats; /// Restore a tenant from a fully-buffered backup envelope. @@ -51,14 +46,27 @@ pub async fn restore_tenant( .into()); } - if !dry_run && env.meta.snapshot_watermark != 0 { - let current_high_water = state.tenant_write_hlc(tenant_id); + // Every group the restore reads or writes has a reachable majority, or + // the restore fails here, before it proposes anything. + if !dry_run { + super::super::quorum::require_quorum(state)?; + } + + let newest = if !dry_run && env.meta.snapshot_watermark != 0 { + super::super::guard::newest_committed_write(state, tenant_id).await? + } else { + None + }; + if let Some(mark) = newest { + let current_high_water = mark.hlc; if env.meta.snapshot_watermark < current_high_water { if force { tracing::warn!( tenant_id, envelope_watermark = env.meta.snapshot_watermark, current_high_water, + newest_write_site = mark.site.as_str(), + newest_write_collection = mark.collection.as_deref().unwrap_or(""), "restore staleness protection explicitly overridden via FORCE: \ envelope watermark is older than the destination cluster's last \ observed write-HLC for this tenant — newer writes will be overwritten" @@ -68,8 +76,13 @@ pub async fn restore_tenant( detail: format!( "restore refused: envelope watermark {} is older than the \ destination cluster's last observed write-HLC {} for tenant \ - {} — newer writes would be silently overwritten", - env.meta.snapshot_watermark, current_high_water, tenant_id + {} (newest write: {} on collection '{}') — newer writes would \ + be silently overwritten", + env.meta.snapshot_watermark, + current_high_water, + tenant_id, + mark.site, + mark.collection.as_deref().unwrap_or(""), ), }); } @@ -84,8 +97,15 @@ pub async fn restore_tenant( ..Default::default() }; + // Map every backed-up database to its destination, creating each one the + // destination lacks. Every other section names its database by source id. + let database_blobs = decode_databases(&env)?; + let databases = resolve_databases(state, tenant_id, &database_blobs, dry_run)?; + stats.databases = database_blobs.len(); + stats.databases_created = databases.created(); + if !dry_run { - let restored_collections = apply_metadata_sections(state, tenant_id, &env)?; + let restored_collections = apply_metadata_sections(state, tenant_id, &env, &databases)?; // Every restored collection's declaration reaches this node's Data // Plane before any of its rows do. The catalog row alone leaves // `doc_configs` empty for the collection, and the re-issue below @@ -97,219 +117,65 @@ pub async fn restore_tenant( // applier's own register hook and a later boot seed are both // idempotent with it. A registration failure fails the restore. for coll in &restored_collections { + // A classified error keeps its class. Only a machinery failure + // gains the restore context. dispatch_register_from_stored(state, coll) .await - .map_err(|e| Error::Internal { - detail: format!( - "restore: Data Plane registration of collection '{}' failed: {e}", - coll.name - ), + .map_err(|e| { + if crate::error_classify::is_unclassified_failure(&e) { + Error::Internal { + detail: format!( + "restore: Data Plane registration of collection '{}' failed: {e}", + coll.name + ), + } + } else { + e + } })?; } } - let mut merged = merge_sections(&env.sections)?; - stats.documents = merged.documents.len(); - stats.indexes = merged.indexes.len(); - stats.edges = merged.edges.len(); - stats.vectors = merged.vectors.len(); - stats.kv_tables = merged.kv_tables.len(); - // CRDT state is one entry per (tenant, collection). - stats.crdt_state = merged.crdt_state.len(); - stats.timeseries = merged.timeseries.len(); - stats.flushed_ts_segments = merged.flushed_ts_segments.len(); - stats.surrogate_pk = merged.surrogate_pk.len(); - - rebind::warn_on_tombstoned_restores(state, tenant_id, &merged, env.meta.snapshot_watermark); + let merged = merge_sections(&env.sections)?; + for source in merged.keys() { + if !database_blobs + .iter() + .any(|blob| blob.database_id == *source) + { + return Err(Error::Internal { + detail: format!( + "invalid backup format: a data section names database {source}, which the \ + backup's database section does not list" + ), + }); + } + } + for (source, snap) in &merged { + stats.count_sections(snap); + if dry_run { + stats.columnar_engines += snap.columnar_engines.len(); + } + // A dry run has no target for a database this cluster lacks: no + // tombstone of this cluster names it. + if let Some(target) = databases.get(*source) { + rebind::warn_on_tombstoned_restores( + state, + tenant_id, + target, + snap, + env.meta.snapshot_watermark, + ); + } + } if dry_run { - stats.columnar_engines = merged.columnar_engines.len(); return Ok(stats); } - // Plain-columnar engine state is NOT installed via the snapshot path (that - // lands in in-memory-only Data Plane maps — lost on restart, never - // replicated). Drain it here and re-issue durably below as - // `ColumnarOp::Insert`s. The topology split must therefore never see - // columnar engines. - let columnar_snapshots = std::mem::take(&mut merged.columnar_engines); - - // Timeseries engine state (memtable section + flushed on-disk segments) is - // likewise NOT installed via the snapshot path — `restore_timeseries` and - // `restore_flushed_ts_segments` do a per-node DIRECT install that is never - // Raft-replicated, so on a multi-replica cluster the data lands on only one - // node. Drain both sections here and re-issue durably below as - // `TimeseriesOp::Ingest`s (Raft-replicated in cluster mode; WAL-appended - // then installed in single-node mode). The topology split must therefore - // never see timeseries data — otherwise it would be double-installed. - let timeseries_memtables = std::mem::take(&mut merged.timeseries); - let flushed_ts_segments = std::mem::take(&mut merged.flushed_ts_segments); - - // CRDT state is NOT installed via the per-node snapshot fan-out: that - // dispatch is race-prone (skips data groups with no leader yet) and not - // durable across restart. Drain the per-collection CRDT section here and - // re-issue durably below as `CrdtOp::ImportSnapshot` (Raft-replicated in - // cluster mode; WAL-appended then installed in single-node mode). The - // topology split must therefore never see CRDT state — otherwise the - // coordinator would double-import. - let crdt_state = std::mem::take(&mut merged.crdt_state); - - // Vector engine state is likewise NOT installed via the snapshot path — - // `restore_vector_collection` installs straight into the in-memory-only - // `vector_collections` Data Plane map with no WAL record and no Raft - // entry, so it is lost on restart (single-node) and never replicated - // (cluster). Drain it here and re-issue durably below, one - // `VectorOp::Insert` per restored vector (Raft-replicated in cluster - // mode; WAL-appended then installed in single-node mode). The topology - // split must therefore never see vector data — otherwise it would be - // double-installed. - let vector_snapshots = std::mem::take(&mut merged.vectors); - - // Vector-index HNSW/PQ/IVF configuration (metric, M, ef_construction, - // quantization/index_type) is captured at backup alongside the raw - // vectors above (see `TenantDataSnapshot::vector_params` / - // `::index_configs` doc comments) but is likewise NOT installed via the - // snapshot path. Drain both here and re-issue durably below as - // `VectorOp::SetParams` — BEFORE the vector `Insert` re-issue, since - // `get_or_create_vector_index` lazily creates the Data Plane HNSW index - // from `self.vector_params` on the first `Insert` it sees for a - // (collection, field), defaulting silently if no `SetParams` landed - // first. The topology split must therefore never see these sections. - let vector_params_snapshots = std::mem::take(&mut merged.vector_params); - let index_config_snapshots = std::mem::take(&mut merged.index_configs); - - // Drain the PK→surrogate identity map before the topology split (the split - // only routes per-key engine data). It is rebound into the destination - // catalog after the data install dispatches succeed — without it restored - // documents are unreachable by PK point-lookup (`WHERE id=`). - let surrogate_binds = std::mem::take(&mut merged.surrogate_pk); - - let SplitOutput { - buckets, - malformed_keys, - route_fallbacks, - } = split_by_current_topology(state, tenant_id, merged); - stats.nodes_dispatched = buckets.len(); - stats.malformed_keys = malformed_keys; - stats.route_fallbacks = route_fallbacks; - if malformed_keys > 0 { - tracing::warn!( - tenant_id, - count = malformed_keys, - "restore: snapshot contained keys that did not parse — possible corruption" - ); - } - if route_fallbacks > 0 { - tracing::warn!( - tenant_id, - count = route_fallbacks, - "restore: routed some entries to local node because no current leader was visible" - ); - } - - let mut local_plan: Option = None; - let mut remote_futs = Vec::with_capacity(buckets.len()); - for (node_id, sub) in buckets { - let payload = zerompk::to_msgpack_vec(&sub).map_err(|e| Error::Internal { - detail: format!("restore: snapshot encode failed: {e}"), - })?; - let plan = PhysicalPlan::Meta(MetaOp::RestoreTenantSnapshot { - tenant_id, - snapshot: payload, - // User RESTORE keeps the fail-closed collision behavior. - replace_mode: false, - clear_vshards: Vec::new(), - collections_to_clear: Vec::new(), - }); - if is_self(state, node_id) { - local_plan = Some(plan); - } else { - let state = state.clone(); - remote_futs - .push(async move { dispatch_remote(&state, node_id, tenant_id, plan).await }); - } - } - if let Some(plan) = local_plan { - sync_dispatch::dispatch_system( - state, - sync_dispatch::SystemTask::new( - sync_dispatch::SystemReason::BackupRestore, - TenantId::new(tenant_id), - // TODO(A8-followup): backup/restore not yet multi-database. - crate::types::DatabaseId::DEFAULT, - "__system", - plan, - ), - NODE_RESTORE_TIMEOUT, - ) - .await?; - } - let results = futures::future::join_all(remote_futs).await; - if let Some(first_err) = results.into_iter().find_map(Result::err) { - return Err(first_err); + // Each database re-issues its rows into its destination database. + for (source, snap) in merged { + let target = databases.target(source)?; + super::database::reissue_database(state, tenant_id, target, snap, &mut stats).await?; } - - // Rebind the PK→surrogate identity map into the destination catalog now - // that the data is installed. The catalog is the SOURCE OF TRUTH the - // planner consults for PK point-lookups (`surrogate_assigner.lookup(pk)`); - // a missing binding makes a restored row unreachable by PK even though it - // is present in the doc store. A rebind failure is FATAL — silently - // shipping unqueryable rows is the partial-success anti-pattern this - // codebase forbids. - rebind::rebind_surrogates(state, surrogate_binds)?; - - // Durable re-issue of plain-columnar rows. Each restored collection's live - // rows are decoded from the snapshot and replayed as a durable - // `ColumnarOp::Insert` (Raft-replicated in cluster mode; WAL-appended then - // installed in single-node mode). Collections that decode to zero live rows - // are skipped. Any failure is fatal — no warn-and-continue. - stats.columnar_engines = - reissue::reissue_columnar_snapshots(state, tenant_id, columnar_snapshots).await?; - - // Durable re-issue of timeseries rows. Each restored collection's memtable - // rows plus every flushed partition's rows are decoded from the snapshot and - // replayed as a durable `TimeseriesOp::Ingest` (Raft-replicated in cluster - // mode; WAL-appended then installed in single-node mode). Collections that - // decode to zero live rows are skipped. Any failure is fatal — no - // warn-and-continue. - stats.timeseries_reissued = reissue::reissue_timeseries_snapshots( - state, - tenant_id, - timeseries_memtables, - flushed_ts_segments, - ) - .await?; - - // Durable re-issue of CRDT state. Each collection's Loro snapshot is - // proposed through Raft to the data group owning that collection's vshard - // (Raft-replicated in cluster mode; WAL-appended then installed in - // single-node mode). Every replica applies the same idempotent Loro merge - // and converges deterministically. Any failure is fatal — no - // warn-and-continue. - stats.crdt_reissued = - super::super::crdt_reissue::reissue_crdt_snapshots(state, crdt_state).await?; - - // Durable re-issue of vector-index configuration. Each restored - // (collection, field) HNSW/PQ/IVF config is replayed as a - // `VectorOp::SetParams` (Raft-replicated in cluster mode; WAL-appended - // then installed in single-node mode). MUST run before the vector-insert - // re-issue below — see the `vector_params_snapshots` drain comment - // above. Any failure is fatal — no warn-and-continue. - stats.vector_params_reissued = reissue::reissue_vector_params( - state, - tenant_id, - vector_params_snapshots, - index_config_snapshots, - ) - .await?; - - // Durable re-issue of vector rows. Each restored vector is replayed as an - // individual `VectorOp::Insert` (Raft-replicated in cluster mode; - // WAL-appended then installed in single-node mode). Collections that - // decode to zero vectors are skipped. Any failure is fatal — no - // warn-and-continue. - stats.vectors_reissued = - reissue::reissue_vector_snapshots(state, tenant_id, vector_snapshots).await?; - Ok(stats) } diff --git a/nodedb/src/control/backup/restore/orchestrate/stats.rs b/nodedb/src/control/backup/restore/orchestrate/stats.rs index 5a72d210e..fc9559539 100644 --- a/nodedb/src/control/backup/restore/orchestrate/stats.rs +++ b/nodedb/src/control/backup/restore/orchestrate/stats.rs @@ -4,12 +4,18 @@ use serde::Serialize; +use crate::types::TenantDataSnapshot; + /// Aggregate stats returned to the client at the end of a restore. #[derive(Debug, Default, Clone, Serialize)] pub struct RestoreStats { pub tenant_id: u64, pub dry_run: bool, pub sections: u16, + /// Number of databases the backup covers. + pub databases: usize, + /// Number of those databases the restore created on this cluster. + pub databases_created: usize, pub source_vshard_count: u16, pub documents: usize, pub indexes: usize, @@ -27,14 +33,36 @@ pub struct RestoreStats { pub crdt_reissued: usize, /// Number of individual vectors re-issued durably (Raft/WAL) on restore. pub vectors_reissued: usize, + /// Number of individual KV rows re-issued durably (Raft/WAL) on restore. + pub kv_reissued: usize, /// Number of (collection, field) vector-index HNSW/PQ/IVF configs /// re-issued durably (Raft/WAL) on restore. pub vector_params_reissued: usize, /// Number of PK→surrogate identity bindings rebound into the catalog. pub surrogate_pk: usize, - pub nodes_dispatched: usize, - /// Non-zero = snapshot contained unparseable keys (possible corruption). - pub malformed_keys: usize, - /// Non-zero = some entries were routed to local node due to missing shard leader. - pub route_fallbacks: usize, + /// Document sub-records re-issued: one per current row, one per version + /// of a `bitemporal=true` row. + pub documents_reissued: usize, + /// Edge versions re-issued. + pub edges_reissued: usize, + /// Redo records the document and edge re-issue committed. + pub redo_records: usize, +} + +impl RestoreStats { + /// Add the section sizes of one database's merged snapshot. The + /// columnar count is the number re-issued, so a restore adds it as it + /// re-issues and a dry run adds the section size. + pub fn count_sections(&mut self, snap: &TenantDataSnapshot) { + self.documents += snap.documents.len() + snap.documents_versioned.len(); + self.indexes += snap.indexes.len() + snap.indexes_versioned.len(); + self.edges += snap.edges.len(); + self.vectors += snap.vectors.len(); + self.kv_tables += snap.kv_tables.len(); + // CRDT state is one entry per (tenant, collection). + self.crdt_state += snap.crdt_state.len(); + self.timeseries += snap.timeseries.len(); + self.flushed_ts_segments += snap.flushed_ts_segments.len(); + self.surrogate_pk += snap.surrogate_pk.len(); + } } diff --git a/nodedb/src/control/backup/restore/quorum.rs b/nodedb/src/control/backup/restore/quorum.rs new file mode 100644 index 000000000..506759f21 --- /dev/null +++ b/nodedb/src/control/backup/restore/quorum.rs @@ -0,0 +1,110 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! RESTORE's quorum check: every Raft group the restore reads or writes has a +//! reachable majority before the restore proposes anything. +//! +//! A restore reads every data group's write marks, writes catalog rows +//! through the metadata group, places a cut marker in the sequencer group, +//! and re-issues rows through the data groups. A group with no reachable +//! majority commits nothing, so every step against it waits out its +//! deadline. Checked first, the restore fails at once with the group and the +//! nodes it cannot reach, and nothing of it is applied anywhere. +//! +//! The check reads this node's membership view: the routing table's voters +//! and the topology's active nodes. A majority lost after the check fails the +//! step that needs it at its deadline instead. + +use std::collections::BTreeSet; + +use crate::Error; +use crate::control::state::SharedState; + +/// Fail with [`Error::GroupQuorumUnavailable`] for the first Raft group whose +/// voters have no reachable majority. A node with no cluster routing has no +/// group to check. +pub(super) fn require_quorum(state: &SharedState) -> Result<(), Error> { + let (Some(routing), Some(topology)) = ( + state.cluster_routing.as_ref(), + state.cluster_topology.as_ref(), + ) else { + return Ok(()); + }; + let active: BTreeSet = topology + .read() + .unwrap_or_else(|p| p.into_inner()) + .active_nodes() + .iter() + .map(|node| node.node_id) + .collect(); + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + let mut group_ids = routing.group_ids(); + group_ids.sort_unstable(); + for group_id in group_ids { + let Some(info) = routing.group_info(group_id) else { + continue; + }; + if let Some(error) = quorum_error(group_id, &info.members, &active) { + return Err(error); + } + } + Ok(()) +} + +/// The error for `group_id` when fewer than a majority of `voters` are in +/// `active`, else `None`. +fn quorum_error(group_id: u64, voters: &[u64], active: &BTreeSet) -> Option { + if voters.is_empty() { + return None; + } + let mut unreachable: Vec = voters + .iter() + .copied() + .filter(|voter| !active.contains(voter)) + .collect(); + unreachable.sort_unstable(); + let reachable = voters.len() - unreachable.len(); + if reachable * 2 > voters.len() { + return None; + } + let mut voters = voters.to_vec(); + voters.sort_unstable(); + Some(Error::GroupQuorumUnavailable { + group_id, + voters, + unreachable, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_group_with_a_reachable_majority_passes() { + let active = BTreeSet::from([1, 2]); + assert!(quorum_error(4, &[1, 2, 3], &active).is_none()); + } + + #[test] + fn a_group_without_a_reachable_majority_names_its_unreachable_voters() { + let active = BTreeSet::from([1]); + match quorum_error(4, &[3, 1, 2], &active) { + Some(Error::GroupQuorumUnavailable { + group_id, + voters, + unreachable, + }) => { + assert_eq!(group_id, 4); + assert_eq!(voters, vec![1, 2, 3]); + assert_eq!(unreachable, vec![2, 3]); + } + other => panic!("expected GroupQuorumUnavailable, got {other:?}"), + } + } + + #[test] + fn half_of_an_even_voter_set_is_not_a_majority() { + let active = BTreeSet::from([1, 2]); + assert!(quorum_error(4, &[1, 2, 3, 4], &active).is_some()); + } +} diff --git a/nodedb/src/control/backup/restore/redo_reissue/commit.rs b/nodedb/src/control/backup/restore/redo_reissue/commit.rs new file mode 100644 index 000000000..eae5c35bb --- /dev/null +++ b/nodedb/src/control/backup/restore/redo_reissue/commit.rs @@ -0,0 +1,241 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Commit restored units as redo records through their vShard's apply log. +//! +//! A restored row installs exactly as a committed transaction's row does. With +//! Raft the record is proposed to the collection's data group, and every +//! replica binds its identities, appends it to its own WAL and installs it. +//! With no Raft this node runs the same apply alone. Either way the call +//! returns once the record is durable and installed here. + +use std::collections::HashSet; + +use nodedb_physical::physical_plan::RedoOrigin; + +use crate::bridge::envelope::{ErrorCode, Status}; +use crate::control::state::SharedState; +use crate::control::surrogate::CarriedIdentity; +use crate::control::wal_replication::encode::transaction_redo_entry; +use crate::control::wal_replication::propose_replicated_entry; +use crate::control::wal_replication::transaction_redo::{ + RedoTarget, TransactionRedoPayload, apply_transaction_redo, +}; +use crate::event::EventSource; +use crate::types::TenantId; +use crate::wal::{RedoRecord, RedoSubRecord}; + +use super::units::{CollectionUnits, RowUnit}; + +/// Most sub-records one restore record carries. +const MAX_OPS_PER_RECORD: usize = 512; + +/// Most encoded bytes one restore record carries. A single unit larger than +/// this still commits, alone in its own record. +const MAX_BYTES_PER_RECORD: usize = 4 * 1024 * 1024; + +/// Split `units` into record-sized batches, in order. A unit never splits. +fn batch_units(units: Vec) -> Vec> { + let mut batches = Vec::new(); + let mut current: Vec = Vec::new(); + let (mut ops, mut bytes) = (0usize, 0usize); + for unit in units { + let (unit_ops, unit_bytes) = (unit.ops.len(), unit.byte_len()); + if !current.is_empty() + && (ops + unit_ops > MAX_OPS_PER_RECORD || bytes + unit_bytes > MAX_BYTES_PER_RECORD) + { + batches.push(std::mem::take(&mut current)); + (ops, bytes) = (0, 0); + } + ops += unit_ops; + bytes += unit_bytes; + current.push(unit); + } + if !current.is_empty() { + batches.push(current); + } + batches +} + +/// One batch as the payload every replica applies. `stored` is the name the +/// Data Plane stores the collection under, as a committed transaction's +/// written-collection list names it. +fn batch_payload(stored: &str, batch: Vec) -> TransactionRedoPayload { + let mut ops: Vec = Vec::new(); + let mut identities: Vec = Vec::new(); + let mut seen: HashSet<(String, Vec)> = HashSet::new(); + for unit in batch { + ops.extend(unit.ops); + for identity in unit.identities { + if seen.insert((identity.collection.clone(), identity.pk_bytes.clone())) { + identities.push(identity); + } + } + } + TransactionRedoPayload { + redo: RedoRecord { + version: 1, + ops, + calvin_stamp: None, + }, + collections: vec![stored.to_string()], + // The backup holds every target row with its total already folded in. + sum_targets: Vec::new(), + identities, + // Every replica applies the rows as restored: AFTER triggers fired + // when the rows were first written, and do not fire again. + event_source: EventSource::Restore, + origin: RedoOrigin::Restore, + } +} + +/// Commit one record and wait until it is durable and installed here. +async fn commit_record( + state: &SharedState, + target: RedoTarget, + payload: &TransactionRedoPayload, +) -> crate::Result<()> { + super::super::durable::log_reissue_step( + state, + "redo", + payload.collections.first().map_or("", String::as_str), + target.vshard_id, + payload.redo.ops.len(), + ); + if let Some(proposer) = state.async_raft_proposer() { + let entry = transaction_redo_entry( + target.tenant_id, + target.database_id, + target.vshard_id, + payload, + ); + propose_replicated_entry(state, proposer, entry).await?; + return Ok(()); + } + let outcome = apply_transaction_redo(state, target, payload, 0, None).await?; + if outcome.response.status == Status::Ok { + return Ok(()); + } + Err(crate::Error::DataPlane( + outcome + .response + .error_code + .as_deref() + .cloned() + .unwrap_or_else(|| ErrorCode::Internal { + detail: "restore redo apply returned an error status with no error code".into(), + }), + )) +} + +/// Commit every unit of `units` in order. Returns the records committed. +pub(super) async fn commit_collection( + state: &SharedState, + tenant_id: TenantId, + units: CollectionUnits, +) -> crate::Result { + let CollectionUnits { + database_id, + collection, + units, + } = units; + let target = RedoTarget { + tenant_id, + database_id, + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(), + }; + let stored = nodedb_types::QualifiedCollection::new(database_id, &collection); + let mut records = 0usize; + for batch in batch_units(units) { + let payload = batch_payload(stored.as_str(), batch); + // A classified error keeps its class. Only a machinery failure gains + // the restore context. + commit_record(state, target, &payload).await.map_err(|e| { + if crate::error_classify::is_unclassified_failure(&e) { + crate::Error::Internal { + detail: format!("restore: re-issuing rows of '{collection}' failed: {e}"), + } + } else { + e + } + })?; + records += 1; + } + Ok(records) +} + +#[cfg(test)] +mod tests { + use nodedb_types::Surrogate; + + use super::*; + use crate::types::VShardId; + + fn unit(ops: usize, payload_len: usize, pk: &str) -> RowUnit { + RowUnit { + ops: (0..ops) + .map(|_| RedoSubRecord { + record_type: 0, + payload: vec![0; payload_len], + }) + .collect(), + identities: vec![CarriedIdentity { + collection: "c".into(), + pk_bytes: pk.as_bytes().to_vec(), + surrogate: Surrogate::new(1), + }], + } + } + + #[test] + fn batches_cut_between_units_at_the_op_limit() { + let units = (0..3) + .map(|i| unit(MAX_OPS_PER_RECORD / 2, 1, &i.to_string())) + .collect(); + let batches = batch_units(units); + let sizes: Vec = batches.iter().map(Vec::len).collect(); + assert_eq!(sizes, vec![2, 1]); + } + + #[test] + fn an_oversized_unit_commits_alone() { + let units = vec![ + unit(1, 1, "a"), + unit(1, MAX_BYTES_PER_RECORD + 1, "b"), + unit(1, 1, "c"), + ]; + let sizes: Vec = batch_units(units).iter().map(Vec::len).collect(); + assert_eq!(sizes, vec![1, 1, 1]); + } + + #[test] + fn a_payload_carries_each_identity_once_and_restores_without_folds() { + let payload = batch_payload("c", vec![unit(1, 1, "a"), unit(1, 1, "a")]); + assert_eq!(payload.redo.ops.len(), 2); + assert_eq!(payload.identities.len(), 1); + assert!(payload.sum_targets.is_empty()); + assert_eq!(payload.origin, RedoOrigin::Restore); + } + + #[test] + fn a_restored_record_carries_the_restore_source_to_every_replica() { + let payload = batch_payload("c", vec![unit(1, 1, "a")]); + assert_eq!(payload.event_source, EventSource::Restore); + let entry = transaction_redo_entry( + TenantId::new(1), + crate::types::DatabaseId::DEFAULT, + VShardId::new(0), + &payload, + ); + assert_eq!( + crate::event::EventSource::from(entry.event_source), + EventSource::Restore + ); + match entry.write { + crate::control::wal_replication::ReplicatedWrite::TransactionRedo { + event_source, + .. + } => assert_eq!(EventSource::from(event_source), EventSource::Restore), + other => panic!("expected a transaction redo, got {other:?}"), + } + } +} diff --git a/nodedb/src/control/backup/restore/redo_reissue/documents.rs b/nodedb/src/control/backup/restore/redo_reissue/documents.rs new file mode 100644 index 000000000..0d13cd069 --- /dev/null +++ b/nodedb/src/control/backup/restore/redo_reissue/documents.rs @@ -0,0 +1,392 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Restored document rows as redo units. +//! +//! A backup carries each row in its stored form, keyed by its storage key: +//! +//! * `documents` — `"{db}:{tid}:{collection}:{storage_key}"`, the current row +//! of a collection that keeps no history; +//! * `documents_versioned` — `"{db}:{tid}:{collection}:{storage_key}\x00{sys:020}"`, +//! every version of a `bitemporal=true` row. +//! +//! `collection` is the name the source Data Plane stored the collection +//! under: database-qualified outside the default database. Each row +//! re-issues under the destination-qualified name, and binds its identity +//! under the bare name in the destination database. +//! +//! Each row becomes one unit: its sub-records in version order plus the +//! identity every replica binds before it installs them. A strict row's Binary +//! Tuple decodes back to MessagePack with the collection's schema, the same +//! conversion a transaction commit applies to a staged strict row. Every +//! replica re-derives the row's secondary index entries as it installs it. + +use std::collections::{BTreeMap, HashMap}; + +use nodedb_types::columnar::StrictSchema; +use nodedb_types::{CollectionType, DocumentMode, RowIdentity, StorageKey}; + +use crate::control::state::SharedState; +use crate::control::surrogate::CarriedIdentity; +use crate::data::executor::strict_format::{binary_tuple_to_msgpack, undecodable_strict_row}; +use crate::engine::sparse::btree_versioned::{TAG_LIVE, TAG_TOMBSTONE, decode_value}; +use crate::types::{DatabaseId, SurrogateBindEntry, TenantId}; + +use super::super::target::DatabaseTarget; +use super::sub_record::{VersionStamp, document_put, document_tombstone}; +use super::units::{CollectionUnits, RowUnit}; + +/// A row as the backup stored it. +enum StoredRow { + /// The one current body of a row with no history. + Current(Vec), + /// Every `(sys_from_ms, versioned value)` of a bitemporal row. + Versions(Vec<(i64, Vec)>), +} + +/// What decoding a collection's rows reads from its catalog entry. +struct CollectionShape { + strict: Option, + declared_primary_key: Option, +} + +fn malformed(key: &str) -> crate::Error { + let prefix: String = key.chars().take(64).collect(); + crate::Error::Serialization { + format: "backup".into(), + detail: format!("restore: document key '{prefix}' is malformed"), + } +} + +/// Split `"{db}:{tid}:{collection}:{rest}"`, checking the tenant. +fn split_key(key: &str, tenant_id: u64) -> crate::Result<(u64, &str, &str)> { + let mut parts = key.splitn(4, ':'); + let (Some(db), Some(tid), Some(collection), Some(rest)) = + (parts.next(), parts.next(), parts.next(), parts.next()) + else { + return Err(malformed(key)); + }; + let db = db.parse::().map_err(|_| malformed(key))?; + if tid.parse::().ok() != Some(tenant_id) || collection.is_empty() { + return Err(malformed(key)); + } + Ok((db, collection, rest)) +} + +/// Group every restored row by `(database, collection)`, then by storage key. +fn group_rows( + tenant_id: u64, + documents: Vec<(String, Vec)>, + documents_versioned: Vec<(String, Vec)>, +) -> crate::Result>> { + let mut grouped: BTreeMap<(u64, String), BTreeMap> = BTreeMap::new(); + for (key, body) in documents { + let (db, collection, rest) = split_key(&key, tenant_id)?; + let storage_key = StorageKey::parse(rest).ok_or_else(|| malformed(&key))?; + grouped + .entry((db, collection.to_string())) + .or_default() + .insert(storage_key, StoredRow::Current(body)); + } + for (key, value) in documents_versioned { + let (db, collection, rest) = split_key(&key, tenant_id)?; + let (hex, sys) = rest.split_once('\x00').ok_or_else(|| malformed(&key))?; + let storage_key = StorageKey::parse(hex).ok_or_else(|| malformed(&key))?; + let sys_from_ms = sys.parse::().map_err(|_| malformed(&key))?; + let rows = grouped.entry((db, collection.to_string())).or_default(); + match rows + .entry(storage_key) + .or_insert_with(|| StoredRow::Versions(Vec::new())) + { + StoredRow::Versions(versions) => versions.push((sys_from_ms, value)), + StoredRow::Current(_) => { + return Err(crate::Error::Serialization { + format: "backup".into(), + detail: format!( + "restore: row {storage_key} of '{collection}' is both current-only \ + and versioned" + ), + }); + } + } + } + for rows in grouped.values_mut() { + for row in rows.values_mut() { + if let StoredRow::Versions(versions) = row { + versions.sort_by_key(|(sys, _)| *sys); + } + } + } + Ok(grouped) +} + +fn collection_shape( + state: &SharedState, + database_id: DatabaseId, + tenant_id: u64, + collection: &str, +) -> crate::Result { + let stored = state + .credentials + .catalog() + .get_collection(database_id, tenant_id, collection)? + .ok_or_else(|| crate::Error::Internal { + detail: format!( + "restore: the backup holds rows of '{collection}' but restored no catalog \ + entry for it" + ), + })?; + // The storage mode the Data Plane registers for the collection, so a row + // decodes with the schema it was encoded with. + let strict = match stored.collection_type { + CollectionType::Document(DocumentMode::Strict(schema)) => Some(schema), + CollectionType::KeyValue(config) => Some(config.schema), + CollectionType::Document(DocumentMode::Schemaless) | CollectionType::Columnar(_) => None, + }; + Ok(CollectionShape { + strict, + declared_primary_key: stored.declared_primary_key, + }) +} + +/// A stored body as the MessagePack a put carries. +fn body_msgpack( + shape: &CollectionShape, + collection: &str, + key: StorageKey, + body: &[u8], +) -> crate::Result> { + match &shape.strict { + Some(schema) => binary_tuple_to_msgpack(body, schema) + .ok_or_else(|| undecodable_strict_row(collection, key.to_identity().as_str())), + None => Ok(body.to_vec()), + } +} + +/// Builds each row's unit for one collection. +struct RowBuilder<'a> { + state: &'a SharedState, + /// The destination database. + database_id: DatabaseId, + tenant: TenantId, + /// Bare catalog name: it keys the identity binds. + collection: &'a str, + /// The name the destination Data Plane stores the collection under: the + /// sub-records carry it. + stored: &'a str, + shape: CollectionShape, + /// `storage surrogate → primary key` the backup bound for this collection. + binds: HashMap, +} + +impl RowBuilder<'_> { + /// The row's client identity: the backup's binding, else the identity + /// INSERT derives from the row body. + fn identity(&self, key: StorageKey, body: Option<&[u8]>) -> crate::Result { + if let Some(pk) = self.binds.get(&key.surrogate().as_u32()) { + let pk = std::str::from_utf8(pk).map_err(|_| crate::Error::Serialization { + format: "backup".into(), + detail: format!( + "restore: the backup binds row {key} of '{}' to a key that is not UTF-8", + self.collection + ), + })?; + return Ok(RowIdentity::from_user_key(pk)); + } + Ok(match body { + Some(body) => { + RowIdentity::of_stored_row(body, self.shape.declared_primary_key.as_deref(), key) + } + None => key.to_identity(), + }) + } + + /// Bind the row's identity on this node. The backup's surrogate wins + /// unless this node already binds the identity: the row then installs + /// under that surrogate, over the row it names. + fn bind(&self, identity: &RowIdentity, key: StorageKey) -> crate::Result { + let surrogate = self.state.surrogate_assigner.bind( + nodedb_types::CollectionKey::from_bare(self.database_id, self.collection), + self.tenant, + identity.as_str().as_bytes(), + key.surrogate(), + )?; + Ok(CarriedIdentity { + collection: self.collection.to_string(), + pk_bytes: identity.as_str().as_bytes().to_vec(), + surrogate, + }) + } + + fn current(&self, key: StorageKey, body: &[u8]) -> crate::Result { + let value = body_msgpack(&self.shape, self.collection, key, body)?; + let identity = self.identity(key, Some(&value))?; + let carried = self.bind(&identity, key)?; + let op = document_put( + self.stored, + identity.as_str(), + value, + carried.surrogate.as_u32(), + None, + )?; + Ok(RowUnit { + ops: vec![op], + identities: vec![carried], + }) + } + + fn versions(&self, key: StorageKey, versions: &[(i64, Vec)]) -> crate::Result { + // Decode every version first: the identity comes from a live body. + let mut decoded = Vec::with_capacity(versions.len()); + for (sys_from_ms, raw) in versions { + let version = decode_value(raw)?; + let body = match version.tag { + TAG_LIVE => Some(body_msgpack( + &self.shape, + self.collection, + key, + version.body, + )?), + TAG_TOMBSTONE => None, + tag => { + return Err(crate::Error::Serialization { + format: "versioned-doc".into(), + detail: format!( + "restore: version {sys_from_ms} of row {key} of '{}' carries tag \ + {tag:#04x}, which no write path records", + self.collection + ), + }); + } + }; + let stamp = VersionStamp { + sys_from_ms: *sys_from_ms, + valid_from_ms: version.valid_from_ms, + valid_until_ms: version.valid_until_ms, + }; + decoded.push((stamp, body)); + } + let first_live = decoded.iter().find_map(|(_, body)| body.as_deref()); + let identity = self.identity(key, first_live)?; + let carried = self.bind(&identity, key)?; + let surrogate = carried.surrogate.as_u32(); + let mut ops = Vec::with_capacity(decoded.len()); + for (stamp, body) in decoded { + ops.push(match body { + Some(value) => document_put( + self.stored, + identity.as_str(), + value, + surrogate, + Some(stamp), + )?, + None => document_tombstone( + self.stored, + identity.as_str(), + surrogate, + stamp.sys_from_ms, + )?, + }); + } + Ok(RowUnit { + ops, + identities: vec![carried], + }) + } +} + +/// Every restored row of `tenant_id` in one database, one unit per row, +/// grouped by collection. Every row key must name `target.source`. `binds` +/// is the backup's primary-key section of that database. +pub(super) fn document_units( + state: &SharedState, + tenant_id: u64, + target: DatabaseTarget, + documents: Vec<(String, Vec)>, + documents_versioned: Vec<(String, Vec)>, + binds: &[SurrogateBindEntry], +) -> crate::Result> { + let grouped = group_rows(tenant_id, documents, documents_versioned)?; + let mut out = Vec::with_capacity(grouped.len()); + for ((db, stored), rows) in grouped { + if db != target.source.as_u64() { + return Err(crate::Error::Serialization { + format: "backup".into(), + detail: format!( + "restore: rows of '{stored}' name database {db}, but sit with database {}", + target.source.as_u64() + ), + }); + } + let name = target.resolve(&stored)?; + let database_id = target.dest; + let builder = RowBuilder { + state, + database_id, + tenant: TenantId::new(tenant_id), + collection: &name.bare, + stored: name.stored.as_str(), + shape: collection_shape(state, database_id, tenant_id, &name.bare)?, + binds: binds + .iter() + .filter(|b| b.tenant_id == tenant_id && b.collection == name.bare) + .map(|b| (b.surrogate, b.pk.as_slice())) + .collect(), + }; + let mut units = Vec::with_capacity(rows.len()); + for (key, row) in &rows { + units.push(match row { + StoredRow::Current(body) => builder.current(*key, body)?, + StoredRow::Versions(versions) => builder.versions(*key, versions)?, + }); + } + out.push(CollectionUnits { + database_id, + collection: name.bare.clone(), + units, + }); + } + Ok(out) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn keys_split_into_collection_and_storage_key() { + let (db, collection, rest) = split_key("0:7:users:0000002a", 7).unwrap(); + assert_eq!((db, collection, rest), (0, "users", "0000002a")); + assert!(split_key("0:8:users:0000002a", 7).is_err()); + assert!(split_key("0:7:users", 7).is_err()); + } + + #[test] + fn versions_group_under_their_row_in_system_time_order() { + let grouped = group_rows( + 7, + vec![("0:7:plain:00000001".into(), vec![1])], + vec![ + ( + "0:7:ledger:00000002\x0000000000000000000200".into(), + vec![2], + ), + ( + "0:7:ledger:00000002\x0000000000000000000100".into(), + vec![1], + ), + ], + ) + .unwrap(); + let ledger = &grouped[&(0, "ledger".to_string())]; + let key = StorageKey::parse("00000002").unwrap(); + let StoredRow::Versions(versions) = &ledger[&key] else { + panic!("a versioned row groups as versions"); + }; + let order: Vec = versions.iter().map(|(sys, _)| *sys).collect(); + assert_eq!(order, vec![100, 200]); + assert!(matches!( + grouped[&(0, "plain".to_string())][&StorageKey::parse("00000001").unwrap()], + StoredRow::Current(_) + )); + } +} diff --git a/nodedb/src/control/backup/restore/redo_reissue/edges.rs b/nodedb/src/control/backup/restore/redo_reissue/edges.rs new file mode 100644 index 000000000..b3dcbd958 --- /dev/null +++ b/nodedb/src/control/backup/restore/redo_reissue/edges.rs @@ -0,0 +1,143 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Restored graph edges as redo units. +//! +//! A backup carries every edge version under its versioned key, +//! `"{collection}\x00{src}\x00{label}\x00{dst}\x00{system_from:020}"`. Each +//! version re-issues at its original `system_from`, so the restored edge keeps +//! its history and its valid-from time. A tombstone version re-issues as a +//! delete at its `system_from`. Every replica updates its CSR index and its +//! node identities as it installs each version. +//! +//! The key's collection is the name the source Data Plane stored it under. +//! Each version re-issues under the destination-qualified name, and binds its +//! node identities under the bare name in the destination database. + +use std::collections::BTreeMap; + +use crate::control::state::SharedState; +use crate::control::surrogate::CarriedIdentity; +use crate::engine::graph::edge_store::{ + EdgeValuePayload, is_gdpr_erasure, is_tombstone, parse_versioned_edge_key, +}; +use crate::types::{DatabaseId, TenantId}; +use crate::wal::{EdgeDeleteRedo, EdgePutRedo}; + +use super::super::target::{DatabaseTarget, RestoredName}; +use super::sub_record::{edge_delete, edge_put}; +use super::units::{CollectionUnits, RowUnit}; + +fn malformed(key: &str) -> crate::Error { + let prefix: String = key.chars().take(64).collect(); + crate::Error::Serialization { + format: "backup".into(), + detail: format!("restore: edge key '{prefix:?}' is malformed"), + } +} + +/// The identity of node `node_id` in edge collection `collection`, bound +/// through the surrogate assigner as a live edge write binds it. +fn node_identity( + state: &SharedState, + database_id: DatabaseId, + tenant: TenantId, + collection: &str, + node_id: &str, +) -> crate::Result { + let surrogate = state.surrogate_assigner.assign( + nodedb_types::CollectionKey::from_bare(database_id, collection), + tenant, + node_id.as_bytes(), + )?; + Ok(CarriedIdentity { + collection: collection.to_string(), + pk_bytes: node_id.as_bytes().to_vec(), + surrogate, + }) +} + +/// One edge version of the collection `name` as a unit. +fn edge_unit( + state: &SharedState, + database_id: DatabaseId, + tenant: TenantId, + name: &RestoredName, + key: &str, + value: &[u8], +) -> crate::Result { + let (_, src_id, label, dst_id, system_from) = + parse_versioned_edge_key(key).ok_or_else(|| malformed(key))?; + let collection = name.bare.as_str(); + if is_tombstone(value) { + let op = edge_delete(&EdgeDeleteRedo { + collection: name.stored.to_string(), + src_id: src_id.to_string(), + label: label.to_string(), + dst_id: dst_id.to_string(), + system_from: Some(system_from), + })?; + return Ok(RowUnit { + ops: vec![op], + identities: Vec::new(), + }); + } + if is_gdpr_erasure(value) { + return Err(crate::Error::Serialization { + format: "backup".into(), + detail: format!( + "restore: edge version {system_from} in '{collection}' is an erasure marker, \ + which no write path records" + ), + }); + } + let payload = EdgeValuePayload::decode(value)?; + let src = node_identity(state, database_id, tenant, collection, src_id)?; + let dst = node_identity(state, database_id, tenant, collection, dst_id)?; + let op = edge_put(&EdgePutRedo { + collection: name.stored.to_string(), + src_id: src_id.to_string(), + label: label.to_string(), + dst_id: dst_id.to_string(), + properties: payload.properties, + src_surrogate: src.surrogate.as_u32(), + dst_surrogate: dst.surrogate.as_u32(), + system_from: Some(system_from), + })?; + Ok(RowUnit { + ops: vec![op], + identities: vec![src, dst], + }) +} + +/// Every restored edge version of `tenant_id` in one database, one unit per +/// version, grouped by edge collection in key order: each edge's versions in +/// system-time order. The edge section of a database's data section holds +/// that database's edges only. +pub(super) fn edge_units( + state: &SharedState, + tenant_id: u64, + target: DatabaseTarget, + edges: Vec<(String, Vec)>, +) -> crate::Result> { + let database_id = target.dest; + let tenant = TenantId::new(tenant_id); + let mut by_key: BTreeMap> = BTreeMap::new(); + for (key, value) in edges { + by_key.insert(key, value); + } + let mut grouped: BTreeMap> = BTreeMap::new(); + for (key, value) in &by_key { + let (stored, ..) = parse_versioned_edge_key(key).ok_or_else(|| malformed(key))?; + let name = target.resolve(stored)?; + let unit = edge_unit(state, database_id, tenant, &name, key, value)?; + grouped.entry(name.bare).or_default().push(unit); + } + Ok(grouped + .into_iter() + .map(|(collection, units)| CollectionUnits { + database_id, + collection, + units, + }) + .collect()) +} diff --git a/nodedb/src/control/backup/restore/redo_reissue/mod.rs b/nodedb/src/control/backup/restore/redo_reissue/mod.rs new file mode 100644 index 000000000..1a4a1478a --- /dev/null +++ b/nodedb/src/control/backup/restore/redo_reissue/mod.rs @@ -0,0 +1,13 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Durable, replicated RESTORE of document rows, their index entries, and +//! graph edges, as committed redo records. + +mod commit; +mod documents; +mod edges; +mod reissue; +mod sub_record; +mod units; + +pub(in crate::control::backup::restore) use reissue::{RestoredRows, reissue_rows_and_edges}; diff --git a/nodedb/src/control/backup/restore/redo_reissue/reissue.rs b/nodedb/src/control/backup/restore/redo_reissue/reissue.rs new file mode 100644 index 000000000..b7a32bf4f --- /dev/null +++ b/nodedb/src/control/backup/restore/redo_reissue/reissue.rs @@ -0,0 +1,65 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Re-issue every restored document row and graph edge as committed redo. +//! +//! Rows go first, then edges, so an edge's endpoints are in place when it +//! installs. A backup's index entries are not re-issued: every replica +//! derives a row's secondary index entries as it installs the row, exactly as +//! for a committed transaction's row. + +use crate::control::state::SharedState; +use crate::types::{SurrogateBindEntry, TenantId}; + +use super::super::target::DatabaseTarget; +use super::commit::commit_collection; +use super::documents::document_units; +use super::edges::edge_units; + +/// The backup sections this re-issue consumes. +pub(in crate::control::backup::restore) struct RestoredRows<'a> { + pub documents: Vec<(String, Vec)>, + pub documents_versioned: Vec<(String, Vec)>, + pub edges: Vec<(String, Vec)>, + /// The backup's primary-key section: each row's client identity. + pub binds: &'a [SurrogateBindEntry], +} + +/// What the re-issue committed. +#[derive(Debug, Default, Clone, Copy)] +pub(in crate::control::backup::restore) struct RedoReissueStats { + /// Document sub-records: one per current row, one per version. + pub documents: usize, + /// Edge sub-records: one per edge version. + pub edges: usize, + /// Redo records committed. + pub records: usize, +} + +/// Re-issue `rows` of `tenant_id`, backed up from `target.source`, durably +/// into `target.dest`. The first error fails the restore. +pub(in crate::control::backup::restore) async fn reissue_rows_and_edges( + state: &SharedState, + tenant_id: u64, + target: DatabaseTarget, + rows: RestoredRows<'_>, +) -> crate::Result { + let tenant = TenantId::new(tenant_id); + let mut stats = RedoReissueStats::default(); + let documents = document_units( + state, + tenant_id, + target, + rows.documents, + rows.documents_versioned, + rows.binds, + )?; + for collection in documents { + stats.documents += collection.units.iter().map(|u| u.ops.len()).sum::(); + stats.records += commit_collection(state, tenant, collection).await?; + } + for collection in edge_units(state, tenant_id, target, rows.edges)? { + stats.edges += collection.units.iter().map(|u| u.ops.len()).sum::(); + stats.records += commit_collection(state, tenant, collection).await?; + } + Ok(stats) +} diff --git a/nodedb/src/control/backup/restore/redo_reissue/sub_record.rs b/nodedb/src/control/backup/restore/redo_reissue/sub_record.rs new file mode 100644 index 000000000..84bb6f763 --- /dev/null +++ b/nodedb/src/control/backup/restore/redo_reissue/sub_record.rs @@ -0,0 +1,148 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Redo sub-record encoders for restored document rows and graph edges. +//! +//! Each encoder writes the exact payload shape the transaction resolver emits +//! and the replay arms decode, so a restored row installs through the same +//! arm a committed transaction's row does: +//! +//! * document put: `(collection, document_id, value, prov, surrogate)`, plus +//! `(sys_from_ms, valid_from_ms, valid_until_ms)` for a version of a +//! `bitemporal=true` collection; +//! * document delete: `(collection, document_id, prov, surrogate, sys_from_ms)`, +//! a tombstone version of a `bitemporal=true` collection; +//! * edge put and delete: [`EdgePutRedo`] and [`EdgeDeleteRedo`]. + +use nodedb_types::sync::wire::SyncProvenance; +use nodedb_wal::record::RecordType; + +use crate::wal::{EdgeDeleteRedo, EdgePutRedo, RedoSubRecord}; + +/// The system and valid time a document version was stored at. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) struct VersionStamp { + pub sys_from_ms: i64, + pub valid_from_ms: i64, + pub valid_until_ms: i64, +} + +fn encode_error(what: &str, e: impl std::fmt::Display) -> crate::Error { + crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("restore redo: encode {what}: {e}"), + } +} + +/// A document put. `value` is the row as MessagePack. `stamp` is `Some` for a +/// version of a `bitemporal=true` collection, which installs at that exact +/// system time. +pub(super) fn document_put( + collection: &str, + document_id: &str, + value: Vec, + surrogate: u32, + stamp: Option, +) -> crate::Result { + let prov: Option = None; + let payload = match stamp { + Some(s) => zerompk::to_msgpack_vec(&( + collection, + document_id, + value, + prov, + surrogate, + s.sys_from_ms, + s.valid_from_ms, + s.valid_until_ms, + )), + None => zerompk::to_msgpack_vec(&(collection, document_id, value, prov, surrogate)), + } + .map_err(|e| encode_error("document put", e))?; + Ok(RedoSubRecord { + record_type: RecordType::Put as u32, + payload, + }) +} + +/// The tombstone version of a `bitemporal=true` document at `sys_from_ms`. +pub(super) fn document_tombstone( + collection: &str, + document_id: &str, + surrogate: u32, + sys_from_ms: i64, +) -> crate::Result { + let prov: Option = None; + let payload = zerompk::to_msgpack_vec(&(collection, document_id, prov, surrogate, sys_from_ms)) + .map_err(|e| encode_error("document tombstone", e))?; + Ok(RedoSubRecord { + record_type: RecordType::Delete as u32, + payload, + }) +} + +/// One edge version put at its original `system_from`. +pub(super) fn edge_put(put: &EdgePutRedo) -> crate::Result { + Ok(RedoSubRecord { + record_type: RecordType::Put as u32, + payload: zerompk::to_msgpack_vec(put).map_err(|e| encode_error("edge put", e))?, + }) +} + +/// One edge tombstone at its original `system_from`. +pub(super) fn edge_delete(delete: &EdgeDeleteRedo) -> crate::Result { + Ok(RedoSubRecord { + record_type: RecordType::Delete as u32, + payload: zerompk::to_msgpack_vec(delete).map_err(|e| encode_error("edge delete", e))?, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + type BitemporalPut = ( + String, + String, + Vec, + Option, + u32, + i64, + i64, + i64, + ); + type PlainPut = (String, String, Vec, Option, u32); + type BitemporalDelete = (String, String, Option, u32, i64); + + #[test] + fn a_current_row_encodes_the_plain_put_shape() { + let sub = document_put("users", "u1", vec![0x80], 7, None).unwrap(); + assert_eq!(sub.record_type, RecordType::Put as u32); + let (collection, id, value, prov, surrogate): PlainPut = + zerompk::from_msgpack(&sub.payload).unwrap(); + assert_eq!( + (collection.as_str(), id.as_str(), value, surrogate), + ("users", "u1", vec![0x80], 7) + ); + assert!(prov.is_none()); + assert!(zerompk::from_msgpack::(&sub.payload).is_err()); + } + + #[test] + fn a_version_keeps_its_stamp() { + let stamp = VersionStamp { + sys_from_ms: 1_000, + valid_from_ms: 10, + valid_until_ms: 20, + }; + let sub = document_put("ledger", "e1", vec![0x80], 9, Some(stamp)).unwrap(); + let (_, _, _, _, surrogate, sys, vf, vu): BitemporalPut = + zerompk::from_msgpack(&sub.payload).unwrap(); + assert_eq!((surrogate, sys, vf, vu), (9, 1_000, 10, 20)); + + let tomb = document_tombstone("ledger", "e1", 9, 2_000).unwrap(); + assert_eq!(tomb.record_type, RecordType::Delete as u32); + let (_, id, _, surrogate, sys): BitemporalDelete = + zerompk::from_msgpack(&tomb.payload).unwrap(); + assert_eq!((id.as_str(), surrogate, sys), ("e1", 9, 2_000)); + } +} diff --git a/nodedb/src/control/backup/restore/redo_reissue/units.rs b/nodedb/src/control/backup/restore/redo_reissue/units.rs new file mode 100644 index 000000000..27161bd26 --- /dev/null +++ b/nodedb/src/control/backup/restore/redo_reissue/units.rs @@ -0,0 +1,36 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The unit a restored row or edge re-issues as. + +use crate::control::surrogate::CarriedIdentity; +use crate::types::DatabaseId; +use crate::wal::RedoSubRecord; + +/// One restored row or edge: its sub-records in apply order, and the +/// identities every replica binds before it installs them. A unit never +/// splits across two redo records. +pub(super) struct RowUnit { + pub ops: Vec, + pub identities: Vec, +} + +impl RowUnit { + /// The unit's encoded size, for sizing the records it goes into. + pub(super) fn byte_len(&self) -> usize { + self.ops.iter().map(|op| op.payload.len()).sum::() + + self + .identities + .iter() + .map(|identity| identity.collection.len() + identity.pk_bytes.len()) + .sum::() + } +} + +/// Every unit of one collection. All of them write that collection's vShard. +pub(super) struct CollectionUnits { + /// The destination database. + pub database_id: DatabaseId, + /// Bare catalog name of the collection. + pub collection: String, + pub units: Vec, +} diff --git a/nodedb/src/control/backup/restore/remote.rs b/nodedb/src/control/backup/restore/remote.rs deleted file mode 100644 index 911d73122..000000000 --- a/nodedb/src/control/backup/restore/remote.rs +++ /dev/null @@ -1,96 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Remote-node dispatch helpers for RESTORE TENANT. - -use std::sync::Arc; -use std::time::Duration; - -use nodedb_cluster::rpc_codec::{ExecuteRequest, ExecuteResponse, RaftRpc, TypedClusterError}; - -use crate::Error; -use crate::bridge::envelope::PhysicalPlan; -use crate::control::state::SharedState; -use crate::types::TraceId; -use nodedb_physical::physical_plan::wire as plan_wire; - -pub(super) const NODE_RESTORE_TIMEOUT: Duration = Duration::from_secs(120); - -pub(super) async fn dispatch_remote( - state: &Arc, - node_id: u64, - tenant_id: u64, - plan: PhysicalPlan, -) -> Result<(), Error> { - let transport = state - .cluster_transport - .as_ref() - .ok_or_else(|| Error::Internal { - detail: format!("restore: cluster_transport unavailable but node {node_id} is remote"), - })?; - let plan_bytes = plan_wire::encode(&plan).map_err(|e| Error::Internal { - detail: format!("restore: plan encode failed: {e}"), - })?; - let req = RaftRpc::ExecuteRequest(ExecuteRequest { - plan_bytes, - tenant_id, - database_id: nodedb_types::id::DatabaseId::DEFAULT.as_u64(), - deadline_remaining_ms: NODE_RESTORE_TIMEOUT.as_millis() as u64, - trace_id: TraceId::generate().0, - descriptor_versions: Vec::new(), - // Restore dispatch is not session-transaction-scoped. - txn_id: None, - }); - let resp = transport - .send_rpc(node_id, req) - .await - .map_err(|e| Error::Internal { - detail: format!("restore RPC to node {node_id} failed: {e}"), - })?; - match resp { - RaftRpc::ExecuteResponse(ExecuteResponse { success: true, .. }) => Ok(()), - RaftRpc::ExecuteResponse(ExecuteResponse { - error: Some(err), .. - }) => Err(map_typed_error(err, node_id)), - RaftRpc::ExecuteResponse(_) => Err(Error::Internal { - detail: format!("restore: empty error response from node {node_id}"), - }), - other => Err(Error::Internal { - detail: format!( - "restore: unexpected RPC response variant from node {node_id}: {other:?}" - ), - }), - } -} - -pub(super) fn map_typed_error(err: TypedClusterError, node_id: u64) -> Error { - match err { - TypedClusterError::Internal { message, .. } => Error::Internal { - detail: format!("restore node {node_id}: {message}"), - }, - TypedClusterError::DeadlineExceeded { elapsed_ms } => Error::Internal { - detail: format!("restore node {node_id}: deadline exceeded after {elapsed_ms}ms"), - }, - TypedClusterError::NotLeader { .. } => Error::Internal { - detail: format!("restore node {node_id}: routed to non-leader"), - }, - TypedClusterError::DescriptorMismatch { collection, .. } => Error::Internal { - detail: format!( - "restore node {node_id}: descriptor mismatch on collection {collection}" - ), - }, - // Keep the shard's verdict typed: a restore refused by the Data Plane - // must not read as a generic internal restore fault. - TypedClusterError::DataPlane { code } => Error::DataPlane(code.into()), - // A constraint verdict keeps its collection and kind, so the client - // reads the SQLSTATE the refusing shard meant. - TypedClusterError::RejectedConstraint { - collection, - constraint, - detail, - } => Error::RejectedConstraint { - collection, - constraint, - detail, - }, - } -} diff --git a/nodedb/src/control/backup/restore/sections.rs b/nodedb/src/control/backup/restore/sections.rs index 7614001b6..464dceeae 100644 --- a/nodedb/src/control/backup/restore/sections.rs +++ b/nodedb/src/control/backup/restore/sections.rs @@ -2,86 +2,138 @@ //! Catalog-section and data-section helpers for RESTORE TENANT. -use nodedb_types::DatabaseId; +use std::collections::BTreeMap; use std::sync::Arc; +use nodedb_types::backup_envelope::{ + DatabaseDataSection, Envelope, SECTION_ORIGIN_CATALOG_ROWS, SECTION_ORIGIN_DATABASES, + SECTION_ORIGIN_SOURCE_TOMBSTONES, SECTION_ORIGIN_SURROGATE_PK, Section, SourceTombstoneEntry, + StoredCollectionBlob, SurrogateBindBlob, +}; + use crate::Error; +use crate::control::catalog_entry::CatalogEntry; +use crate::control::metadata_proposer::propose_catalog_entry; use crate::control::security::catalog::StoredCollection; use crate::control::state::SharedState; use crate::types::{SurrogateBindEntry, TenantDataSnapshot}; -pub(super) fn merge_sections( - sections: &[nodedb_types::backup_envelope::Section], -) -> Result { - use nodedb_types::backup_envelope::{SECTION_ORIGIN_SURROGATE_PK, SurrogateBindBlob}; +use super::databases::DatabaseMap; - let mut merged = TenantDataSnapshot::default(); +/// Merge every data section and the surrogate-pk section into one +/// `TenantDataSnapshot` per source database, keyed by source database id. +pub(super) fn merge_sections( + sections: &[Section], +) -> Result, Error> { + let mut merged: BTreeMap = BTreeMap::new(); for section in sections { // The surrogate-pk metadata section carries the PK→surrogate identity - // map (not a tenant snapshot); decode it into `merged.surrogate_pk` so - // the restore orchestrator can rebind it after the data install. + // map (not a tenant snapshot). Each bind goes to its database's + // snapshot, so the restore rebinds it in that database. if section.origin_node_id == SECTION_ORIGIN_SURROGATE_PK { let binds: Vec = zerompk::from_msgpack(§ion.body).map_err(|_| Error::Internal { detail: "invalid backup format: surrogate-pk section payload is not decodable" .into(), })?; - merged - .surrogate_pk - .extend(binds.into_iter().map(|b| SurrogateBindEntry { - tenant_id: b.tenant_id, - collection: b.collection, - pk: b.pk, - surrogate: b.surrogate, - })); + for b in binds { + merged + .entry(b.database_id) + .or_default() + .surrogate_pk + .push(SurrogateBindEntry { + database_id: b.database_id, + tenant_id: b.tenant_id, + collection: b.collection, + pk: b.pk, + surrogate: b.surrogate, + }); + } continue; } if is_metadata_section(section) { continue; } - let snap: TenantDataSnapshot = + let data: DatabaseDataSection = zerompk::from_msgpack(§ion.body).map_err(|_| Error::Internal { + detail: "invalid backup format: section payload is not a database data section" + .into(), + })?; + let snap: TenantDataSnapshot = + zerompk::from_msgpack(&data.snapshot).map_err(|_| Error::Internal { detail: "invalid backup format: section payload is not a tenant snapshot".into(), })?; - merged.documents.extend(snap.documents); - merged.indexes.extend(snap.indexes); - merged.edges.extend(snap.edges); - merged.vectors.extend(snap.vectors); - merged.vector_params.extend(snap.vector_params); - merged.index_configs.extend(snap.index_configs); - merged.kv_tables.extend(snap.kv_tables); - // CRDT state is per-collection and tenant-explicit: - // `(tenant_id, collection, loro_bytes)`. Loro import is a monotonic - // merge so concatenating section contributions is safe. - merged.crdt_state.extend(snap.crdt_state); - merged.timeseries.extend(snap.timeseries); - merged.flushed_ts_segments.extend(snap.flushed_ts_segments); - merged.columnar_engines.extend(snap.columnar_engines); - // surrogate_pk on a per-node data section (Raft snapshots carry it - // there); merge it too so both transports converge here. - merged.surrogate_pk.extend(snap.surrogate_pk); + append_snapshot(merged.entry(data.database_id).or_default(), snap); } Ok(merged) } -pub(super) fn is_metadata_section(section: &nodedb_types::backup_envelope::Section) -> bool { +/// Concatenate every section of `snap` onto `into`. Each source node holds +/// disjoint vShards, so the sections never overlap. +fn append_snapshot(into: &mut TenantDataSnapshot, snap: TenantDataSnapshot) { + // Destructure exhaustively so a new field fails to compile here rather + // than being dropped from the restore. + let TenantDataSnapshot { + documents, + indexes, + edges, + vectors, + kv_tables, + crdt_state, + crdt_constraints, + timeseries, + flushed_ts_segments, + columnar_engines, + vector_params, + index_configs, + surrogate_pk, + tenant_edges, + group_write_marks, + documents_versioned, + indexes_versioned, + } = snap; + into.documents.extend(documents); + into.indexes.extend(indexes); + into.documents_versioned.extend(documents_versioned); + into.indexes_versioned.extend(indexes_versioned); + into.edges.extend(edges); + into.vectors.extend(vectors); + into.vector_params.extend(vector_params); + into.index_configs.extend(index_configs); + into.kv_tables.extend(kv_tables); + // CRDT state is per-collection and tenant-explicit. Loro import is a + // monotonic merge so concatenating section contributions is safe. + into.crdt_state.extend(crdt_state); + into.crdt_constraints.extend(crdt_constraints); + into.timeseries.extend(timeseries); + into.flushed_ts_segments.extend(flushed_ts_segments); + into.columnar_engines.extend(columnar_engines); + into.surrogate_pk.extend(surrogate_pk); + into.tenant_edges.extend(tenant_edges); + // A backup carries no marks: the guard reads the destination's marks. + into.group_write_marks.extend(group_write_marks); +} + +pub(super) fn is_metadata_section(section: &Section) -> bool { matches!( section.origin_node_id, - nodedb_types::backup_envelope::SECTION_ORIGIN_CATALOG_ROWS - | nodedb_types::backup_envelope::SECTION_ORIGIN_SOURCE_TOMBSTONES - | nodedb_types::backup_envelope::SECTION_ORIGIN_SURROGATE_PK + SECTION_ORIGIN_CATALOG_ROWS + | SECTION_ORIGIN_SOURCE_TOMBSTONES + | SECTION_ORIGIN_SURROGATE_PK + | SECTION_ORIGIN_DATABASES ) } /// Apply catalog-row and source-tombstone sections to the destination catalog. -/// Runs BEFORE the data-section restore. +/// Runs BEFORE the data-section restore, after `databases` maps every source +/// database to its destination. /// -/// Catalog rows are proposed cluster-wide through the metadata Raft -/// group (group 0) — exactly like CREATE COLLECTION — so every node's -/// catalog learns the restored collection and can serve it. A -/// catalog-propose failure on this path is FATAL: returning the data -/// restored but unqueryable on non-coordinator nodes is the -/// silent-partial-success anti-pattern this codebase forbids. +/// Each catalog row moves to its destination database. Catalog rows are +/// proposed cluster-wide through the metadata Raft group (group 0) — exactly +/// like CREATE COLLECTION — so every node's catalog learns the restored +/// collection and can serve it. A catalog-propose failure on this path is +/// FATAL: returning the data restored but unqueryable on non-coordinator nodes +/// is the silent-partial-success anti-pattern this codebase forbids. /// /// Returns every collection written to the catalog, in section order. The /// caller registers each one with this node's Data Plane before any restored @@ -91,35 +143,31 @@ pub(super) fn is_metadata_section(section: &nodedb_types::backup_envelope::Secti pub(super) fn apply_metadata_sections( state: &Arc, tenant_id: u64, - env: &nodedb_types::backup_envelope::Envelope, + env: &Envelope, + databases: &DatabaseMap, ) -> Result, Error> { - use nodedb_types::backup_envelope::{ - SECTION_ORIGIN_CATALOG_ROWS, SECTION_ORIGIN_SOURCE_TOMBSTONES, SourceTombstoneEntry, - StoredCollectionBlob, - }; let catalog = state.credentials.catalog(); let mut restored: Vec = Vec::new(); for section in &env.sections { match section.origin_node_id { SECTION_ORIGIN_CATALOG_ROWS => { - let Ok(blobs) = zerompk::from_msgpack::>(§ion.body) - else { - tracing::warn!( - tenant_id, - "restore: catalog-rows section failed to decode — skipping" - ); - continue; - }; + let blobs = zerompk::from_msgpack::>(§ion.body) + .map_err(|_| Error::Internal { + detail: "invalid backup format: catalog-rows section is not decodable" + .into(), + })?; for blob in blobs { - let Ok(coll) = zerompk::from_msgpack::(&blob.bytes) else { - tracing::warn!( - tenant_id, - name = %blob.name, - "restore: catalog row failed to decode — skipping" - ); - continue; - }; + let mut coll = + zerompk::from_msgpack::(&blob.bytes).map_err(|_| { + Error::Internal { + detail: format!( + "invalid backup format: catalog row of '{}' is not decodable", + blob.name + ), + } + })?; + coll.database_id = databases.target(blob.database_id)?.dest; // Propose the collection through the metadata Raft // group so every node's applier (`catalog_entry:: // apply::collection::put`) writes the row — mirroring @@ -127,61 +175,47 @@ pub(super) fn apply_metadata_sections( // blocks on its local applied-index watcher, so on the // cluster path it has already applied the put via the // same applier — we must NOT also put locally (double-put). - let entry = crate::control::catalog_entry::CatalogEntry::PutCollection( - Box::new(coll.clone()), - ); - let outcome = - crate::control::metadata_proposer::propose_catalog_entry(state, &entry)?; - if outcome.needs_local_apply() { + let entry = CatalogEntry::PutCollection(Box::new(coll.clone())); + if propose_catalog_entry(state, &entry)?.needs_local_apply() { // Single-node / no-cluster fallback: apply the // catalog mutation directly, matching what the // applier would have done on a clustered deployment. // A failure here is FATAL — the collection would be // unqueryable otherwise. - catalog.put_collection(DatabaseId::DEFAULT, &coll)?; + catalog.put_collection(coll.database_id, &coll)?; } restored.push(coll); } } SECTION_ORIGIN_SOURCE_TOMBSTONES => { - let Ok(tombs) = zerompk::from_msgpack::>(§ion.body) - else { - tracing::warn!( - tenant_id, - "restore: source-tombstones section failed to decode — skipping" - ); - continue; - }; + let tombs = zerompk::from_msgpack::>(§ion.body) + .map_err(|_| Error::Internal { + detail: "invalid backup format: source-tombstones section is not \ + decodable" + .into(), + })?; for t in tombs { + let database_id = databases.target(t.database_id)?.dest.as_u64(); // Replicate via the metadata Raft group so every node's boot WAL // replay barrier matches — a coordinator-local tombstone lets purged // writes resurrect on follower restart. - let entry = crate::control::catalog_entry::CatalogEntry::RecordWalTombstone { - database_id: DatabaseId::DEFAULT.as_u64(), + let entry = CatalogEntry::RecordWalTombstone { + database_id, tenant_id, - collection: t.collection, + collection: t.collection.clone(), purge_lsn: t.purge_lsn, }; - let outcome = - crate::control::metadata_proposer::propose_catalog_entry(state, &entry)?; - if outcome.needs_local_apply() { + if propose_catalog_entry(state, &entry)?.needs_local_apply() { // Single-node / no-cluster fallback: apply directly, // matching the applier. A failure here is FATAL — a // silently-skipped tombstone means purged writes resurrect - // on restart, which is the bug this change fixes. - if let crate::control::catalog_entry::CatalogEntry::RecordWalTombstone { - collection, - purge_lsn, - .. - } = entry - { - catalog.record_wal_tombstone( - DatabaseId::DEFAULT.as_u64(), - tenant_id, - &collection, - purge_lsn, - )?; - } + // on restart. + catalog.record_wal_tombstone( + database_id, + tenant_id, + &t.collection, + t.purge_lsn, + )?; } } } diff --git a/nodedb/src/control/backup/restore/target.rs b/nodedb/src/control/backup/restore/target.rs new file mode 100644 index 000000000..d1ea9e20a --- /dev/null +++ b/nodedb/src/control/backup/restore/target.rs @@ -0,0 +1,150 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Where one backed-up database's rows restore to. +//! +//! A backup names each database by its id on the source cluster. The +//! restore maps that id to the destination database of the same name. A +//! section's collection names come from the source Data Plane, qualified for +//! the source database. Each re-issue resolves them to the bare catalog name, +//! then qualifies that name for the destination database. + +use nodedb_types::{CollectionKey, DatabaseId, QualifiedCollection}; + +/// The source database of a backup section and the destination database +/// its rows restore into. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) struct DatabaseTarget { + /// The database id the backup recorded. + pub source: DatabaseId, + /// The database id the rows restore into. + pub dest: DatabaseId, +} + +/// A restored collection's names in the destination database. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct RestoredName { + /// Bare catalog name. It keys the vShard, the surrogate binds and the + /// catalog row. + pub bare: String, + /// The name the destination Data Plane stores the collection under. + pub stored: QualifiedCollection, +} + +impl RestoredName { + /// The canonical key of the collection in `database_id`. + pub fn key(&self, database_id: DatabaseId) -> CollectionKey<'_> { + CollectionKey::from_bare(database_id, &self.bare) + } +} + +fn malformed(key: &str, why: &str) -> crate::Error { + let prefix: String = key.chars().take(64).collect(); + crate::Error::Serialization { + format: "backup".into(), + detail: format!("restore: section key '{prefix}' is malformed: {why}"), + } +} + +impl DatabaseTarget { + /// The destination names of a collection the source Data Plane stored + /// as `source_stored`. + pub fn resolve(&self, source_stored: &str) -> crate::Result { + let key = CollectionKey::from_qualified_str(self.source, source_stored)?; + Ok(self.bare(key.name())) + } + + /// The destination names of the bare catalog name `bare`. + pub fn bare(&self, bare: &str) -> RestoredName { + RestoredName { + bare: bare.to_string(), + stored: QualifiedCollection::new(self.dest, bare), + } + } + + /// The part after `"{db}:{tid}:"` of a scoped section key. The key's + /// database must be this target's source database and its tenant must be + /// `tenant_id`. + pub fn scoped_rest<'k>(&self, key: &'k str, tenant_id: u64) -> crate::Result<&'k str> { + let mut parts = key.splitn(3, ':'); + let (Some(db), Some(tid), Some(rest)) = (parts.next(), parts.next(), parts.next()) else { + return Err(malformed(key, "expected '{db}:{tenant}:{collection}'")); + }; + if db.parse::().ok() != Some(self.source.as_u64()) { + return Err(malformed( + key, + &format!( + "it sits in the section of database {}, but names another database", + self.source.as_u64() + ), + )); + } + if tid.parse::().ok() != Some(tenant_id) { + return Err(malformed( + key, + &format!("it names a tenant other than {tenant_id}"), + )); + } + if rest.is_empty() { + return Err(malformed(key, "the collection is empty")); + } + Ok(rest) + } + + /// The destination names of the collection a `"{db}:{tid}:{collection}"` + /// section key names. + pub fn resolve_scoped(&self, key: &str, tenant_id: u64) -> crate::Result { + self.resolve(self.scoped_rest(key, tenant_id)?) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + const TARGET: DatabaseTarget = DatabaseTarget { + source: DatabaseId::new(1025), + dest: DatabaseId::new(2048), + }; + + #[test] + fn a_source_name_moves_to_the_destination_database() { + let name = TARGET + .resolve("1025/orders") + .expect("qualified for the source"); + assert_eq!(name.bare, "orders"); + assert_eq!(name.stored.as_str(), "2048/orders"); + assert_eq!( + name.key(TARGET.dest), + CollectionKey::from_bare(DatabaseId::new(2048), "orders") + ); + } + + #[test] + fn a_name_qualified_for_another_database_is_refused() { + assert!(TARGET.resolve("7/orders").is_err()); + assert!(TARGET.resolve("orders").is_err()); + } + + #[test] + fn the_default_database_keeps_bare_names() { + let target = DatabaseTarget { + source: DatabaseId::DEFAULT, + dest: DatabaseId::DEFAULT, + }; + let name = target + .resolve("orders") + .expect("bare in the default database"); + assert_eq!(name.stored.as_str(), "orders"); + } + + #[test] + fn a_scoped_key_must_name_the_source_database_and_tenant() { + let name = TARGET + .resolve_scoped("1025:7:1025/metrics", 7) + .expect("scoped key of the source database"); + assert_eq!(name.stored.as_str(), "2048/metrics"); + assert!(TARGET.resolve_scoped("0:7:metrics", 7).is_err()); + assert!(TARGET.resolve_scoped("1025:8:1025/metrics", 7).is_err()); + assert!(TARGET.resolve_scoped("1025:7", 7).is_err()); + } +} diff --git a/nodedb/src/control/backup/restore/timeseries_reissue.rs b/nodedb/src/control/backup/restore/timeseries_reissue.rs index 3329ad0a9..2950e23e1 100644 --- a/nodedb/src/control/backup/restore/timeseries_reissue.rs +++ b/nodedb/src/control/backup/restore/timeseries_reissue.rs @@ -10,7 +10,6 @@ //! identity re-derived from tag columns. use std::collections::HashMap; -use std::time::Duration; use nodedb_types::RlsWriteCheck; use nodedb_types::columnar::schema::TS_SYSTEM; @@ -19,20 +18,13 @@ use nodedb_types::value::Value; use crate::Error; use crate::bridge::envelope::PhysicalPlan; -use crate::control::server::shared::ddl::sync_dispatch; -use crate::control::server::wal_dispatch::wal_append_if_write; -use crate::control::state::SharedState; use crate::engine::timeseries::columnar_memtable::{ ColumnData, ColumnType, ColumnarMemtable, ColumnarMemtableConfig, MemtableSnapshot, }; use crate::engine::timeseries::columnar_segment::ColumnarSegmentReader; -use crate::types::{DatabaseId, TenantId, TsFlushedCollectionBlob, VShardId}; +use crate::types::TsFlushedCollectionBlob; use nodedb_physical::physical_plan::TimeseriesOp; -/// Per-collection re-issue dispatch timeout. Generous: a restored collection may -/// carry many flushed partitions' worth of rows in one ingest. -const REISSUE_TIMEOUT: Duration = Duration::from_secs(120); - /// Server-stamped reserved column — re-derived by the ingest path, so it must /// NOT be carried back into the re-issued rows (the ingest handler restamps it). /// `_ts_valid_from` / `_ts_valid_until` ARE client-provided and preserved. @@ -312,54 +304,6 @@ pub fn build_timeseries_ingest_plan( })) } -/// Re-issue a restored timeseries collection's rows durably. -/// -/// Branches identically to a normal write (and to `reissue_columnar_durably`): -/// - Cluster: `to_replicated_entry` + `propose_replicated_entry`. -/// - Single-node: `wal_append_if_write` then `sync_dispatch::dispatch_system`. -pub async fn reissue_timeseries_durably( - state: &SharedState, - tenant_id: TenantId, - database_id: DatabaseId, - collection: &str, - plan: PhysicalPlan, -) -> crate::Result<()> { - let vshard = VShardId::from_collection_in_database(database_id, collection); - - if let Some(proposer) = state.async_raft_proposer() { - let entry = crate::control::wal_replication::to_replicated_entry( - tenant_id, - database_id, - vshard, - &crate::control::wal_replication::ReplicableWrite::decide_for_replication(&plan)?, - )? - .ok_or_else(|| Error::Internal { - detail: format!( - "restore reissue: timeseries plan for '{collection}' did not map to a \ - replicated write" - ), - })?; - crate::control::wal_replication::propose_replicated_entry(state, proposer, entry).await?; - return Ok(()); - } - - // Single-node: WAL first (durable for restart replay), then install live. - wal_append_if_write(&state.wal, tenant_id, vshard, database_id, &plan)?; - sync_dispatch::dispatch_system( - state, - sync_dispatch::SystemTask::new( - sync_dispatch::SystemReason::BackupRestore, - tenant_id, - database_id, - collection, - plan, - ), - REISSUE_TIMEOUT, - ) - .await?; - Ok(()) -} - #[cfg(test)] mod tests { use super::*; diff --git a/nodedb/src/control/backup/restore/topology.rs b/nodedb/src/control/backup/restore/topology.rs deleted file mode 100644 index ff6215f63..000000000 --- a/nodedb/src/control/backup/restore/topology.rs +++ /dev/null @@ -1,188 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Topology-aware snapshot bucketing for RESTORE TENANT. - -use std::collections::BTreeMap; - -use nodedb_cluster::routing::{VSHARD_COUNT, vshard_for_collection}; -use nodedb_types::id::DatabaseId; - -use crate::control::backup::snapshot_keys::{ - extract_db_scoped_collection, extract_db_tenant_scoped_collection, -}; -use crate::control::state::SharedState; -use crate::types::TenantDataSnapshot; - -/// Bucketed output from `split_by_current_topology`. -pub(super) struct SplitOutput { - pub buckets: BTreeMap, - pub malformed_keys: usize, - pub route_fallbacks: usize, -} - -enum RouteOutcome { - Routed(u64), - Malformed, - NoLeader, -} - -/// Bucket the merged snapshot per current vshard ownership. -/// -/// Replicated-by-design data (graph edges, CRDT state) goes to every -/// owning node. Single-node mode is the degenerate case: everything to self. -pub(super) fn split_by_current_topology( - state: &SharedState, - tenant_id: u64, - merged: TenantDataSnapshot, -) -> SplitOutput { - let routing = state - .cluster_routing - .as_ref() - .map(|r| r.read().unwrap_or_else(|poisoned| poisoned.into_inner())); - let single_node = routing.is_none() || state.cluster_transport.is_none(); - - if single_node { - let mut out = BTreeMap::new(); - out.insert(state.node_id, merged); - return SplitOutput { - buckets: out, - malformed_keys: 0, - route_fallbacks: 0, - }; - } - let routing = - routing.expect("invariant: single_node is false, so routing.is_some() is guaranteed"); - - let mut all_owners = BTreeMap::::new(); - for vshard in 0..VSHARD_COUNT { - if let Ok(node) = routing.leader_for_vshard(vshard) - && node != 0 - { - all_owners.entry(node).or_default(); - } - } - if all_owners.is_empty() { - all_owners.insert(state.node_id, TenantDataSnapshot::default()); - } - - // Restore today operates on `DatabaseId::DEFAULT`; the snapshot/topology - // wire format gains a database_id alongside tenant_id when multi-database - // restore lands, at which point this binding moves up to a parameter. - let database_id = DatabaseId::DEFAULT; - let route_collection = |coll: &str| -> RouteOutcome { - let v = vshard_for_collection(database_id, coll); - match routing.leader_for_vshard(v) { - Ok(leader) if leader != 0 => RouteOutcome::Routed(leader), - _ => RouteOutcome::NoLeader, - } - }; - // Documents / indexes / vectors / timeseries keys are db-tenant-scoped: - // `"{db}:{tid}:{collection}[:suffix]"` (collection never contains ':'/'\0'). - let route_key = |key: &str| -> RouteOutcome { - match extract_db_tenant_scoped_collection(key, tenant_id) { - Some(coll) => route_collection(coll), - None => RouteOutcome::Malformed, - } - }; - // Columnar / flushed-ts keys are db-scoped: `"{db}:{tid}:{collection}"` - // where the collection is the whole remainder and may itself contain ':'. - let route_db_scoped_key = |key: &str| -> RouteOutcome { - match extract_db_scoped_collection(key, tenant_id) { - Some(coll) => route_collection(coll), - None => RouteOutcome::Malformed, - } - }; - - let mut malformed = 0usize; - let mut fallbacks = 0usize; - let mut resolve = |outcome: RouteOutcome, key: Option<&str>| -> u64 { - match outcome { - RouteOutcome::Routed(node) => node, - RouteOutcome::Malformed => { - malformed += 1; - if let Some(k) = key { - let prefix: String = k.chars().take(64).collect(); - tracing::warn!(tenant_id, key_prefix = %prefix, "restore: malformed key"); - } - state.node_id - } - RouteOutcome::NoLeader => { - fallbacks += 1; - state.node_id - } - } - }; - - for entry in merged.documents { - let node = resolve(route_key(&entry.0), Some(&entry.0)); - all_owners.entry(node).or_default().documents.push(entry); - } - for entry in merged.indexes { - let node = resolve(route_key(&entry.0), Some(&entry.0)); - all_owners.entry(node).or_default().indexes.push(entry); - } - for entry in merged.kv_tables { - let node = resolve(route_collection(&entry.0), Some(&entry.0)); - all_owners.entry(node).or_default().kv_tables.push(entry); - } - for entry in merged.timeseries { - let node = resolve(route_key(&entry.0), Some(&entry.0)); - all_owners.entry(node).or_default().timeseries.push(entry); - } - // Plain-columnar engine state is NOT installed via the snapshot path: the - // snapshot-install lands data in in-memory-only Data Plane maps with no WAL - // record and no Raft entry, so it is lost on restart (single-node) and never - // reaches replicas (cluster). RESTORE re-issues columnar rows as durable - // `ColumnarOp::Insert`s instead (see `columnar_reissue`); `merged.columnar_engines` - // is therefore drained by the caller before this split and never bucketed here. - debug_assert!( - merged.columnar_engines.is_empty(), - "columnar engines must be drained before topology split" - ); - // Vector engine state is likewise NOT installed via the snapshot path: - // RESTORE re-issues each restored vector as a durable `VectorOp::Insert` - // instead (see `vector_reissue`); `merged.vectors` is therefore drained - // by the caller before this split and never bucketed here. - debug_assert!( - merged.vectors.is_empty(), - "vectors must be drained before topology split" - ); - for blob in merged.flushed_ts_segments { - let node = resolve( - route_db_scoped_key(&blob.collection_key), - Some(&blob.collection_key), - ); - all_owners - .entry(node) - .or_default() - .flushed_ts_segments - .push(blob); - } - - // Replicated-by-design: every owning node gets a copy. - for entry in &merged.edges { - for snap in all_owners.values_mut() { - snap.edges.push(entry.clone()); - } - } - // CRDT state is NOT bucketed here: the per-node snapshot fan-out is - // race-prone (skips data groups that have not elected a leader yet) and not - // durable across restart. RESTORE drains the CRDT section before this split - // and re-issues each collection's Loro snapshot durably through Raft to its - // owning data group (see `crdt_reissue`). The vec is therefore empty here by - // contract. - debug_assert!( - merged.crdt_state.is_empty(), - "CRDT state must be drained before topology split" - ); - - SplitOutput { - buckets: all_owners, - malformed_keys: malformed, - route_fallbacks: fallbacks, - } -} - -pub(super) fn is_self(state: &SharedState, node_id: u64) -> bool { - node_id == state.node_id || node_id == 0 || state.cluster_transport.is_none() -} diff --git a/nodedb/src/control/backup/restore/vector_reissue.rs b/nodedb/src/control/backup/restore/vector_reissue.rs index a8c54b3f6..391449082 100644 --- a/nodedb/src/control/backup/restore/vector_reissue.rs +++ b/nodedb/src/control/backup/restore/vector_reissue.rs @@ -7,17 +7,10 @@ //! re-issues each vector as a durable `VectorOp::Insert`: Raft-proposed on //! cluster, WAL-appended + dispatched on single-node. -use std::time::Duration; - use nodedb_types::surrogate::Surrogate; -use crate::Error; use crate::bridge::envelope::PhysicalPlan; -use crate::control::server::shared::ddl::sync_dispatch; -use crate::control::server::wal_dispatch::wal_append_if_write; -use crate::control::state::SharedState; use crate::engine::vector::index_config::{IndexConfig, IndexType}; -use crate::types::{DatabaseId, TenantId, VShardId}; use nodedb_physical::physical_plan::VectorOp; use nodedb_types::vector_distance::DistanceMetric; @@ -107,55 +100,3 @@ pub fn build_vector_set_params_plan( ivf_nprobe: config.ivf_nprobe, }) } - -/// Re-issue a restored vector insert durably. -/// -/// Branches identically to a normal write: -/// - Cluster: `to_replicated_entry` + `propose_replicated_entry`. -/// - Single-node: `wal_append_if_write` then `sync_dispatch::dispatch_system`. -pub async fn reissue_vector_durably( - state: &SharedState, - tenant_id: TenantId, - database_id: DatabaseId, - collection: &str, - plan: PhysicalPlan, -) -> crate::Result<()> { - let vshard = VShardId::from_collection_in_database(database_id, collection); - - if let Some(proposer) = state.async_raft_proposer() { - let entry = crate::control::wal_replication::to_replicated_entry( - tenant_id, - database_id, - vshard, - &crate::control::wal_replication::ReplicableWrite::decide_for_replication(&plan)?, - )? - .ok_or_else(|| Error::Internal { - detail: format!( - "restore reissue: vector plan for '{collection}' did not map to a \ - replicated write" - ), - })?; - crate::control::wal_replication::propose_replicated_entry(state, proposer, entry).await?; - return Ok(()); - } - - // Single-node: WAL first (durable for restart replay), then install live. - wal_append_if_write(&state.wal, tenant_id, vshard, database_id, &plan)?; - sync_dispatch::dispatch_system( - state, - sync_dispatch::SystemTask::new( - sync_dispatch::SystemReason::BackupRestore, - tenant_id, - database_id, - collection, - plan, - ), - REISSUE_TIMEOUT, - ) - .await?; - Ok(()) -} - -/// Per-vector re-issue dispatch timeout. Mirrors the columnar/timeseries -/// reissue timeout; a single-vector `Insert` completes far under this. -const REISSUE_TIMEOUT: Duration = Duration::from_secs(120); diff --git a/nodedb/src/control/backup/snapshot_keys.rs b/nodedb/src/control/backup/snapshot_keys.rs index d1d60dcfa..d3b13685b 100644 --- a/nodedb/src/control/backup/snapshot_keys.rs +++ b/nodedb/src/control/backup/snapshot_keys.rs @@ -12,13 +12,14 @@ //! [`extract_db_tenant_scoped_collection`]. //! - **db-scoped (collection-last)** — `"{db}:{tid}:{collection}"` where the //! collection is the remainder and may itself contain `':'` (flushed-ts -//! segments, columnar engines). Use [`extract_db_scoped_collection`]. -//! - **collection-name-only** — the key IS the bare collection name (kv tables). -//! Routed directly; no extractor needed. +//! segments, columnar engines, kv tables). Use [`extract_db_scoped_collection`]. //! -//! Both the RESTORE topology splitter and the Raft snapshot SEND builder filter -//! sections by which vshard each entry's collection routes to, so the parsing -//! lives here once and is shared by both — never duplicated ad-hoc. +//! Every extracted collection is the name the Data Plane stores it under: +//! database-qualified (`"{db}/{name}"`) outside the default database. +//! [`vshard_of_stored`] maps such a name to its vShard. +//! +//! The Raft snapshot SEND builder filters sections by which vshard each +//! entry's collection routes to, so the parsing lives here once. //! //! The backup orchestrator additionally needs to filter a fully-gathered, //! single-tenant [`TenantDataSnapshot`] *in place* to a set of source vshards @@ -30,9 +31,24 @@ use std::collections::HashSet; +use nodedb_types::{CollectionKey, DatabaseId}; + use crate::engine::graph::edge_store::parse_versioned_edge_key; use crate::types::TenantDataSnapshot; +/// The vShard of a collection named as the Data Plane stores it in +/// `database_id`. +/// +/// A database-qualified name routes by its bare name, exactly as the write +/// that stored it did. A name without the qualifier routes as a bare name. +/// Both are deterministic, so every source node that filters the same entry +/// assigns it the same vShard. +pub fn vshard_of_stored(database_id: DatabaseId, stored: &str) -> u32 { + let key = CollectionKey::from_qualified_str(database_id, stored) + .unwrap_or_else(|_| CollectionKey::from_bare(database_id, stored)); + nodedb_cluster::routing::vshard_for_collection(key) +} + /// Extract the collection from a `"{db}:{tid}:{collection}[:suffix...]"` key. /// /// Used by documents, indexes, vectors, and timeseries-memtable sections, @@ -80,18 +96,19 @@ pub fn extract_db_scoped_collection(key: &str, tenant_id: u64) -> Option<&str> { /// section shapes are unchanged, so the RESTORE merge path (`merge_sections`) /// consumes the output exactly as before. /// -/// `vshard_of` maps a collection name to its vshard (the caller passes the -/// canonical `vshard_for_collection(DEFAULT, _)`), matching the Raft snapshot -/// SEND builder. Every section kind the snapshot carries is classified here so -/// adding a section without updating this filter is impossible to miss: +/// `vshard_of` maps a collection name to its vshard (the caller passes +/// [`vshard_of_stored`] for the snapshot's database), matching the Raft +/// snapshot SEND builder. Every section kind the snapshot carries is +/// classified here so adding a section without updating this filter is +/// impossible to miss: /// -/// - db-tenant-scoped keys (`documents`, `indexes`, `vectors`, `timeseries`) -/// via [`extract_db_tenant_scoped_collection`]. -/// - db-scoped keys (`flushed_ts_segments`, `columnar_engines`) via -/// [`extract_db_scoped_collection`]. -/// - collection-name-only keys (`kv_tables`) routed directly. +/// - db-tenant-scoped keys (`documents`, `indexes`, `documents_versioned`, +/// `indexes_versioned`, `vectors`, `timeseries`) via +/// [`extract_db_tenant_scoped_collection`]. +/// - db-scoped keys (`flushed_ts_segments`, `columnar_engines`, `kv_tables`) +/// via [`extract_db_scoped_collection`]. /// - graph `edges` via [`parse_versioned_edge_key`] (key embeds the collection). -/// - `surrogate_pk` by its explicit `collection` field. +/// - `surrogate_pk` by its explicit `collection` field (the bare name). /// - CRDT (`crdt_state`): per-collection, tenant-explicit. Each entry carries /// its single collection, so it is kept iff that collection's vshard is in /// `source_vshards` — the node owning the collection keeps it, every other @@ -115,15 +132,18 @@ pub fn retain_tenant_data_for_vshards( snap.documents.retain(|(k, _)| in_group_db_tenant_scoped(k)); snap.indexes.retain(|(k, _)| in_group_db_tenant_scoped(k)); + snap.documents_versioned + .retain(|(k, _)| in_group_db_tenant_scoped(k)); + snap.indexes_versioned + .retain(|(k, _)| in_group_db_tenant_scoped(k)); snap.vectors.retain(|(k, _)| in_group_db_tenant_scoped(k)); snap.timeseries .retain(|(k, _)| in_group_db_tenant_scoped(k)); snap.flushed_ts_segments .retain(|b| in_group_db_scoped(&b.collection_key)); snap.columnar_engines.retain(|(k, _)| in_group_db_scoped(k)); - // kv_tables / surrogate_pk: the key / field IS the collection name. - snap.kv_tables - .retain(|(k, _)| source_vshards.contains(&vshard_of(k))); + snap.kv_tables.retain(|(k, _)| in_group_db_scoped(k)); + // surrogate_pk: the field IS the collection name. snap.surrogate_pk .retain(|e| source_vshards.contains(&vshard_of(&e.collection))); // Graph edges: collection is the first '\0'-delimited key component. An @@ -146,6 +166,26 @@ pub fn retain_tenant_data_for_vshards( mod tests { use super::{extract_db_scoped_collection, extract_db_tenant_scoped_collection}; + /// A qualified Data-Plane name in a named database routes to the vShard + /// of its bare catalog key, never to the vShard of the qualified string. + #[test] + fn a_stored_name_routes_by_its_bare_key() { + use nodedb_types::{CollectionKey, DatabaseId, QualifiedCollection}; + + let db = DatabaseId::new(1025); + let stored = QualifiedCollection::new(db, "orders"); + let expected = + nodedb_cluster::routing::vshard_for_collection(CollectionKey::from_bare(db, "orders")); + assert_eq!(super::vshard_of_stored(db, stored.as_str()), expected); + assert_eq!( + super::vshard_of_stored(DatabaseId::DEFAULT, "orders"), + nodedb_cluster::routing::vshard_for_collection(CollectionKey::from_bare( + DatabaseId::DEFAULT, + "orders" + )) + ); + } + #[test] fn extract_db_tenant_scoped_collection_parses_key() { // Documents / indexes: collection is the 3rd token, suffix follows. @@ -228,7 +268,7 @@ mod tests { partitions: vec![], }, ], - kv_tables: vec![("alpha".into(), b"a".to_vec())], + kv_tables: vec![(format!("0:{TID}:alpha"), b"a".to_vec())], ..Default::default() }; @@ -278,7 +318,7 @@ mod tests { let mut snap = TenantDataSnapshot { timeseries: vec![("0:1:alpha".into(), b"a".to_vec())], columnar_engines: vec![("0:1:beta".into(), b"b".to_vec())], - kv_tables: vec![("gamma".into(), b"g".to_vec())], + kv_tables: vec![("0:1:gamma".into(), b"g".to_vec())], ..Default::default() }; let all: HashSet = ["alpha", "beta", "gamma"] diff --git a/nodedb/src/control/catalog_entry/apply/collection.rs b/nodedb/src/control/catalog_entry/apply/collection.rs index 5fa9eafd0..eb5631a6b 100644 --- a/nodedb/src/control/catalog_entry/apply/collection.rs +++ b/nodedb/src/control/catalog_entry/apply/collection.rs @@ -178,9 +178,8 @@ pub fn finalize_purge( catalog, )?; catalog.delete_all_surrogates_for_collection( - database_id, + nodedb_types::CollectionKey::from_bare(database_id, name), nodedb_types::TenantId::new(tenant_id), - name, )?; // An index cannot outlive the collection it indexes. Its identity rows, // its ownership rows, and any engine-side build parameters go with the diff --git a/nodedb/src/control/catalog_entry/apply/dispatch.rs b/nodedb/src/control/catalog_entry/apply/dispatch.rs index 5a75212b6..b89cbb37e 100644 --- a/nodedb/src/control/catalog_entry/apply/dispatch.rs +++ b/nodedb/src/control/catalog_entry/apply/dispatch.rs @@ -8,6 +8,8 @@ use crate::control::catalog_entry::entry::CatalogEntry; use crate::control::security::catalog::SystemCatalog; use crate::control::security::catalog::types::CheckpointDoc; +use super::outcome::ApplyOutcome; + use super::{ alert_rule, api_key, auth_user, change_stream, checkpoint, collection, column_stats, consumer_group, continuous_aggregate, custom_type, database, function, index_registry, @@ -20,22 +22,34 @@ use super::{ /// Apply `entry` to `catalog`. /// /// A failed catalog write raises: skipping a committed metadata entry -/// diverges this node from the quorum. `Ok(false)` reports that the entry -/// wrote nothing, which still concludes its DDL. Debug builds verify +/// diverges this node from the quorum. [`ApplyOutcome::Unchanged`] reports +/// that the entry wrote nothing, which still concludes its DDL. +/// [`ApplyOutcome::Refused`] reports an entry that breaks a role rule at its +/// log position; every node refuses it alike. Debug builds verify /// referential integrity after every apply — release-gated because a full /// rescan would wedge `raft_tick_loop` on a node with a pre-existing orphan. -pub fn apply_to(entry: &CatalogEntry, catalog: &SystemCatalog) -> Result { - let applied = match entry { +pub fn apply_to( + entry: &CatalogEntry, + catalog: &SystemCatalog, +) -> Result { + let outcome = match entry { CatalogEntry::PutTenantWithAdmin { tenant, admin } => { - tenant::put_with_admin(tenant, admin, catalog)? + if tenant::put_with_admin(tenant, admin, catalog)? { + ApplyOutcome::Applied + } else { + ApplyOutcome::Unchanged + } } + CatalogEntry::PutUser(stored) => user::put(stored, catalog)?, + CatalogEntry::PutRole(stored) => role::put(stored, catalog)?, + CatalogEntry::DeleteRole { name } => role::delete(name, catalog)?, _ => { apply_to_inner(entry, catalog)?; - true + ApplyOutcome::Applied } }; - if !applied { - return Ok(false); + if !outcome.wrote() { + return Ok(outcome); } #[cfg(debug_assertions)] { @@ -62,7 +76,7 @@ pub fn apply_to(entry: &CatalogEntry, catalog: &SystemCatalog) -> Result crate::Result<()> { @@ -144,10 +158,13 @@ fn apply_to_inner(entry: &CatalogEntry, catalog: &SystemCatalog) -> crate::Resul tenant_id, name, } => change_stream::delete(*database_id, *tenant_id, name, catalog), - CatalogEntry::PutUser(stored) => user::put(stored, catalog), + // Applied by `apply_to`, which reports a refused entry. + CatalogEntry::PutUser(_) => Ok(()), CatalogEntry::DropUser { username } => user::delete(username, catalog), - CatalogEntry::PutRole(stored) => role::put(stored, catalog), - CatalogEntry::DeleteRole { name } => role::delete(name, catalog), + // Applied by `apply_to`, which reports a refused entry. + CatalogEntry::PutRole(_) => Ok(()), + // Applied by `apply_to`, which reports a refused entry. + CatalogEntry::DeleteRole { .. } => Ok(()), CatalogEntry::PutApiKey(stored) => api_key::put(stored, catalog), CatalogEntry::RevokeApiKey { key_id } => api_key::revoke(key_id, catalog), CatalogEntry::PutAuthUser(stored) => auth_user::put(stored, catalog), diff --git a/nodedb/src/control/catalog_entry/apply/mod.rs b/nodedb/src/control/catalog_entry/apply/mod.rs index 24d244877..2a2380265 100644 --- a/nodedb/src/control/catalog_entry/apply/mod.rs +++ b/nodedb/src/control/catalog_entry/apply/mod.rs @@ -23,6 +23,7 @@ pub mod index_registry; pub mod local; pub mod materialized_view; pub mod oidc_provider; +pub mod outcome; pub mod owner; pub mod permission; pub mod procedure; @@ -45,3 +46,4 @@ pub mod vector; pub mod wal_tombstone; pub use dispatch::apply_to; +pub use outcome::ApplyOutcome; diff --git a/nodedb/src/control/catalog_entry/apply/outcome.rs b/nodedb/src/control/catalog_entry/apply/outcome.rs new file mode 100644 index 000000000..ce88e89af --- /dev/null +++ b/nodedb/src/control/catalog_entry/apply/outcome.rs @@ -0,0 +1,26 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! What applying one catalog entry did. + +use crate::control::security::role_assignment::RoleRefusal; + +/// The result of applying a committed catalog entry. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum ApplyOutcome { + /// The entry was written. Its post-apply side effects run. + Applied, + /// The entry wrote nothing and still concludes its DDL, such as an + /// if-absent create for a descriptor that exists. No side effects run. + Unchanged, + /// The entry breaks a role rule at its log position, and is skipped. + /// Every node applies the same log in the same order, so every node + /// skips it alike. No side effects run. + Refused(RoleRefusal), +} + +impl ApplyOutcome { + /// Whether the entry was written, so its side effects must run. + pub fn wrote(&self) -> bool { + matches!(self, Self::Applied) + } +} diff --git a/nodedb/src/control/catalog_entry/apply/role.rs b/nodedb/src/control/catalog_entry/apply/role.rs index 482625fd9..8d2710d42 100644 --- a/nodedb/src/control/catalog_entry/apply/role.rs +++ b/nodedb/src/control/catalog_entry/apply/role.rs @@ -1,17 +1,54 @@ // SPDX-License-Identifier: BUSL-1.1 //! Apply Role catalog entries to `SystemCatalog` redb. +//! +//! Both entries check the role rules against the catalog at their log +//! position. The statement checked before proposing; a change that +//! committed in between is caught here, on every node alike. use crate::control::security::catalog::{StoredRole, SystemCatalog, catalog_err}; +use crate::control::security::role_assignment::{self, RoleRefusal}; -pub fn put(stored: &StoredRole, catalog: &SystemCatalog) -> crate::Result<()> { +use super::outcome::ApplyOutcome; + +/// Write the role, unless its inheritance parent is neither built in nor +/// defined. +pub fn put(stored: &StoredRole, catalog: &SystemCatalog) -> crate::Result { + let parent = stored.parent.as_str(); + if !parent.is_empty() && !role_assignment::is_builtin_role_name(parent) { + let roles = catalog.load_all_roles()?; + if !roles.iter().any(|role| role.name == parent) { + let refusal = RoleRefusal::Undefined { + name: parent.to_string(), + }; + tracing::warn!( + role = %stored.name, + %refusal, + "catalog_entry: role entry refused; its parent is undefined" + ); + return Ok(ApplyOutcome::Refused(refusal)); + } + } catalog .put_role(stored) - .map_err(|e| catalog_err(&format!("put_role '{}'", stored.name), e)) + .map_err(|e| catalog_err(&format!("put_role '{}'", stored.name), e))?; + Ok(ApplyOutcome::Applied) } -pub fn delete(name: &str, catalog: &SystemCatalog) -> crate::Result<()> { +/// Delete the role, unless a user holds it or a role inherits from it. +pub fn delete(name: &str, catalog: &SystemCatalog) -> crate::Result { + let users = catalog.load_all_users()?; + let roles = catalog.load_all_roles()?; + if let Err(refusal) = role_assignment::check_stored_drop(name, &users, &roles) { + tracing::warn!( + role = %name, + %refusal, + "catalog_entry: role drop refused; users or roles still depend on it" + ); + return Ok(ApplyOutcome::Refused(refusal)); + } catalog .delete_role(name) - .map_err(|e| catalog_err(&format!("delete_role '{name}'"), e)) + .map_err(|e| catalog_err(&format!("delete_role '{name}'"), e))?; + Ok(ApplyOutcome::Applied) } diff --git a/nodedb/src/control/catalog_entry/apply/user.rs b/nodedb/src/control/catalog_entry/apply/user.rs index 7b47f8323..91adcb9cf 100644 --- a/nodedb/src/control/catalog_entry/apply/user.rs +++ b/nodedb/src/control/catalog_entry/apply/user.rs @@ -3,11 +3,30 @@ //! Apply User catalog entries to `SystemCatalog` redb. use crate::control::security::catalog::{StoredUser, SystemCatalog, catalog_err}; +use crate::control::security::role_assignment; -pub fn put(stored: &StoredUser, catalog: &SystemCatalog) -> crate::Result<()> { +use super::outcome::ApplyOutcome; + +/// Write the user, unless an active user names a role that is neither built +/// in nor defined in its tenant at this log position. The statement checked +/// before proposing; a role dropped after that check and before this entry +/// committed is caught here, on every node alike. +pub fn put(stored: &StoredUser, catalog: &SystemCatalog) -> crate::Result { + if stored.is_active { + let roles = catalog.load_all_roles()?; + if let Err(refusal) = role_assignment::check_stored_user(stored, &roles) { + tracing::warn!( + user = %stored.username, + %refusal, + "catalog_entry: user entry refused; it names an undefined role" + ); + return Ok(ApplyOutcome::Refused(refusal)); + } + } catalog .put_user(stored) - .map_err(|e| catalog_err(&format!("put_user '{}'", stored.username), e)) + .map_err(|e| catalog_err(&format!("put_user '{}'", stored.username), e))?; + Ok(ApplyOutcome::Applied) } /// Fully remove the user record from redb. `delete_user` is idempotent — a diff --git a/nodedb/src/control/catalog_entry/apply/wal_tombstone.rs b/nodedb/src/control/catalog_entry/apply/wal_tombstone.rs index e6577d7fe..e8e55aaec 100644 --- a/nodedb/src/control/catalog_entry/apply/wal_tombstone.rs +++ b/nodedb/src/control/catalog_entry/apply/wal_tombstone.rs @@ -41,6 +41,8 @@ mod tests { fn record_wal_tombstone_entry_applies_and_is_monotone() { let (store, _tmp) = make_catalog(); let catalog = store.catalog(); + let users = + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::new(7), "users"); // Apply via the top-level apply_to path (entry → apply_to_inner → wal_tombstone::record). let entry = CatalogEntry::RecordWalTombstone { @@ -53,7 +55,7 @@ mod tests { let set = catalog.load_wal_tombstones().expect("load"); assert_eq!( - set.purge_lsn(7, 1, "users"), + set.purge_lsn(users, 1), Some(100), "initial tombstone not recorded" ); @@ -68,7 +70,7 @@ mod tests { apply_to(&entry_lower, catalog).expect("apply record_wal_tombstone (lower)"); let set = catalog.load_wal_tombstones().expect("load after lower"); assert_eq!( - set.purge_lsn(7, 1, "users"), + set.purge_lsn(users, 1), Some(100), "lower purge_lsn must not regress stored tombstone" ); @@ -83,7 +85,7 @@ mod tests { apply_to(&entry_higher, catalog).expect("apply record_wal_tombstone (higher)"); let set = catalog.load_wal_tombstones().expect("load after higher"); assert_eq!( - set.purge_lsn(7, 1, "users"), + set.purge_lsn(users, 1), Some(200), "higher purge_lsn must raise stored tombstone" ); diff --git a/nodedb/src/control/catalog_entry/authorization.rs b/nodedb/src/control/catalog_entry/authorization.rs new file mode 100644 index 000000000..4b9c613cf --- /dev/null +++ b/nodedb/src/control/catalog_entry/authorization.rs @@ -0,0 +1,124 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Which catalog entries change authorization state. +//! +//! An entry that changes who may do what, or what a statement may see, is +//! acknowledged only after it binds every node (see the authorization +//! lease). The match is exhaustive, so a new entry kind is classified when +//! it is added. + +use super::entry::CatalogEntry; + +impl CatalogEntry { + /// Whether applying this entry can change what a statement is allowed to + /// read or write. + pub fn bears_authorization(&self) -> bool { + match self { + // A collection carries its owner and its permission tree. Dropping + // or purging it removes the owner and the grants on it. + Self::PutCollection(_) + | Self::PutCollectionIfAbsent(_) + | Self::DeactivateCollection { .. } + | Self::PurgeCollection { .. } => true, + Self::PutUser(_) + | Self::DropUser { .. } + | Self::PutRole(_) + | Self::DeleteRole { .. } + | Self::PutApiKey(_) + | Self::RevokeApiKey { .. } + | Self::PutAuthUser(_) + | Self::PutOidcProvider(_) + | Self::DeleteOidcProvider { .. } => true, + Self::PutTenant(_) | Self::PutTenantWithAdmin { .. } | Self::DeleteTenant { .. } => { + true + } + Self::PutRlsPolicy(_) + | Self::DeleteRlsPolicy { .. } + | Self::PutRedactionPolicy(_) + | Self::DeleteRedactionPolicy { .. } => true, + Self::PutPermission(_) + | Self::DeletePermission { .. } + | Self::PutScopeGrant(_) + | Self::DeleteScopeGrant { .. } + | Self::PutOwner(_) + | Self::DeleteOwner { .. } => true, + Self::PutDatabase(_) + | Self::DeleteDatabase { .. } + | Self::PutDatabaseGrant { .. } + | Self::DeleteDatabaseGrant { .. } + | Self::CloneDatabase { .. } => true, + Self::PutSequence(_) + | Self::DeleteSequence { .. } + | Self::PutSequenceState(_) + | Self::PutTrigger(_) + | Self::DeleteTrigger { .. } + | Self::PutFunction(_) + | Self::DeleteFunction { .. } + | Self::PutProcedure(_) + | Self::DeleteProcedure { .. } + | Self::PutSchedule(_) + | Self::DeleteSchedule { .. } + | Self::PutChangeStream(_) + | Self::DeleteChangeStream { .. } + | Self::PutMaterializedView(_) + | Self::DeleteMaterializedView { .. } + | Self::PutStreamingMaterializedView(_) + | Self::DeleteStreamingMaterializedView { .. } + | Self::PutContinuousAggregate(_) + | Self::DeleteContinuousAggregate { .. } + | Self::PutIndexRecord(_) + | Self::DeleteIndexRecord { .. } + | Self::PutSynonymGroup(_) + | Self::DeleteSynonymGroup { .. } + | Self::PutCustomType(_) + | Self::DeleteCustomType { .. } + | Self::RecordWalTombstone { .. } + | Self::MoveTenantCutover { .. } + | Self::PutDatabaseQuota { .. } + | Self::DeleteDatabaseQuota { .. } + | Self::PutTenantQuota { .. } + | Self::DeleteTenantQuota { .. } + | Self::PutScopeQuota(_) + | Self::DeleteScopeQuota { .. } + | Self::PutRetentionPolicy(_) + | Self::DeleteRetentionPolicy { .. } + | Self::PutAlertRule(_) + | Self::DeleteAlertRule { .. } + | Self::CreateTopicIfAbsent(_) + | Self::DeleteTopicWithConsumerGroups { .. } + | Self::PutConsumerGroupIfAbsent(_) + | Self::DeleteConsumerGroup { .. } + | Self::MigrateConsumerGroupStream { .. } + | Self::PutCheckpoint(_) + | Self::DeleteCheckpoint { .. } + | Self::CompactHistory { .. } + | Self::PutVectorModel(_) + | Self::DeleteVectorModel { .. } + | Self::PutVectorIndexParams(_) + | Self::PutColumnStats(_) + | Self::DeleteVectorIndexParams { .. } => false, + } + } +} + +#[cfg(test)] +mod tests { + use crate::control::catalog_entry::entry::CatalogEntry; + use crate::control::security::catalog::StoredCollection; + + #[test] + fn a_collection_bears_authorization_and_a_sequence_does_not() { + assert!( + CatalogEntry::PutCollection(Box::new(StoredCollection::new(1, "a", "b"))) + .bears_authorization() + ); + assert!( + !CatalogEntry::DeleteSequence { + database_id: 0, + tenant_id: 1, + name: "c".into(), + } + .bears_authorization() + ); + } +} diff --git a/nodedb/src/control/catalog_entry/mod.rs b/nodedb/src/control/catalog_entry/mod.rs index 5a36c9895..e175ee82a 100644 --- a/nodedb/src/control/catalog_entry/mod.rs +++ b/nodedb/src/control/catalog_entry/mod.rs @@ -31,6 +31,7 @@ //! compile error everywhere a caller needs to handle it. pub mod apply; +pub mod authorization; pub mod codec; pub mod descriptor_stamp; pub mod descriptor_validate; @@ -38,6 +39,7 @@ pub mod entry; pub mod kind; pub mod persist_collection; pub mod post_apply; +pub mod role_rules; pub use codec::{decode, encode}; pub use entry::CatalogEntry; diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/collection.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/collection.rs index 4786c6155..c3b9ace3b 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/collection.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/collection.rs @@ -102,6 +102,7 @@ pub(crate) async fn reclaim_collection_storage( // predecessor writes after a same-name CREATE. shared .wal + .appender(crate::wal::manager::NO_APPLY_KEY) .append_collection_tombstone( TenantId::new(tenant_id), DatabaseId::new(database_id), diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/core_fanout.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/core_fanout.rs index ac7a3eae0..f5a8a0804 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/core_fanout.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/core_fanout.rs @@ -11,12 +11,13 @@ use std::time::Duration; use tracing::debug; -use crate::bridge::envelope::{PhysicalPlan, Priority, Request, Status}; +use crate::bridge::envelope::{PhysicalPlan, Priority, Request, Response, Status}; +use crate::control::ResponseReceiver; use crate::control::state::SharedState; use crate::types::{DatabaseId, ReadConsistency, TenantId, TraceId, VShardId}; /// Deadline for one core's acknowledgement of a post-apply meta op. -const DISPATCH_TIMEOUT: Duration = Duration::from_secs(30); +pub(super) const DISPATCH_TIMEOUT: Duration = Duration::from_secs(30); /// Scope and naming for one fan-out, as every log line and error reports it. pub(super) struct CoreFanout<'a> { @@ -30,6 +31,31 @@ pub(super) struct CoreFanout<'a> { pub detail: &'a str, } +/// Every core's answer to one fan-out, as far as the dispatch deadline saw. +pub(super) struct FanoutAnswers { + /// Cores that were not reached, or that answered anything but `Ok`. + pub(super) refused: Vec, + /// Cores still working at the deadline, each with the receiver its + /// answer arrives on. + pub(super) pending: Vec<(usize, ResponseReceiver)>, +} + +impl FanoutAnswers { + /// Every core that did not apply the plan, once each pending core gave + /// its final answer. Waits with no deadline. A core whose receiver + /// closes without a final answer counts as not applied. + pub(super) async fn into_final_refusals(self) -> Vec { + let mut refused = self.refused; + for (core_id, mut rx) in self.pending { + match final_response(&mut rx).await { + Some(resp) if resp.status == Status::Ok => {} + _ => refused.push(core_id), + } + } + refused + } +} + /// Dispatch `plan` to every core on this node, raising when any core failed to /// apply it. pub(super) async fn dispatch_to_every_core( @@ -47,21 +73,35 @@ pub(super) async fn dispatch_to_every_core( } /// Dispatch `plan` to every core on this node, returning the cores that -/// answered with anything but `Ok`. -/// -/// A caller that dispatches two plans covering each other reads the two lists -/// per core instead of raising on the first refusal. -pub(super) async fn unacked_cores( +/// did not answer `Ok` by the dispatch deadline. +async fn unacked_cores( shared: &SharedState, target: &CoreFanout<'_>, plan: &PhysicalPlan, ) -> Vec { + let answers = fan_out(shared, target, plan).await; + let mut unacked = answers.refused; + unacked.extend(answers.pending.into_iter().map(|(core_id, _)| core_id)); + unacked +} + +/// Dispatch `plan` to every core on this node and collect the answers that +/// arrive by the dispatch deadline. +/// +/// A caller that dispatches two plans covering each other reads the two +/// answer sets per core instead of raising on the first refusal. +pub(super) async fn fan_out( + shared: &SharedState, + target: &CoreFanout<'_>, + plan: &PhysicalPlan, +) -> FanoutAnswers { let num_cores = { let d = shared.dispatcher.lock().unwrap_or_else(|p| p.into_inner()); d.num_cores() }; + let deadline = std::time::Instant::now() + DISPATCH_TIMEOUT; let mut receivers = Vec::with_capacity(num_cores); - let mut unreached: Vec = Vec::new(); + let mut refused: Vec = Vec::new(); { let mut d = shared.dispatcher.lock().unwrap_or_else(|p| p.into_inner()); @@ -73,7 +113,7 @@ pub(super) async fn unacked_cores( database_id: DatabaseId::new(target.database_id), vshard_id: VShardId::new(core_id as u32), plan: plan.clone(), - deadline: std::time::Instant::now() + DISPATCH_TIMEOUT, + deadline, priority: Priority::Background, trace_id: TraceId::generate(), consistency: ReadConsistency::Eventual, @@ -92,16 +132,18 @@ pub(super) async fn unacked_cores( let rx = shared.tracker.register(request_id); if d.dispatch_to_core(core_id, request).is_err() { shared.tracker.cancel(&request_id); - unreached.push(core_id); + refused.push(core_id); continue; } receivers.push((core_id, rx)); } } + let mut pending: Vec<(usize, ResponseReceiver)> = Vec::new(); + let wait_until = tokio::time::Instant::from_std(deadline); for (core_id, mut rx) in receivers { - match tokio::time::timeout(DISPATCH_TIMEOUT, async { rx.recv().await.ok_or(()) }).await { - Ok(Ok(resp)) if resp.status == Status::Ok => { + match tokio::time::timeout_at(wait_until, final_response(&mut rx)).await { + Ok(Some(resp)) if resp.status == Status::Ok => { debug!( tenant = target.tenant_id, collection = %target.collection, @@ -111,9 +153,79 @@ pub(super) async fn unacked_cores( "post-apply core ack" ); } - _ => unreached.push(core_id), + Ok(_) => refused.push(core_id), + Err(_) => pending.push((core_id, rx)), + } + } + + FanoutAnswers { refused, pending } +} + +/// The request's final response, skipping partial frames. `None` once the +/// receiver closes without one. Cancel-safe. +async fn final_response(rx: &mut ResponseReceiver) -> Option { + while let Some(resp) = rx.recv().await { + if !resp.partial { + return Some(resp); } } + None +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::Payload; + use crate::control::RequestTracker; + use crate::types::{Lsn, RequestId}; - unreached + fn answer(id: u64, status: Status, partial: bool) -> Response { + Response { + request_id: RequestId::new(id), + status, + attempt: 1, + partial, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: None, + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + /// A core that outlived the dispatch deadline counts by its final + /// answer. Partial frames before it do not decide anything. + #[tokio::test] + async fn a_pending_core_counts_by_its_final_answer() { + let tracker = RequestTracker::new(); + let applied = tracker.register(RequestId::new(1)); + let refused = tracker.register(RequestId::new(2)); + let answers = FanoutAnswers { + refused: vec![3], + pending: vec![(0, applied), (1, refused)], + }; + let waiter = tokio::spawn(answers.into_final_refusals()); + + assert!(tracker.complete(answer(1, Status::Partial, true))); + assert!(tracker.complete(answer(1, Status::Ok, false))); + assert!(tracker.complete(answer(2, Status::Error, false))); + + assert_eq!(waiter.await.expect("waiter"), vec![3, 1]); + } + + /// A receiver that closes without a final answer leaves the core's + /// outcome unknown, so it counts as not applied. + #[tokio::test] + async fn a_core_whose_receiver_closes_counts_as_not_applied() { + let tracker = RequestTracker::new(); + let rx = tracker.register(RequestId::new(4)); + tracker.cancel(&RequestId::new(4)); + let answers = FanoutAnswers { + refused: Vec::new(), + pending: vec![(2, rx)], + }; + + assert_eq!(answers.into_final_refusals().await, vec![2]); + } } diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/dispatcher.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/dispatcher.rs index 8eaeadbcd..c2afb6bf2 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/dispatcher.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/dispatcher.rs @@ -63,11 +63,12 @@ use super::collection; /// Dispatch post-apply side effects of `entry`. Runs on every node (leader /// and followers) so each node's local Data Plane observes catalog mutations /// symmetrically. -pub fn spawn_post_apply_async_side_effects( - entry: CatalogEntry, - shared: Arc, - raft_index: u64, -) { +/// +/// A storage reclaim takes its purge boundary from this node's own WAL. Replay +/// compares the boundary against WAL record LSNs, so it must be a WAL LSN, +/// never a Raft log index. Every write of the reclaimed collection on this +/// node sits below the next LSN this WAL assigns. +pub fn spawn_post_apply_async_side_effects(entry: CatalogEntry, shared: Arc) { match entry { CatalogEntry::PutCollection(stored) => { // SYNCHRONOUS: Register must complete before the applied-index @@ -124,6 +125,7 @@ pub fn spawn_post_apply_async_side_effects( tenant_id, name, } => { + let purge_lsn = shared.wal.next_lsn().as_u64(); let result = tokio::task::block_in_place(|| { tokio::runtime::Handle::current().block_on(async move { collection::reclaim_collection_storage( @@ -131,7 +133,7 @@ pub fn spawn_post_apply_async_side_effects( database_id, tenant_id, &name, - raft_index, + purge_lsn, false, ) .await @@ -152,13 +154,14 @@ pub fn spawn_post_apply_async_side_effects( tenant_id, name, } => { + let purge_lsn = shared.wal.next_lsn().as_u64(); let result = tokio::task::block_in_place(|| { tokio::runtime::Handle::current().block_on(async move { super::materialized_view::delete_async( database_id, tenant_id, name, - raft_index, + purge_lsn, shared, ) .await @@ -185,7 +188,7 @@ pub fn spawn_post_apply_async_side_effects( CatalogEntry::PutVectorIndexParams(stored) => { tokio::task::block_in_place(|| { tokio::runtime::Handle::current().block_on(async move { - super::vector::put_async(*stored, &shared).await; + super::vector::put_async(*stored, Arc::clone(&shared)).await; }); }); } @@ -204,7 +207,7 @@ pub fn spawn_post_apply_async_side_effects( tenant_id, collection, field_name, - &shared, + Arc::clone(&shared), ) .await; }); @@ -364,7 +367,6 @@ pub fn spawn_post_apply_async_side_effects( | CatalogEntry::PutColumnStats(_) | CatalogEntry::MoveTenantCutover { .. } => { let _ = shared; - let _ = raft_index; } } } diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs index 52fb4f832..25bb5fe9c 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs @@ -26,13 +26,25 @@ //! committed. Every failed stage files a `Capture` instead, because a node //! silently missing an index is the defect this module exists to stop. +use std::sync::Arc; + +use tokio::sync::oneshot; + use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; use crate::control::state::SharedState; -use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; +use crate::types::{DatabaseId, Lsn, TenantId}; use nodedb_physical::physical_plan::VectorOp; use nodedb_types::StoredVectorIndexParams; -use super::core_fanout::{CoreFanout, dispatch_to_every_core, unacked_cores}; +use super::core_fanout::{CoreFanout, DISPATCH_TIMEOUT, FanoutAnswers, fan_out}; + +/// The longest a parameter install waits for its cores to answer: the +/// `SetParams` dispatch deadline, then the reshape's. Its record's window +/// closes within this of its open. +pub(crate) fn longest_core_wait() -> std::time::Duration { + DISPATCH_TIMEOUT.saturating_mul(2) +} /// One vector index, named the way every stage below reports it. struct IndexTarget<'a> { @@ -42,6 +54,30 @@ struct IndexTarget<'a> { field_name: &'a str, } +impl<'a> IndexTarget<'a> { + fn of(entry: &'a StoredVectorIndexParams) -> Self { + Self { + database_id: entry.database_id, + tenant_id: entry.tenant_id, + collection: &entry.collection, + field_name: &entry.field_name, + } + } +} + +/// Where a parameter install stands when a core outlives the dispatch +/// deadline. +enum PutStage { + /// Some cores have not answered `SetParams` yet. + SetParams(FanoutAnswers), + /// Every core answered `SetParams`. `refused` took the reshape instead, + /// and some cores have not answered it yet. + Rebuild { + refused: Vec, + rebuild: FanoutAnswers, + }, +} + /// Build the `SetParams` plan the boot seed and the CREATE handler both /// reproduce, so runtime and restart install identical parameters. fn set_params_plan(entry: &StoredVectorIndexParams) -> PhysicalPlan { @@ -94,66 +130,168 @@ fn drop_index_plan(database_id: u64, collection: &str, field_name: &str) -> Phys /// Install one vector index's build parameters on this node: append the redo /// record, then bring every core to the committed parameters. /// +/// The install owns its record's outcome-floor window, so it runs in a task +/// the caller does not own: a caller dropped mid-install leaves the task to +/// close the window. The call returns once every core answered or the +/// dispatch deadline passed. Cores still working then are waited for by the +/// task. +/// /// The single-node DDL handlers call this directly, where no applier runs and /// the post-apply lane never fires. -pub async fn put_async(entry: StoredVectorIndexParams, shared: &SharedState) { - let target = IndexTarget { - database_id: entry.database_id, - tenant_id: entry.tenant_id, - collection: &entry.collection, - field_name: &entry.field_name, - }; +pub async fn put_async(entry: StoredVectorIndexParams, shared: Arc) { + let (ready_tx, ready_rx) = oneshot::channel(); + tokio::spawn(install_params(entry, shared, ready_tx)); + if ready_rx.await.is_err() { + tracing::error!("the vector index install task ended before it reported"); + } +} + +async fn install_params( + entry: StoredVectorIndexParams, + shared: Arc, + ready: oneshot::Sender<()>, +) { + let target = IndexTarget::of(&entry); let plan = set_params_plan(&entry); // The record makes this node's log self-sufficient: replay rebuilds the - // index from it in LSN order alongside the vector writes around it. - if let Err(error) = append_redo(shared, &target, &plan) { + // index from it in LSN order alongside the vector writes around it. Its + // outcome-floor window closes once every core gave its final answer. + let minted = MintedRecords::open(&shared.outcome_floor); + if let Err(error) = append_redo(&shared, &target, &plan, &minted) { report(&error, "set_params_wal_append", &target); } - let target_fanout = fanout(&target); - let refused = unacked_cores(shared, &target_fanout, &plan).await; - if refused.is_empty() { - return; - } + // The cores hold the record from here. It closes from their answers. + minted.mark_sent(); + let set_params = fan_out(&shared, &fanout(&target), &plan).await; + let stage = if set_params.pending.is_empty() { + let refused = set_params.refused; + if refused.is_empty() { + minted.settle(); + // The caller can be gone. The install completes either way. + let _ = ready.send(()); + return; + } + // Every refused core already holds a materialized index, which only + // the in-place reshape reaches. A core that took `SetParams` answers + // this with `NotFound` and stays on the parameters it accepted. + let rebuild = fan_out(&shared, &fanout(&target), &rebuild_plan(&entry)).await; + if rebuild.pending.is_empty() { + close_put(&target, refused, rebuild.refused, minted); + let _ = ready.send(()); + return; + } + PutStage::Rebuild { refused, rebuild } + } else { + PutStage::SetParams(set_params) + }; + // Cores outlived the dispatch deadline. The caller moves on while this + // task waits for their final answers. + let _ = ready.send(()); + finish_put(&shared, &entry, stage, minted).await; +} + +/// Resolve a handed-off parameter install once every core answered. +async fn finish_put( + shared: &SharedState, + entry: &StoredVectorIndexParams, + stage: PutStage, + minted: MintedRecords, +) { + let target = IndexTarget::of(entry); + let (refused, rebuild) = match stage { + PutStage::SetParams(set_params) => { + let refused = set_params.into_final_refusals().await; + if refused.is_empty() { + minted.settle(); + return; + } + let rebuild = fan_out(shared, &fanout(&target), &rebuild_plan(entry)).await; + (refused, rebuild) + } + PutStage::Rebuild { refused, rebuild } => (refused, rebuild), + }; + let reshaped = rebuild.into_final_refusals().await; + close_put(&target, refused, reshaped, minted); +} - // Every refused core already holds a materialized index, which only the - // in-place reshape reaches. A core that took `SetParams` answers this with - // `NotFound` and stays on the parameters it just accepted. - let reshaped = unacked_cores(shared, &target_fanout, &rebuild_plan(&entry)).await; +/// Close a parameter install's window from both answer sets. A core that +/// refused `SetParams` and the reshape missed the change. +fn close_put( + target: &IndexTarget<'_>, + refused: Vec, + reshaped: Vec, + minted: MintedRecords, +) { let missed: Vec = refused .into_iter() .filter(|core_id| reshaped.contains(core_id)) .collect(); - if !missed.is_empty() { - let error = crate::Error::Internal { - detail: format!("cores did not apply the vector index change: {missed:?}"), - }; - report(&error, "set_params_dispatch", &target); + if missed.is_empty() { + minted.settle(); + return; } + let error = crate::Error::Internal { + detail: format!("cores did not apply the vector index change: {missed:?}"), + }; + report(&error, "set_params_dispatch", target); + // A core that missed the change still needs restart replay to reach the + // record. + minted.hold(); } /// Remove one vector index from this node: append and fsync the drop record, /// then dispatch `DropIndex` to every core. +/// +/// Runs in a task the caller does not own, and returns once every core +/// answered or the dispatch deadline passed, as [`put_async`] does. pub async fn delete_async( database_id: u64, tenant_id: u64, collection: String, field_name: String, - shared: &SharedState, + shared: Arc, ) { + let (ready_tx, ready_rx) = oneshot::channel(); + tokio::spawn(drop_index( + IndexName { + database_id, + tenant_id, + collection, + field_name, + }, + shared, + ready_tx, + )); + if ready_rx.await.is_err() { + tracing::error!("the vector index drop task ended before it reported"); + } +} + +/// An owned index name, for the task that drops the index. +struct IndexName { + database_id: u64, + tenant_id: u64, + collection: String, + field_name: String, +} + +async fn drop_index(name: IndexName, shared: Arc, ready: oneshot::Sender<()>) { let target = IndexTarget { - database_id, - tenant_id, - collection: &collection, - field_name: &field_name, + database_id: name.database_id, + tenant_id: name.tenant_id, + collection: &name.collection, + field_name: &name.field_name, }; - let plan = drop_index_plan(database_id, &collection, &field_name); + let plan = drop_index_plan(name.database_id, &name.collection, &name.field_name); // The vector writes this drop cancels are already fsynced in this node's // log, so replay rebuilds the dropped index unless the drop record is - // durable too. Append and fsync before touching the cores. - match append_redo(shared, &target, &plan) { + // durable too. Append and fsync before touching the cores. The record's + // outcome-floor window closes once every core gave its final answer. + let minted = MintedRecords::open(&shared.outcome_floor); + match append_redo(&shared, &target, &plan, &minted) { Ok(Some(lsn)) => { if let Err(error) = shared.wal.wait_durable(lsn).await { report(&error, "drop_index_fsync", &target); @@ -168,25 +306,52 @@ pub async fn delete_async( Err(error) => report(&error, "drop_index_wal_append", &target), } - if let Err(error) = dispatch_to_every_core(shared, &fanout(&target), &plan).await { - report(&error, "drop_index_dispatch", &target); + // The cores hold the record from here. It closes from their answers. + minted.mark_sent(); + let answers = fan_out(&shared, &fanout(&target), &plan).await; + // Cores still working past the deadline answer later. The caller moves + // on while this task waits for their final answers. + let _ = ready.send(()); + let refused = answers.into_final_refusals().await; + close_drop(&target, refused, minted); +} + +/// Close a drop's window from the cores that did not drop the index. +fn close_drop(target: &IndexTarget<'_>, refused: Vec, minted: MintedRecords) { + if refused.is_empty() { + minted.settle(); + return; } + let error = crate::Error::Internal { + detail: format!("cores did not apply the vector index change: {refused:?}"), + }; + report(&error, "drop_index_dispatch", target); + // A core that kept the index still needs restart replay to reach the + // drop record. + minted.hold(); } -/// Append `plan`'s redo record to this node's WAL, returning its LSN. +/// Append `plan`'s redo record to this node's WAL under `minted`'s window, +/// returning its LSN. fn append_redo( shared: &SharedState, target: &IndexTarget<'_>, plan: &PhysicalPlan, + minted: &MintedRecords, ) -> crate::Result> { let database_id = DatabaseId::new(target.database_id); - let vshard = VShardId::from_collection_in_database(database_id, target.collection); - let outcome = crate::control::server::wal_dispatch::wal_append_if_write( - &shared.wal, - TenantId::new(target.tenant_id), - vshard, + let owner = RecordOwner { + tenant_id: TenantId::new(target.tenant_id), database_id, + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, target.collection).vshard(), + }; + let outcome = minted.append_plan( + &shared.wal, + owner, plan, + // A vector index change writes no row; its records carry no row + // image, and the source names the committed DDL that ran it. + crate::event::EventSource::User, )?; Ok(outcome.lsn) } diff --git a/nodedb/src/control/catalog_entry/post_apply/collection.rs b/nodedb/src/control/catalog_entry/post_apply/collection.rs index 4eaabf5d4..a33819699 100644 --- a/nodedb/src/control/catalog_entry/post_apply/collection.rs +++ b/nodedb/src/control/catalog_entry/post_apply/collection.rs @@ -6,8 +6,10 @@ use std::sync::Arc; use tracing::debug; +use crate::control::security::auth_fence::TreeDefChange; use crate::control::security::catalog::{StoredCollection, StoredOwner}; use crate::control::state::SharedState; +use crate::types::DatabaseId; /// Synchronous half of `PutCollection` post-apply: install the owner /// record into the in-memory `PermissionStore`. Called inline by the @@ -42,6 +44,50 @@ pub fn put_owner_sync(stored: &StoredCollection, shared: Arc) { } } +/// Queue the tree-definition change a committed collection descriptor makes. +/// Every node runs this, so each node's permission cache learns the tree +/// defined through any node. Planning and lease coverage move the queue into +/// the cache. +pub fn queue_tree_def_sync(stored: &StoredCollection, shared: &SharedState) { + match TreeDefChange::from_collection(stored) { + Ok(Some(change)) => { + change.note_committed(shared.authorization_fence.sources()); + shared.authorization_fence.tree_defs().push(change); + } + Ok(None) => {} + Err(e) => { + // The DDL commits the serialization of a parsed definition, so + // this JSON always parses. The prior definition stays in place: + // removing it would drop the filter and expose rows. + tracing::error!( + collection = %stored.name, + tenant = stored.tenant_id, + error = %e, + "post_apply: PERMISSION_TREE of a committed collection could not be read" + ); + } + } +} + +/// Queue the removal of a collection's tree definition, for a collection +/// that was dropped or purged. +pub fn queue_tree_def_removal_sync( + database_id: u64, + tenant_id: u64, + name: &str, + shared: &SharedState, +) { + if database_id != DatabaseId::DEFAULT.as_u64() { + return; + } + let change = TreeDefChange::Unregister { + tenant_id, + collection: name.to_owned(), + }; + change.note_committed(shared.authorization_fence.sources()); + shared.authorization_fence.tree_defs().push(change); +} + /// Register-dispatch half: dispatch a `Register` request to this node's /// Data Plane so subsequent `DocumentOp::Scan` calls find the collection /// in `doc_configs` and decode strict (Binary Tuple) documents correctly. diff --git a/nodedb/src/control/catalog_entry/post_apply/mod.rs b/nodedb/src/control/catalog_entry/post_apply/mod.rs index f0e63bd7c..564e1af61 100644 --- a/nodedb/src/control/catalog_entry/post_apply/mod.rs +++ b/nodedb/src/control/catalog_entry/post_apply/mod.rs @@ -56,5 +56,6 @@ pub(crate) use async_dispatch::crdt_compact::compact_async; pub use async_dispatch::spawn_post_apply_async_side_effects; pub(crate) use async_dispatch::synonym_group::delete_async as remove_synonym_group; pub(crate) use async_dispatch::synonym_group::put_async as install_synonym_group; +pub(crate) use async_dispatch::vector::longest_core_wait as vector_install_longest_core_wait; pub(crate) use async_dispatch::vector::put_async as install_vector_index_params; pub use sync::apply_post_apply_side_effects_sync; diff --git a/nodedb/src/control/catalog_entry/post_apply/sync.rs b/nodedb/src/control/catalog_entry/post_apply/sync.rs index b75ff0a2d..0f9d854f8 100644 --- a/nodedb/src/control/catalog_entry/post_apply/sync.rs +++ b/nodedb/src/control/catalog_entry/post_apply/sync.rs @@ -45,6 +45,7 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { // Install owner from the CANONICAL catalog collection, not the @@ -60,15 +61,18 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc collection::put_owner_sync(&canonical, Arc::clone(shared)), - None => collection::put_owner_sync(stored, Arc::clone(shared)), - } + let canonical = canonical.as_ref().unwrap_or(&**stored); + collection::put_owner_sync(canonical, Arc::clone(shared)); + collection::queue_tree_def_sync(canonical, shared); } CatalogEntry::DeactivateCollection { - tenant_id, name, .. + database_id, + tenant_id, + name, + .. } => { collection::deactivate(*tenant_id, name.clone(), Arc::clone(shared)); + collection::queue_tree_def_removal_sync(*database_id, *tenant_id, name, shared); } CatalogEntry::PurgeCollection { database_id, @@ -76,6 +80,7 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { collection::purge_sync(*database_id, *tenant_id, name.clone(), Arc::clone(shared)); + collection::queue_tree_def_removal_sync(*database_id, *tenant_id, name, shared); } CatalogEntry::PutSequence(stored) => { sequence::put((**stored).clone(), Arc::clone(shared)); diff --git a/nodedb/src/control/catalog_entry/role_rules.rs b/nodedb/src/control/catalog_entry/role_rules.rs new file mode 100644 index 000000000..e6a1b2d76 --- /dev/null +++ b/nodedb/src/control/catalog_entry/role_rules.rs @@ -0,0 +1,74 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Role rules over a batch of catalog entries, in order. +//! +//! A transaction buffers its DDL and applies it at COMMIT. One statement can +//! break a role rule for another in the same batch: `DROP ROLE r` after +//! `CREATE USER u ROLE r`, for example. Each statement's own check saw only +//! the state before the batch. This check replays the batch's role and user +//! entries over the committed catalog, in order, and refuses the whole batch +//! before anything is written, so a COMMIT never applies part of a batch. + +use std::collections::HashMap; + +use crate::control::security::catalog::{StoredRole, StoredUser, SystemCatalog}; +use crate::control::security::role_assignment::{self, RoleRefusal}; + +use super::entry::CatalogEntry; + +/// Check every role and user entry of `entries` against the committed +/// catalog and the entries before it. +pub fn check_batch<'a>( + entries: impl IntoIterator, + catalog: &SystemCatalog, +) -> crate::Result<()> { + let mut users: HashMap = catalog + .load_all_users()? + .into_iter() + .map(|user| (user.username.clone(), user)) + .collect(); + let mut roles: HashMap = catalog + .load_all_roles()? + .into_iter() + .map(|role| (role.name.clone(), role)) + .collect(); + for entry in entries { + match entry { + CatalogEntry::PutUser(user) => { + if user.is_active { + let defined: Vec = roles.values().cloned().collect(); + role_assignment::check_stored_user(user, &defined)?; + } + users.insert(user.username.clone(), (**user).clone()); + } + CatalogEntry::PutTenantWithAdmin { admin, .. } => { + users.insert(admin.username.clone(), (**admin).clone()); + } + CatalogEntry::DropUser { username } => { + users.remove(username); + } + CatalogEntry::PutRole(role) => { + let parent = role.parent.as_str(); + if !parent.is_empty() + && !role_assignment::is_builtin_role_name(parent) + && !roles.contains_key(parent) + { + return Err(RoleRefusal::Undefined { + name: parent.to_string(), + } + .into()); + } + roles.insert(role.name.clone(), (**role).clone()); + } + CatalogEntry::DeleteRole { name } => { + let held: Vec = users.values().cloned().collect(); + let defined: Vec = roles.values().cloned().collect(); + role_assignment::check_stored_drop(name, &held, &defined)?; + roles.remove(name); + } + // Every other entry leaves users and roles as they are. + _ => {} + } + } + Ok(()) +} diff --git a/nodedb/src/control/catalog_overlay/index_record.rs b/nodedb/src/control/catalog_overlay/index_record.rs index 037bddec7..8377662c7 100644 --- a/nodedb/src/control/catalog_overlay/index_record.rs +++ b/nodedb/src/control/catalog_overlay/index_record.rs @@ -51,6 +51,49 @@ pub fn resolve_index_record( ) } +/// Every index record of one `(database, tenant)`, with this connection's +/// uncommitted DDL replayed over the committed list in statement order. A +/// buffered create is listed, a buffered drop is not, and the result stays +/// in name order. +pub fn resolve_index_records( + database_id: u64, + tenant_id: u64, + committed: Vec, +) -> Vec { + let replayed = crate::control::server::shared::session::ddl_buffer::with_buffered(|buffered| { + let mut by_name: std::collections::BTreeMap = committed + .iter() + .map(|record| (record.name.clone(), record.clone())) + .collect(); + let mut touched = false; + for item in buffered { + match &item.entry { + CatalogEntry::PutIndexRecord(stored) + if stored.database_id == database_id && stored.tenant_id == tenant_id => + { + by_name.insert(stored.name.clone(), (**stored).clone()); + touched = true; + } + CatalogEntry::DeleteIndexRecord { + database_id: entry_db, + tenant_id: entry_tenant, + name, + .. + } if *entry_db == database_id && *entry_tenant == tenant_id => { + by_name.remove(name); + touched = true; + } + _ => {} + } + } + touched.then(|| by_name.into_values().collect::>()) + }); + match replayed { + Some(Some(records)) => records, + Some(None) | None => committed, + } +} + #[cfg(test)] mod tests { use super::*; @@ -108,6 +151,21 @@ mod tests { .await; } + #[tokio::test] + async fn the_listing_shows_a_buffered_create_and_hides_a_buffered_drop() { + conn_scope::scoped(async { + ddl_buffer::activate(); + ddl_buffer::try_buffer(put("idx_new")); + ddl_buffer::try_buffer(delete("idx_old")); + let names: Vec = resolve_index_records(0, 1, vec![stored("idx_old")]) + .into_iter() + .map(|record| record.name) + .collect(); + assert_eq!(names, vec!["idx_new".to_string()]); + }) + .await; + } + #[tokio::test] async fn outside_a_transaction_the_committed_row_wins() { conn_scope::scoped(async { diff --git a/nodedb/src/control/catalog_overlay/mod.rs b/nodedb/src/control/catalog_overlay/mod.rs index 744ee3c7d..f09f42da5 100644 --- a/nodedb/src/control/catalog_overlay/mod.rs +++ b/nodedb/src/control/catalog_overlay/mod.rs @@ -26,10 +26,12 @@ mod index_record; mod materialized_view; mod procedure; mod trigger; +mod vector_index_params; pub use self::collection::{resolve_collection, resolve_tenant_collections}; pub use self::function::resolve_function; -pub use self::index_record::resolve_index_record; +pub use self::index_record::{resolve_index_record, resolve_index_records}; pub use self::materialized_view::resolve_materialized_view; pub use self::procedure::resolve_procedure; pub use self::trigger::resolve_trigger; +pub use self::vector_index_params::resolve_vector_index_params; diff --git a/nodedb/src/control/catalog_overlay/vector_index_params.rs b/nodedb/src/control/catalog_overlay/vector_index_params.rs new file mode 100644 index 000000000..791cf3323 --- /dev/null +++ b/nodedb/src/control/catalog_overlay/vector_index_params.rs @@ -0,0 +1,135 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Uncommitted-DDL overlay for vector index build parameters. +//! +//! See [`super::collection`] for the mechanism this mirrors: a `CREATE +//! VECTOR INDEX` buffered inside an open transaction must be visible to a +//! later `ALTER VECTOR INDEX` or duplicate check in that same transaction, +//! before COMMIT writes the row. + +use nodedb_types::StoredVectorIndexParams; + +use crate::control::catalog_entry::CatalogEntry; + +/// The vector index one `(database, tenant, collection, field)` names. +struct Target<'a> { + database_id: u64, + tenant_id: u64, + collection: &'a str, + field_name: &'a str, +} + +/// True when `entry` mutates the vector index `target` names. +fn targets(entry: &CatalogEntry, target: &Target<'_>) -> bool { + match entry { + CatalogEntry::PutVectorIndexParams(stored) => { + stored.database_id == target.database_id + && stored.tenant_id == target.tenant_id + && stored.collection == target.collection + && stored.field_name == target.field_name + } + CatalogEntry::DeleteVectorIndexParams { + database_id, + tenant_id, + collection, + field_name, + } => { + *database_id == target.database_id + && *tenant_id == target.tenant_id + && collection == target.collection + && field_name == target.field_name + } + _ => false, + } +} + +/// Replay one buffered entry over the state resolved so far. +fn step( + current: Option, + entry: &CatalogEntry, +) -> Option { + match entry { + CatalogEntry::PutVectorIndexParams(stored) => Some((**stored).clone()), + CatalogEntry::DeleteVectorIndexParams { .. } => None, + _ => current, + } +} + +/// Resolve one vector index's build parameters through this connection's +/// uncommitted DDL. +pub fn resolve_vector_index_params( + database_id: u64, + tenant_id: u64, + collection: &str, + field_name: &str, + committed: Option, +) -> Option { + let target = Target { + database_id, + tenant_id, + collection, + field_name, + }; + super::core::resolve(committed, |entry| targets(entry, &target), step) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::server::shared::session::{conn_scope, ddl_buffer}; + + fn params(m: usize) -> StoredVectorIndexParams { + StoredVectorIndexParams { + database_id: 0, + tenant_id: 1, + collection: "docs".to_owned(), + field_name: "emb".to_owned(), + dim: 3, + metric: "cosine".to_owned(), + m, + ef_construction: 200, + index_type: "hnsw".to_owned(), + pq_m: 0, + ivf_cells: 0, + ivf_nprobe: 0, + } + } + + fn resolve(committed: Option) -> Option { + resolve_vector_index_params(0, 1, "docs", "emb", committed) + } + + #[tokio::test] + async fn a_buffered_create_then_alter_resolves_to_the_altered_row() { + conn_scope::scoped(async { + ddl_buffer::activate(); + ddl_buffer::try_buffer(CatalogEntry::PutVectorIndexParams(Box::new(params(16)))); + ddl_buffer::try_buffer(CatalogEntry::PutVectorIndexParams(Box::new(params(32)))); + assert_eq!(resolve(None).map(|p| p.m), Some(32)); + }) + .await; + } + + #[tokio::test] + async fn a_buffered_drop_hides_the_committed_row() { + conn_scope::scoped(async { + ddl_buffer::activate(); + ddl_buffer::try_buffer(CatalogEntry::DeleteVectorIndexParams { + database_id: 0, + tenant_id: 1, + collection: "docs".to_owned(), + field_name: "emb".to_owned(), + }); + assert!(resolve(Some(params(16))).is_none()); + }) + .await; + } + + #[tokio::test] + async fn outside_a_transaction_the_committed_row_wins() { + conn_scope::scoped(async { + assert_eq!(resolve(Some(params(16))).map(|p| p.m), Some(16)); + }) + .await; + } +} diff --git a/nodedb/src/control/checkpoint_manager.rs b/nodedb/src/control/checkpoint_manager.rs index dbbe5df59..310049ba5 100644 --- a/nodedb/src/control/checkpoint_manager.rs +++ b/nodedb/src/control/checkpoint_manager.rs @@ -129,6 +129,30 @@ pub struct CheckpointCycleInputs<'a> { pub cold_storage: Option>, /// When present, the tombstone set is GC'd to the new truncation point. pub catalog: Option<&'a crate::control::security::catalog::SystemCatalog>, + /// The applied state of every Calvin scheduler on this node. Saved in + /// `catalog` before truncation deletes the applied markers it came from. + pub calvin_mirrors: Option<&'a crate::control::cluster::calvin::scheduler::AppliedMirrors>, +} + +/// The applied state of every mirror, as the catalog stores it. +fn calvin_applied_states( + mirrors: &crate::control::cluster::calvin::scheduler::AppliedMirrors, +) -> Vec { + mirrors.snapshot_all() +} + +/// Save `states` in `catalog`. +fn save_calvin_applied( + states: Vec, + catalog: Option<&crate::control::security::catalog::SystemCatalog>, +) -> crate::Result<()> { + if states.is_empty() { + return Ok(()); + } + let catalog = catalog.ok_or_else(|| crate::Error::Internal { + detail: "checkpoint has no catalog to save the Calvin applied state in".into(), + })?; + catalog.save_calvin_applied(states) } /// Run one checkpoint cycle: dispatch checkpoint to all cores, collect LSNs, @@ -147,6 +171,7 @@ pub async fn run_checkpoint_cycle(inputs: CheckpointCycleInputs<'_>) -> Option) -> Option { debug!( marker_lsn = marker_lsn.as_u64(), @@ -324,6 +356,22 @@ pub async fn run_checkpoint_cycle(inputs: CheckpointCycleInputs<'_>) -> Option ready.map_err(|error| { + warn!(%error, "checkpoint manager not started: startup did not complete"); + }), + // The WAL on disk is intact, so restart replay covers what a final + // cycle would have made redundant. + _ = guard.await_signal() => { + info!("shutdown before startup completed: no final checkpoint"); + Err(()) + } + }; + if started.is_err() { + guard.report_drained(); + return; + } info!( interval_secs = config.interval.as_secs(), "checkpoint manager started" @@ -60,6 +81,7 @@ pub fn spawn_checkpoint_task( timeout: budget, cold_storage: shared.cold_storage.clone(), catalog: Some(shared.credentials.catalog()), + calvin_mirrors: Some(shared.authorization_fence.calvin_mirrors()), }); if tokio::time::timeout(budget, cycle).await.is_err() { warn!( @@ -82,6 +104,7 @@ pub fn spawn_checkpoint_task( timeout: config.core_timeout, cold_storage: shared.cold_storage.clone(), catalog: Some(shared.credentials.catalog()), + calvin_mirrors: Some(shared.authorization_fence.calvin_mirrors()), }) .await; } diff --git a/nodedb/src/control/clone/copyup.rs b/nodedb/src/control/clone/copyup.rs index 953f18b03..ebcde51b9 100644 --- a/nodedb/src/control/clone/copyup.rs +++ b/nodedb/src/control/clone/copyup.rs @@ -18,7 +18,7 @@ use nodedb_types::{DatabaseId, Surrogate, TenantId}; use crate::bridge::envelope::{Priority, Request, Status}; use crate::control::state::SharedState; -use crate::types::{ReadConsistency, RequestId, TraceId, VShardId}; +use crate::types::{ReadConsistency, RequestId, TraceId}; use nodedb_physical::physical_plan::{DocumentOp, KvOp, PhysicalPlan}; /// Parameters for a KV copy-up operation. @@ -47,15 +47,12 @@ pub async fn perform_kv_clone_copyup(params: KvCopyUpParams<'_>) -> crate::Resul source_value_bytes, } = params; - let target_coll_qualified = crate::control::planner::sql_plan_convert::convert::db_qualified( - target_db_id, - target_collection, - ); + let target_key = nodedb_types::CollectionKey::from_bare(target_db_id, target_collection); // Allocate a surrogate for the target KV row. let surrogate = state .surrogate_assigner - .assign(target_db_id, tenant_id, &target_coll_qualified, &kv_key) + .assign(target_key, tenant_id, &kv_key) .map_err(|e| crate::Error::Storage { engine: "clone_kv_copyup".into(), detail: format!("surrogate alloc failed: {e}"), @@ -69,9 +66,10 @@ pub async fn perform_kv_clone_copyup(params: KvCopyUpParams<'_>) -> crate::Resul surrogate, returning: None, rls_filters: Vec::new(), + provenance: None, }); - let vshard_id = VShardId::from_collection_in_database(target_db_id, &target_coll_qualified); + let vshard_id = target_key.vshard(); let req_id = RequestId::new( state .request_id_counter @@ -158,18 +156,14 @@ pub async fn perform_clone_copyup(params: CopyUpParams<'_>) -> crate::Result) -> crate::Result) -> crate::Res let system_time = rewrite_system_time(effective_source_ms, *system_time)?; let Some(source_surrogate) = state .surrogate_assigner - .lookup(source_db_id, tenant_id, source_qualified.as_str(), pk_bytes) + .lookup( + nodedb_types::CollectionKey::from_bare(source_db_id, source_coll), + tenant_id, + pk_bytes, + ) .ok() .flatten() else { diff --git a/nodedb/src/control/cluster/array_cluster_exec/dispatch.rs b/nodedb/src/control/cluster/array_cluster_exec/dispatch.rs index 93a921ea6..2efa24a78 100644 --- a/nodedb/src/control/cluster/array_cluster_exec/dispatch.rs +++ b/nodedb/src/control/cluster/array_cluster_exec/dispatch.rs @@ -145,6 +145,12 @@ impl NexarArrayDispatch { detail: "array shard response: failed to decode VShardEnvelope".into(), } }), + // The shard's handler failed with a typed error. It is rebuilt as + // the error the local short-circuit returns, so `WrongOwner` + // reaches the fan-out's reroute retry. + RaftRpc::VShardRefusal(refusal) => { + Err(nodedb_cluster::error::ClusterError::from(refusal.error)) + } other => Err(nodedb_cluster::error::ClusterError::Transport { detail: format!( "array shard RPC: unexpected response type {:?}", diff --git a/nodedb/src/control/cluster/array_cluster_helpers.rs b/nodedb/src/control/cluster/array_cluster_helpers.rs index 16aa80c1b..482467674 100644 --- a/nodedb/src/control/cluster/array_cluster_helpers.rs +++ b/nodedb/src/control/cluster/array_cluster_helpers.rs @@ -5,6 +5,7 @@ //! fast path. use nodedb_cluster::distributed_array::merge::ArrayAggPartial; +use nodedb_cluster::error::ClusterError; use nodedb_cluster::wire::VShardMessageType; use crate::Error; @@ -94,15 +95,75 @@ pub(super) fn finalize_agg_partials( rows } -pub(super) fn cluster_err(e: nodedb_cluster::error::ClusterError) -> Error { +pub(super) fn cluster_err(e: ClusterError) -> Error { match e { // A shard did not answer within its timeout: surface as a deterministic // deadline rather than an opaque internal error, matching the // `TypedClusterError::DeadlineExceeded` mapping used elsewhere. - nodedb_cluster::error::ClusterError::ShardTimeout { .. } => Error::DeadlineExceeded { + ClusterError::ShardTimeout { .. } => Error::DeadlineExceeded { request_id: crate::types::RequestId::new(0), }, - other => Error::Internal { + // A shard's Data-Plane verdict keeps its code, so the statement + // renders the SQLSTATE a single-node execution renders. + ClusterError::DataPlane { code } => Error::DataPlane(code.into()), + // A shard's typed execution error is rebuilt, so the statement + // renders the SQLSTATE a single-node execution renders. + ClusterError::ShardExecution { error, .. } | ClusterError::StreamTerminal { error, .. } => { + Error::from(*error) + } + // The shard still refused after the fan-out's reroute retry. The + // vShard's owner is moving, so the client retries the statement. + ClusterError::WrongOwner { + vshard_id, + expected_owner_node, + } => match expected_owner_node { + Some(leader_node) => Error::NotLeader { + vshard_id: crate::types::VShardId::new(vshard_id), + leader_node, + leader_addr: String::new(), + }, + None => Error::NoLeader { + vshard_id: crate::types::VShardId::new(vshard_id), + }, + }, + // The vShard is moving to another node. It has no serving owner + // until the cut-over, so the client retries the statement. + ClusterError::MigrationInProgress { vshard_id } => Error::NoLeader { + vshard_id: crate::types::VShardId::new(vshard_id), + }, + // Cluster machinery faults. The client can act on none of them. + other @ (ClusterError::Raft(_) + | ClusterError::VShardNotMapped { .. } + | ClusterError::GroupNotFound { .. } + | ClusterError::LearnerNotCaughtUp { .. } + | ClusterError::MigrationPauseBudgetExceeded { .. } + | ClusterError::NodeUnreachable { .. } + | ClusterError::GhostNotFound { .. } + | ClusterError::Transport { .. } + | ClusterError::Storage { .. } + | ClusterError::Codec { .. } + | ClusterError::UnsupportedWireVersion { .. } + | ClusterError::CircuitOpen { .. } + | ClusterError::JoinGroupDisappeared { .. } + | ClusterError::JoinCommitTimeout { .. } + | ClusterError::ReadIndexNotLeader { .. } + | ClusterError::ReadIndexTimeout { .. } + | ClusterError::Config { .. } + | ClusterError::MigrationCheckpoint(_) + | ClusterError::MigrationRecovery(_) + | ClusterError::Calvin(_) + | ClusterError::SnapshotCrcMismatch { .. } + | ClusterError::SnapshotOffsetRegression { .. } + | ClusterError::PartialSnapshotCorrupt { .. } + | ClusterError::PartialSnapshotCleanupFailed { .. } + | ClusterError::SnapshotApplyFailed { .. } + | ClusterError::Mirror(_) + | ClusterError::BspBarrier(_) + | ClusterError::VectorGather(_) + | ClusterError::SpatialGather(_) + | ClusterError::Bm25Gather(_) + | ClusterError::TsGather(_) + | ClusterError::RemoteUntyped { .. }) => Error::Internal { detail: format!("array cluster: {other}"), }, } @@ -129,3 +190,71 @@ pub(super) fn array_resp_msg_type(opcode: u32) -> Option { _ => None, } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::ErrorCode; + + /// A shard verdict that crossed the cluster keeps its code at the + /// coordinator, never `Internal`. + #[test] + fn a_shard_verdict_keeps_its_code() { + let code = ErrorCode::Unsupported { + detail: "not on this engine".into(), + }; + let wire = nodedb_cluster::error::ClusterError::DataPlane { + code: code.clone().into(), + }; + match cluster_err(wire) { + Error::DataPlane(rebuilt) => assert_eq!(rebuilt, code), + other => panic!("expected the typed verdict, got {other:?}"), + } + } + + /// A shard that still refused after the reroute retry answers the + /// retryable leader class, never `Internal`. + #[test] + fn a_persistent_wrong_owner_is_a_leader_error() { + let known = nodedb_cluster::error::ClusterError::WrongOwner { + vshard_id: 7, + expected_owner_node: Some(3), + }; + assert!(matches!( + cluster_err(known), + Error::NotLeader { leader_node: 3, .. } + )); + let unknown = nodedb_cluster::error::ClusterError::WrongOwner { + vshard_id: 7, + expected_owner_node: None, + }; + assert!(matches!(cluster_err(unknown), Error::NoLeader { .. })); + } + + /// A vShard mid-migration has no serving owner, so the statement answers + /// the retryable no-leader class, never `Internal`. + #[test] + fn a_migrating_vshard_is_a_no_leader_error() { + let wire = ClusterError::MigrationInProgress { vshard_id: 5 }; + match cluster_err(wire) { + Error::NoLeader { vshard_id } => assert_eq!(vshard_id.as_u32(), 5), + other => panic!("expected NoLeader, got {other:?}"), + } + } + + /// A typed terminal error is rebuilt, never flattened to `Internal`. + #[test] + fn a_typed_terminal_error_is_rebuilt() { + let typed = nodedb_cluster::rpc_codec::TypedClusterError::DataPlane { + code: ErrorCode::DivisionByZero.into(), + }; + let wire = ClusterError::StreamTerminal { + error: Box::new(typed), + detail: "division by zero".into(), + }; + match cluster_err(wire) { + Error::DataPlane(code) => assert_eq!(code, ErrorCode::DivisionByZero), + other => panic!("expected the typed verdict, got {other:?}"), + } + } +} diff --git a/nodedb/src/control/cluster/array_executor/executor.rs b/nodedb/src/control/cluster/array_executor/executor.rs index 46d0a381b..de04c03fc 100644 --- a/nodedb/src/control/cluster/array_executor/executor.rs +++ b/nodedb/src/control/cluster/array_executor/executor.rs @@ -7,8 +7,10 @@ use std::time::{Duration, Instant}; use nodedb_array::types::ArrayId; use nodedb_cluster::error::{ClusterError, Result}; +use nodedb_cluster::rpc_codec::DataPlaneErrorCode; -use crate::bridge::envelope::{Priority, Request}; +use super::refusal::execution_error; +use crate::bridge::envelope::{Priority, Request, Response}; use crate::control::state::SharedState; use crate::event::types::EventSource; use crate::types::{ReadConsistency, RequestId, TraceId, TxnId, VShardId}; @@ -47,7 +49,7 @@ impl DataPlaneArrayExecutor { local_vshard_id: VShardId, plan: PhysicalPlan, txn_id: Option, - ) -> Result { + ) -> Result { let request_id = self.state.next_request_id(); let request = local_request(request_id, array_id, local_vshard_id, plan, txn_id); @@ -58,23 +60,28 @@ impl DataPlaneArrayExecutor { Err(poisoned) => poisoned.into_inner().dispatch(request), }; + // A dispatch refusal, such as a capacity limit, keeps its own class. if let Err(e) = dispatch_result { - return Err(ClusterError::Storage { - detail: format!("array executor dispatch: {e}"), - }); + return Err(execution_error("array executor dispatch", e)); } - match tokio::time::timeout(LOCAL_DISPATCH_TIMEOUT, async { rx.recv().await.ok_or(()) }) - .await - { - Ok(Ok(resp)) => Ok(resp), - Ok(Err(_)) => Err(ClusterError::Storage { - detail: "array executor: response channel closed".into(), - }), - Err(_) => Err(ClusterError::Storage { - detail: "array executor: local dispatch timed out".into(), - }), - } + await_local_response(rx.recv()).await + } +} + +/// Await the local Data Plane's response. A timeout is the typed +/// `DeadlineExceeded` verdict, which the coordinator renders as `57014`. +async fn await_local_response( + rx: impl std::future::Future>, +) -> Result { + match tokio::time::timeout(LOCAL_DISPATCH_TIMEOUT, rx).await { + Ok(Some(resp)) => Ok(resp), + Ok(None) => Err(ClusterError::Storage { + detail: "array executor: response channel closed".into(), + }), + Err(_) => Err(ClusterError::DataPlane { + code: DataPlaneErrorCode::DeadlineExceeded, + }), } } @@ -152,6 +159,27 @@ mod tests { assert_eq!(request.vshard_id, vshard_id); } + /// A local timeout crosses as the typed deadline verdict, and the + /// coordinator renders it as `57014`. + #[tokio::test(start_paused = true)] + async fn a_local_timeout_is_a_typed_deadline() { + let error = await_local_response(std::future::pending::>()) + .await + .expect_err("a pending response must time out"); + assert!( + matches!( + &error, + ClusterError::DataPlane { + code: DataPlaneErrorCode::DeadlineExceeded + } + ), + "expected a typed deadline, got {error:?}" + ); + let rebuilt = crate::control::cluster::array_cluster_helpers::cluster_err(error); + let (_, state, _) = crate::control::server::pgwire::types::error_to_sqlstate(&rebuilt); + assert_eq!(state, nodedb_types::error::sqlstate::QUERY_CANCELED.0); + } + #[test] fn nonzero_vshard_is_preserved_for_read_and_write_requests() { let array_id = ArrayId::new(TenantId::new(41), "measurements"); diff --git a/nodedb/src/control/cluster/array_executor/mod.rs b/nodedb/src/control/cluster/array_executor/mod.rs index 5c89db0b1..e65d997cc 100644 --- a/nodedb/src/control/cluster/array_executor/mod.rs +++ b/nodedb/src/control/cluster/array_executor/mod.rs @@ -28,10 +28,12 @@ //! - [`read`]: read/scan handlers (slice, aggregate, surrogate-bitmap scan) plus //! the response-row parsers. //! - [`write`]: write handlers (put, delete). +//! - [`refusal`]: the typed cluster error a Data-Plane refusal answers with. pub mod cells; mod executor; mod read; +mod refusal; mod trait_impl; mod write; diff --git a/nodedb/src/control/cluster/array_executor/read.rs b/nodedb/src/control/cluster/array_executor/read.rs index 3eb9c6472..4ffdf3107 100644 --- a/nodedb/src/control/cluster/array_executor/read.rs +++ b/nodedb/src/control/cluster/array_executor/read.rs @@ -15,6 +15,7 @@ use crate::types::{TxnId, VShardId}; use nodedb_types::SurrogateBitmap; use super::executor::DataPlaneArrayExecutor; +use super::refusal::refusal_error; use crate::data::executor::response_codec::ArraySliceResponse; use nodedb_physical::physical_plan::{ArrayOp, ArrayReducer, PhysicalPlan}; @@ -62,14 +63,7 @@ impl DataPlaneArrayExecutor { .await?; if resp.status == crate::bridge::envelope::Status::Error { - let detail = resp - .error_code - .as_ref() - .map(|c| format!("{c:?}")) - .unwrap_or_else(|| "unknown Data Plane error".into()); - return Err(ClusterError::Storage { - detail: format!("array slice Data Plane error: {detail}"), - }); + return Err(refusal_error("array slice", &resp)); } // Decode the structured `ArraySliceResponse` envelope, then split the @@ -137,14 +131,7 @@ impl DataPlaneArrayExecutor { .await?; if resp.status == crate::bridge::envelope::Status::Error { - let detail = resp - .error_code - .as_ref() - .map(|c| format!("{c:?}")) - .unwrap_or_else(|| "unknown Data Plane error".into()); - return Err(ClusterError::Storage { - detail: format!("array agg Data Plane error: {detail}"), - }); + return Err(refusal_error("array agg", &resp)); } if resp.payload.is_empty() { @@ -189,14 +176,7 @@ impl DataPlaneArrayExecutor { .await?; if resp.status == crate::bridge::envelope::Status::Error { - let detail = resp - .error_code - .as_ref() - .map(|c| format!("{c:?}")) - .unwrap_or_else(|| "unknown Data Plane error".into()); - return Err(ClusterError::Storage { - detail: format!("surrogate bitmap scan Data Plane error: {detail}"), - }); + return Err(refusal_error("surrogate bitmap scan", &resp)); } collect_surrogate_bitmap(&resp.payload) diff --git a/nodedb/src/control/cluster/array_executor/refusal.rs b/nodedb/src/control/cluster/array_executor/refusal.rs new file mode 100644 index 000000000..d3d37ff5c --- /dev/null +++ b/nodedb/src/control/cluster/array_executor/refusal.rs @@ -0,0 +1,316 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The cluster error an array shard answers with when its Data Plane refuses. +//! +//! A coded refusal crosses as `ClusterError::DataPlane`, so the coordinator +//! rebuilds `crate::Error::DataPlane(code)` and renders the SQLSTATE a +//! single-node execution renders. Only a refusal with no code is a storage +//! error. A local-execution error crosses in its typed wire form and keeps +//! its class. + +use nodedb_cluster::error::ClusterError; +use nodedb_cluster::rpc_codec::DataPlaneErrorCode; + +use crate::bridge::envelope::Response; + +/// The cluster error for a Data-Plane response with an error status. +pub(super) fn refusal_error(context: &str, response: &Response) -> ClusterError { + match response.error_code.as_deref() { + Some(code) => ClusterError::DataPlane { + code: code.clone().into(), + }, + None => ClusterError::Storage { + detail: format!("{context}: data plane returned an error status with no error code"), + }, + } +} + +/// The cluster error for a local-execution error. +/// +/// - A Data-Plane verdict keeps its code. +/// - A deadline and a capacity refusal cross as their Data-Plane verdicts. +/// - A missing leader crosses as `WrongOwner`, so the coordinator re-reads +/// its routing and retries. +/// - Every other error crosses as `ShardExecution` with its typed wire form. +/// `context` goes before its message in the log detail. +pub(super) fn execution_error(context: &str, error: crate::Error) -> ClusterError { + match error { + crate::Error::DataPlane(code) => ClusterError::DataPlane { code: code.into() }, + crate::Error::DeadlineExceeded { .. } => ClusterError::DataPlane { + code: DataPlaneErrorCode::DeadlineExceeded, + }, + capacity @ crate::Error::DispatchCapacity { .. } => ClusterError::DataPlane { + code: DataPlaneErrorCode::DispatchCapacity { + reason: capacity.to_string(), + }, + }, + crate::Error::NotLeader { + vshard_id, + leader_node, + .. + } => ClusterError::WrongOwner { + vshard_id: vshard_id.as_u32(), + expected_owner_node: (leader_node != 0).then_some(leader_node), + }, + crate::Error::NoLeader { vshard_id } => ClusterError::WrongOwner { + vshard_id: vshard_id.as_u32(), + expected_owner_node: None, + }, + // Every other error crosses in its typed wire form, so the + // coordinator rebuilds it and renders its own SQLSTATE. + other @ (crate::Error::RejectedConstraint { .. } + | crate::Error::TxnOverlayMemoryExceeded { .. } + | crate::Error::RejectedAuthz { .. } + | crate::Error::OffsetRegression { .. } + | crate::Error::ConflictRetry { .. } + | crate::Error::CalvinSerializationConflict + | crate::Error::CalvinParticipantError + | crate::Error::RejectedPrevalidation { .. } + | crate::Error::RetryableRefusal { .. } + | crate::Error::AppendOnlyViolation { .. } + | crate::Error::BalanceViolation { .. } + | crate::Error::MaterializedSumTargetNotFound { .. } + | crate::Error::MaterializedSumResolutionMissing { .. } + | crate::Error::PeriodLocked { .. } + | crate::Error::PeriodLockMisconfigured { .. } + | crate::Error::RetentionViolation { .. } + | crate::Error::LegalHoldActive { .. } + | crate::Error::StateTransitionViolation { .. } + | crate::Error::TransitionCheckViolation { .. } + | crate::Error::TypeGuardViolation { .. } + | crate::Error::TypeMismatch { .. } + | crate::Error::InsufficientBalance { .. } + | crate::Error::RateExceeded { .. } + | crate::Error::CollectionNotFound { .. } + | crate::Error::DocumentNotFound { .. } + | crate::Error::CollectionDeactivated { .. } + | crate::Error::VShardAdmissionCapacityExceeded { .. } + | crate::Error::CrdtAdmissionRetriesExhausted { .. } + | crate::Error::CrdtAdmissionInvalidPlan { .. } + | crate::Error::CrdtAdmissionCallerFence + | crate::Error::CrdtApplyRequiresAdmission + | crate::Error::CrdtApplyForbiddenInTransaction + | crate::Error::NotInTransactionBlock { .. } + | crate::Error::CrdtAdmissionTimeout { .. } + | crate::Error::FanOutExceeded { .. } + | crate::Error::CrossCollectionNotColocated { .. } + | crate::Error::SourceFrozen { .. } + | crate::Error::CloneWriteRequiresMaterialize { .. } + | crate::Error::BadRequest { .. } + | crate::Error::BackupTenantMismatch { .. } + | crate::Error::BackupKeyMismatch + | crate::Error::QuotaOvercommit { .. } + | crate::Error::PlanError { .. } + | crate::Error::FeatureNotSupported { .. } + | crate::Error::UndefinedFunction { .. } + | crate::Error::UndefinedObject { .. } + | crate::Error::ObjectNotInPrerequisiteState { .. } + | crate::Error::UndefinedColumn { .. } + | crate::Error::AmbiguousColumn { .. } + | crate::Error::UnknownStrictField { .. } + | crate::Error::DivisionByZero + | crate::Error::DataException { .. } + | crate::Error::InvalidLimitValue { .. } + | crate::Error::RetryableSchemaChanged { .. } + | crate::Error::RetryableLeaderChange { .. } + | crate::Error::GroupQuorumUnavailable { .. } + | crate::Error::GroupMarksUnavailable { .. } + | crate::Error::MetadataLeaderUnavailable + | crate::Error::AuthorizationStateBehind { .. } + | crate::Error::ExecutionLimitExceeded { .. } + | crate::Error::LimitExceeded { .. } + | crate::Error::Wal(_) + | crate::Error::Dispatch { .. } + | crate::Error::Storage { .. } + | crate::Error::ColdStorage { .. } + | crate::Error::Serialization { .. } + | crate::Error::Codec { .. } + | crate::Error::SegmentCorrupted { .. } + | crate::Error::MemoryExhausted { .. } + | crate::Error::Backpressure { .. } + | crate::Error::Crdt(_) + | crate::Error::Io(_) + | crate::Error::Config { .. } + | crate::Error::Encryption { .. } + | crate::Error::Bridge { .. } + | crate::Error::VersionCompat { .. } + | crate::Error::Internal { .. } + | crate::Error::Shaping(_) + | crate::Error::Ddl(_) + | crate::Error::RemoteTyped { .. } + | crate::Error::DescriptorVersionAnomaly { .. } + | crate::Error::CollectionPurgeRowMissing { .. } + | crate::Error::CatalogIntegrityViolation { .. } + | crate::Error::Promql(_) + | crate::Error::DependentObjectsExist { .. } + | crate::Error::RoleInUse { .. } + | crate::Error::CascadeCycle { .. } + | crate::Error::CrossShardInExplicitTransaction + | crate::Error::SequencerUnavailable + | crate::Error::SessionCapExceeded { .. } + | crate::Error::SessionIdleTimeout + | crate::Error::SessionTokenExpired + | crate::Error::SessionKilledByAdmin + | crate::Error::SessionUserDropped + | crate::Error::OidcProviderTenantUnbound + | crate::Error::OidcProviderTenantUnavailable { .. } + | crate::Error::ExternalRoleUndefined { .. } + | crate::Error::OidcNoDefaultDatabase { .. } + | crate::Error::TenantVectorDimExceeded { .. } + | crate::Error::TenantGraphDepthExceeded { .. } + | crate::Error::RoleInheritanceCycle { .. } + | crate::Error::RoleInheritanceDepthExceeded { .. } + | crate::Error::OllpExhausted { .. } + | crate::Error::MirrorReadOnly { .. } + | crate::Error::StaleReadNotLeader { .. }) => { + let detail = format!("{context}: {other}"); + ClusterError::ShardExecution { + error: Box::new( + crate::control::cluster::data_plane_error_wire::execution_error_to_typed(other), + ), + detail, + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::{ErrorCode, Payload, Status}; + use crate::types::{Lsn, RequestId}; + + fn refusal(code: Option) -> Response { + Response { + request_id: RequestId::new(1), + status: Status::Error, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: code.map(Box::new), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + fn unsupported() -> ErrorCode { + ErrorCode::Unsupported { + detail: "not on this engine".into(), + } + } + + /// A coded refusal keeps its code through the cluster error and back to + /// the coordinator's typed error. + #[test] + fn a_coded_refusal_keeps_its_code() { + match refusal_error("array slice", &refusal(Some(unsupported()))) { + ClusterError::DataPlane { code } => { + assert_eq!(ErrorCode::from(code), unsupported()); + } + other => panic!("expected the typed refusal, got {other:?}"), + } + } + + #[test] + fn a_refusal_with_no_code_is_a_storage_error() { + match refusal_error("array slice", &refusal(None)) { + ClusterError::Storage { detail } => assert!(detail.starts_with("array slice: ")), + other => panic!("expected a storage error, got {other:?}"), + } + } + + #[test] + fn a_local_deadline_crosses_as_the_deadline_verdict() { + let error = crate::Error::DeadlineExceeded { + request_id: RequestId::new(1), + }; + assert!(matches!( + execution_error("array put", error), + ClusterError::DataPlane { + code: DataPlaneErrorCode::DeadlineExceeded + } + )); + } + + #[test] + fn a_missing_leader_crosses_as_wrong_owner() { + let error = crate::Error::NotLeader { + vshard_id: crate::types::VShardId::new(9), + leader_node: 4, + leader_addr: "10.0.0.4:9000".into(), + }; + assert!(matches!( + execution_error("array put raft propose", error), + ClusterError::WrongOwner { + vshard_id: 9, + expected_owner_node: Some(4) + } + )); + } + + #[test] + fn an_execution_verdict_keeps_its_code() { + let error = crate::Error::DataPlane(unsupported()); + match execution_error("array put", error) { + ClusterError::DataPlane { code } => { + assert_eq!(ErrorCode::from(code), unsupported()); + } + other => panic!("expected the typed refusal, got {other:?}"), + } + } + + /// A classified error with no Data-Plane twin crosses in its typed wire + /// form, and the coordinator renders the SQLSTATE a single-node + /// execution renders. + #[test] + fn a_classified_error_keeps_its_sqlstate_at_the_coordinator() { + use crate::control::cluster::array_cluster_helpers::cluster_err; + use crate::control::server::pgwire::types::error_to_sqlstate; + use nodedb_cluster::rpc_codec::ShardErrorWire; + + let local = || crate::Error::RejectedAuthz { + tenant_id: crate::types::TenantId::new(1), + resource: "grid".into(), + }; + let shard = execution_error("array put", local()); + match &shard { + ClusterError::ShardExecution { detail, .. } => { + assert!(detail.starts_with("array put: "), "{detail}"); + } + other => panic!("expected a typed shard execution error, got {other:?}"), + } + let received = ClusterError::from(ShardErrorWire::from(shard)); + let rebuilt = cluster_err(received); + assert_eq!(error_to_sqlstate(&rebuilt).1, error_to_sqlstate(&local()).1); + } + + /// Every variant keeps its SQLSTATE class through the array shard hop, + /// except those whose class no public numeric code carries. + #[test] + fn every_variant_keeps_its_class_through_the_array_hop() { + use crate::control::cluster::array_cluster_helpers::cluster_err; + use crate::control::gateway::error_map::class_parity::{ + error_samples, error_variant_index, + }; + use crate::control::server::pgwire::types::error_to_sqlstate; + use nodedb_cluster::rpc_codec::ShardErrorWire; + + let gaps = [7, 32, 33, 88, 90]; + for (err, twin) in error_samples().into_iter().zip(error_samples()) { + if gaps.contains(&error_variant_index(&err)) { + continue; + } + let (_, local, _) = error_to_sqlstate(&err); + let wire = ShardErrorWire::from(execution_error("array put", twin)); + let rebuilt = cluster_err(ClusterError::from(wire)); + let (_, remote, _) = error_to_sqlstate(&rebuilt); + assert_eq!( + remote.get(..2), + local.get(..2), + "{err:?} is {local} locally but {remote} at the coordinator" + ); + } + } +} diff --git a/nodedb/src/control/cluster/array_executor/write.rs b/nodedb/src/control/cluster/array_executor/write.rs index 71c9269d9..6d2982a89 100644 --- a/nodedb/src/control/cluster/array_executor/write.rs +++ b/nodedb/src/control/cluster/array_executor/write.rs @@ -14,6 +14,7 @@ use nodedb_cluster::error::{ClusterError, Result}; use super::cells::flatten_blob_vec; use super::executor::DataPlaneArrayExecutor; +use super::refusal::{execution_error, refusal_error}; use crate::control::server::dispatch_utils::{ ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, submit_write, }; @@ -130,9 +131,7 @@ impl DataPlaneArrayExecutor { entry, ) .await - .map_err(|e| ClusterError::Storage { - detail: format!("{op_label} raft propose: {e}"), - })?; + .map_err(|e| execution_error(&format!("{op_label} raft propose"), e))?; let affected = require_affected_count(&apply_payload).map_err(|e| ClusterError::Storage { detail: format!("{op_label}: {e}"), @@ -154,20 +153,10 @@ impl DataPlaneArrayExecutor { single_node_submit(array_id, VShardId::new(local_vshard_id), plan), ) .await - .map_err(|e| ClusterError::Storage { - detail: format!("{op_label}: {e}"), - })?; + .map_err(|e| execution_error(op_label, e))?; if outcome.response.status == crate::bridge::envelope::Status::Error { - let detail = outcome - .response - .error_code - .as_ref() - .map(|c| format!("{c:?}")) - .unwrap_or_else(|| "unknown Data Plane error".into()); - return Err(ClusterError::Storage { - detail: format!("{op_label} Data Plane error: {detail}"), - }); + return Err(refusal_error(op_label, &outcome.response)); } // Ack with the LSN the funnel actually minted. `None` would mean the @@ -210,7 +199,11 @@ fn single_node_submit( event_source: crate::event::EventSource::User, txn_id: None, user_id: None, - durability: WalDurability::AppendHere { now_override: None }, + durability: WalDurability::AppendHere { + now_override: None, + apply_key: 0, + commit_hlc: None, + }, ordering: WriteOrdering::Gate, change_feed: ChangeFeedOwner::Funnel, } diff --git a/nodedb/src/control/cluster/calvin/mod.rs b/nodedb/src/control/cluster/calvin/mod.rs index 51680f5bf..13c67536a 100644 --- a/nodedb/src/control/cluster/calvin/mod.rs +++ b/nodedb/src/control/cluster/calvin/mod.rs @@ -5,6 +5,6 @@ pub mod scheduler; pub use executor::{OllpConfig, OllpError, OllpOrchestrator}; pub use scheduler::{ - CalvinReadResultProposal, ReadResultEvent, Scheduler, SchedulerConfig, SchedulerParams, - propose_calvin_read_result, + CalvinReadResultProposal, RaftSequencerProposer, ReadResultEvent, Scheduler, SchedulerConfig, + SchedulerParams, SequencerProposer, propose_calvin_read_result, }; diff --git a/nodedb/src/control/cluster/calvin/scheduler/applied_gate.rs b/nodedb/src/control/cluster/calvin/scheduler/applied_gate.rs index f1820cc29..5b9a12fe1 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/applied_gate.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/applied_gate.rs @@ -99,6 +99,16 @@ impl AppliedGate { self.fully_applied_epoch == NOT_YET_APPLIED_EPOCH } + /// The highest epoch delivered to this vShard, or `None` when none was. + pub fn highest_seen_epoch(&self) -> Option { + (self.highest_seen_epoch != NOT_YET_APPLIED_EPOCH).then_some(self.highest_seen_epoch) + } + + /// Whether every position of every epoch at or below `epoch` is applied. + pub fn is_fully_applied_through(&self, epoch: u64) -> bool { + !self.is_sentinel() && self.fully_applied_epoch >= epoch + } + /// Record the delivery of `(epoch, position)` on this vShard, carrying the /// sequencer's per-`(epoch, vShard)` position `count`. /// diff --git a/nodedb/src/control/cluster/calvin/scheduler/applied_mirror.rs b/nodedb/src/control/cluster/calvin/scheduler/applied_mirror.rs new file mode 100644 index 000000000..dc84ac9b5 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/applied_mirror.rs @@ -0,0 +1,175 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A shared view of which Calvin positions this node's scheduler for one +//! vShard applied. +//! +//! The scheduler owns its [`super::AppliedGate`] and mutates it on its own +//! task. Other Control-Plane code needs one question answered: did this +//! node's replica of the vShard apply `(epoch, position)`? The scheduler +//! mirrors each applied position and each watermark advance here, so the +//! answer needs no message to the scheduler task. +//! +//! The mirror keeps the same shape as the gate: a fully-applied watermark and +//! the applied positions above it. The tail is pruned as the watermark +//! advances, so it stays as small as the gate's. + +use std::collections::{BTreeSet, HashMap}; +use std::sync::{Arc, Mutex}; + +use super::recovery::NOT_YET_APPLIED_EPOCH; +use crate::control::security::catalog::calvin_applied::StoredCalvinApplied; + +#[derive(Debug)] +struct MirrorState { + /// Every position of every epoch at or below this is applied. + /// [`NOT_YET_APPLIED_EPOCH`] means none is. + fully_applied_epoch: u64, + /// Applied positions of epochs above the watermark. + tail: BTreeSet<(u64, u32)>, +} + +/// Applied positions of one vShard's scheduler on this node. +#[derive(Debug)] +pub struct AppliedMirror { + state: Mutex, +} + +impl AppliedMirror { + /// A mirror seeded from the scheduler's recovery scan. + pub fn new(fully_applied_epoch: u64, tail: BTreeSet<(u64, u32)>) -> Self { + Self { + state: Mutex::new(MirrorState { + fully_applied_epoch, + tail, + }), + } + } + + /// Record that `(epoch, position)` applied. + pub fn mark(&self, epoch: u64, position: u32) { + let mut state = self.state.lock().unwrap_or_else(|p| p.into_inner()); + if state.fully_applied_epoch != NOT_YET_APPLIED_EPOCH && epoch <= state.fully_applied_epoch + { + return; + } + state.tail.insert((epoch, position)); + } + + /// Record that every position of every epoch at or below `watermark` + /// applied. + pub fn fold(&self, watermark: u64) { + let mut state = self.state.lock().unwrap_or_else(|p| p.into_inner()); + if state.fully_applied_epoch != NOT_YET_APPLIED_EPOCH + && watermark <= state.fully_applied_epoch + { + return; + } + state.fully_applied_epoch = watermark; + state.tail = state.tail.split_off(&(watermark.saturating_add(1), 0)); + } + + /// The mirror's state: the fully-applied watermark and the applied + /// positions above it. + pub fn snapshot(&self) -> (u64, BTreeSet<(u64, u32)>) { + let state = self.state.lock().unwrap_or_else(|p| p.into_inner()); + (state.fully_applied_epoch, state.tail.clone()) + } + + /// Whether this node's replica applied `(epoch, position)`. + pub fn is_applied(&self, epoch: u64, position: u32) -> bool { + let state = self.state.lock().unwrap_or_else(|p| p.into_inner()); + (state.fully_applied_epoch != NOT_YET_APPLIED_EPOCH && epoch <= state.fully_applied_epoch) + || state.tail.contains(&(epoch, position)) + } +} + +/// The applied mirror of every vShard scheduler on this node. +#[derive(Debug, Default)] +pub struct AppliedMirrors { + by_vshard: Mutex>>, +} + +impl AppliedMirrors { + /// Register the mirror of a scheduler starting for `vshard_id`. A + /// restarted scheduler replaces its predecessor's mirror. + pub fn register( + &self, + vshard_id: u32, + fully_applied_epoch: u64, + tail: &BTreeSet<(u64, u32)>, + ) -> Arc { + let mirror = Arc::new(AppliedMirror::new(fully_applied_epoch, tail.clone())); + self.by_vshard + .lock() + .unwrap_or_else(|p| p.into_inner()) + .insert(vshard_id, Arc::clone(&mirror)); + mirror + } + + /// Every registered mirror's state, in the shape the catalog stores. + pub fn snapshot_all(&self) -> Vec { + let mirrors: Vec<(u32, Arc)> = self + .by_vshard + .lock() + .unwrap_or_else(|p| p.into_inner()) + .iter() + .map(|(vshard_id, mirror)| (*vshard_id, Arc::clone(mirror))) + .collect(); + mirrors + .into_iter() + .map(|(vshard_id, mirror)| { + let (fully_applied_epoch, tail) = mirror.snapshot(); + StoredCalvinApplied { + vshard_id, + fully_applied_epoch, + tail, + } + }) + .collect() + } + + /// The mirror of `vshard_id`, when this node runs its scheduler. + pub fn get(&self, vshard_id: u32) -> Option> { + self.by_vshard + .lock() + .unwrap_or_else(|p| p.into_inner()) + .get(&vshard_id) + .cloned() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_mirror_answers_like_the_gate() { + let mirror = AppliedMirror::new(NOT_YET_APPLIED_EPOCH, BTreeSet::new()); + assert!(!mirror.is_applied(0, 0)); + mirror.mark(3, 1); + assert!(mirror.is_applied(3, 1)); + assert!(!mirror.is_applied(3, 0)); + mirror.fold(3); + assert!(mirror.is_applied(3, 0)); + assert!(mirror.is_applied(2, 9)); + assert!(!mirror.is_applied(4, 0)); + // A mark at or below the watermark changes nothing. + mirror.mark(1, 0); + assert!(mirror.is_applied(1, 0)); + // A lower watermark never moves it back. + mirror.fold(1); + assert!(mirror.is_applied(3, 0)); + } + + #[test] + fn a_restarted_scheduler_replaces_its_mirror() { + let mirrors = AppliedMirrors::default(); + let first = mirrors.register(7, NOT_YET_APPLIED_EPOCH, &BTreeSet::new()); + first.mark(1, 0); + let second = mirrors.register(7, 4, &BTreeSet::new()); + let current = mirrors.get(7).expect("mirror"); + assert!(Arc::ptr_eq(¤t, &second)); + assert!(current.is_applied(4, 0)); + assert!(mirrors.get(8).is_none()); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/cut_floor.rs b/nodedb/src/control/cluster/calvin/scheduler/cut_floor.rs new file mode 100644 index 000000000..94bbca1d9 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/cut_floor.rs @@ -0,0 +1,173 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Backup cut markers as one vShard's scheduler receives them. +//! +//! A cut marker arrives between two epoch batches, so every transaction of +//! an epoch at or below the highest epoch delivered before the marker came +//! before it, and every transaction of a later epoch came after it. A +//! transaction after a marker commits above the marker's watermark: a +//! restore of the backup that placed the marker refuses it. A transaction +//! before the marker finishes before the scheduler reports the marker, so the +//! backup holds it. +//! +//! The commit HLC of a transaction is the instant its epoch was created, +//! read once on the sequencer leader and replicated with the batch, raised to +//! the floor of every marker it came after. Replicas of a vShard receive the +//! same inputs in the same order, so they stamp every transaction alike. + +use super::recovery::NOT_YET_APPLIED_EPOCH; + +/// Nanoseconds per millisecond, to express an epoch's millisecond wall time +/// on the nanosecond HLC scale. +const NANOS_PER_MILLI: u64 = 1_000_000; + +/// A marker whose floor applies to the epochs above `through`. +#[derive(Debug, Clone, Copy)] +struct Marker { + through: u64, + floor: u64, +} + +/// A marker the scheduler has not reported yet. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct WaitingCut { + /// The marker's watermark. + pub hlc: u64, + /// The highest epoch delivered before the marker. The marker passes once + /// every epoch at or below it is fully applied. + pub through: u64, +} + +/// The cut markers one scheduler received. +#[derive(Debug, Default)] +pub struct CutFloors { + /// The floor every epoch above the folded markers records at least. + base: u64, + /// Markers not folded into `base`, in arrival order. + markers: Vec, + /// Markers not reported yet. + waiting: Vec, +} + +impl CutFloors { + /// Receive a marker carrying `hlc`, after epochs up to `highest_seen` + /// were delivered (`None` when none was). Returns `true` when every + /// transaction before it finished already: nothing came before it. + pub fn receive(&mut self, hlc: u64, highest_seen: Option) -> bool { + let floor = hlc.saturating_add(1); + match highest_seen { + None => { + self.base = self.base.max(floor); + true + } + Some(through) => { + self.markers.push(Marker { through, floor }); + self.waiting.push(WaitingCut { hlc, through }); + false + } + } + } + + /// The commit HLC of a transaction of `epoch` whose epoch was created at + /// `epoch_system_ms`. + pub fn commit_hlc(&self, epoch: u64, epoch_system_ms: i64) -> u64 { + let created = u64::try_from(epoch_system_ms) + .unwrap_or(0) + .saturating_mul(NANOS_PER_MILLI); + self.markers + .iter() + .filter(|marker| marker.through < epoch) + .map(|marker| marker.floor) + .fold(created.max(self.base), u64::max) + } + + /// Take every waiting marker whose epochs are fully applied, given the + /// test `fully_applied_through`. Returns their watermarks. + pub fn take_passed(&mut self, fully_applied_through: impl Fn(u64) -> bool) -> Vec { + let mut passed = Vec::new(); + self.waiting.retain(|cut| { + if fully_applied_through(cut.through) { + passed.push(cut.hlc); + false + } else { + true + } + }); + passed + } + + /// Fold every marker whose floor covers all epochs above `fully_applied` + /// into the base floor. No transaction of an epoch at or below + /// `fully_applied` commits again, so only the epochs above it need the + /// markers told apart. + pub fn fold(&mut self, fully_applied: u64) { + if fully_applied == NOT_YET_APPLIED_EPOCH { + return; + } + let base = &mut self.base; + self.markers.retain(|marker| { + if marker.through <= fully_applied { + *base = (*base).max(marker.floor); + false + } else { + true + } + }); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_transaction_after_a_marker_commits_above_its_watermark() { + let mut floors = CutFloors::default(); + assert!(!floors.receive(5_000_000_000, Some(7))); + assert_eq!( + floors.commit_hlc(7, 1), + NANOS_PER_MILLI, + "epoch 7 came before" + ); + assert_eq!(floors.commit_hlc(8, 1), 5_000_000_001, "epoch 8 came after"); + assert_eq!( + floors.commit_hlc(8, 9_000), + 9_000 * NANOS_PER_MILLI, + "a later creation instant stands" + ); + } + + #[test] + fn a_marker_passes_once_its_epochs_are_fully_applied() { + let mut floors = CutFloors::default(); + floors.receive(100, Some(3)); + floors.receive(200, Some(5)); + assert_eq!(floors.take_passed(|through| through <= 4), vec![100]); + assert_eq!( + floors.take_passed(|through| through <= 4), + Vec::::new() + ); + assert_eq!(floors.take_passed(|through| through <= 5), vec![200]); + } + + #[test] + fn a_marker_before_any_epoch_passes_at_once_and_raises_every_epoch() { + let mut floors = CutFloors::default(); + assert!(floors.receive(100, None)); + assert_eq!(floors.commit_hlc(0, 0), 101); + assert!(floors.take_passed(|_| false).is_empty()); + } + + #[test] + fn folding_keeps_every_later_stamp() { + let mut floors = CutFloors::default(); + floors.receive(100, Some(3)); + floors.receive(50, Some(6)); + let before: Vec = (4..9).map(|epoch| floors.commit_hlc(epoch, 0)).collect(); + floors.fold(4); + let after: Vec = (5..9).map(|epoch| floors.commit_hlc(epoch, 0)).collect(); + assert_eq!(&before[1..], &after[..]); + floors.fold(NOT_YET_APPLIED_EPOCH); + assert_eq!(floors.commit_hlc(7, 0), 101); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/config.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/config.rs index 3b28a1874..a71084d62 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/config.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/config.rs @@ -4,6 +4,8 @@ use std::time::Duration; +use crate::bridge::dispatch::DATA_PLANE_QUEUE_CAPACITY; + /// Tuning parameters for a [`super::core::Scheduler`] instance. #[derive(Debug, Clone)] pub struct SchedulerConfig { @@ -29,17 +31,37 @@ pub struct SchedulerConfig { /// /// Default: `epoch_duration_ms * 250` milliseconds. pub verdict_stall_warn_ms: u64, + /// In-flight backlog at which the scheduler stops taking new sequenced + /// input. The backlog counts pending, blocked, and dependent-barrier txns. + /// + /// The bound applies only while some backlog txn progresses without new + /// input. A backlog of blocked txns alone keeps intake open, because a + /// reservation release they wait on arrives as input. + /// + /// Default: [`DATA_PLANE_QUEUE_CAPACITY`]. Every dispatch of one vShard + /// goes to one Data Plane core, whose queue holds that many requests. A + /// larger backlog cannot hold a dispatch slot per txn at once. + pub max_inflight_backlog: usize, + /// Most sequencer log entries one catch-up drain reads and replays. + /// + /// Default: `channel_capacity`. A drop happens only when the fan-out + /// channel is full, so one window covers about one channel of missed + /// entries. Windows run back to back while intake is open. + pub catch_up_window: u64, } impl Default for SchedulerConfig { fn default() -> Self { let epoch_duration_ms = 20u64; + let channel_capacity = 512usize; Self { - channel_capacity: 512, + channel_capacity, txn_deadline_multiplier: 3, epoch_duration_ms, dependent_read_passive_timeout_ms: epoch_duration_ms * 3, verdict_stall_warn_ms: epoch_duration_ms * 250, + max_inflight_backlog: DATA_PLANE_QUEUE_CAPACITY, + catch_up_window: channel_capacity as u64, } } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs index bc5d3e4dc..41bf6fcfd 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs @@ -16,17 +16,39 @@ //! and thereby reconstructs the missed input deterministically. Replay is //! idempotent — `process_new_txn`'s in-flight guard turns an already-in-flight //! Txn into a no-op, and Reserve/Release re-application is a lock-manager no-op. +//! +//! One drain reads at most [`SchedulerConfig::catch_up_window`] log entries. +//! +//! [`SchedulerConfig::catch_up_window`]: super::super::config::SchedulerConfig::catch_up_window use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use nodedb_cluster::calvin::types::SchedulerInput; use super::scheduler::Scheduler; +/// Result of one [`Scheduler::drain_catch_up`] call. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) enum CatchUpDrain { + /// Nothing is left to replay now. No catch-up is armed, nothing is + /// committed at the armed index yet, the log read failed and the entry + /// stays armed for a later tick, or the armed range is fully replayed. + Settled, + /// The drain stopped before the committed index: the window filled or + /// the intake gate closed. Catch-up stays armed at the first unprocessed + /// index, and the next drain can resume at once. + Remaining, +} + impl Scheduler { /// Replay any sequencer-fan-out inputs dropped on this replica. /// /// Run on the periodic stall tick. O(1) in the common case (no pending /// catch-up → one map probe and return). /// + /// Reads and replays at most `catch_up_window` log entries from the armed + /// index. Stops feeding at the first input after which the intake gate is + /// closed. + /// /// # Lock discipline (deadlock-safety) /// /// The two shared mutexes — the sequencer state machine and MultiRaft — are @@ -34,7 +56,9 @@ impl Scheduler { /// holds the SM lock while fanning out but never takes MultiRaft underneath /// it; this drain takes them strictly one-at-a-time (SM → release → MultiRaft /// → release → SM → release), so the two paths can never form a lock cycle. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn drain_catch_up(&mut self) { + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn drain_catch_up( + &mut self, + ) -> CatchUpDrain { // 1. SM-lock scope: PEEK the earliest armed index for this vShard. // `None` (the common case) means no catch-up is pending — return O(1). // Otherwise pair it with the committed-index watermark as the replay @@ -49,27 +73,30 @@ impl Scheduler { .lock() .unwrap_or_else(|p| p.into_inner()); let Some(lo) = sm.peek_catch_up_from(self.vshard_id) else { - return; + return CatchUpDrain::Settled; }; let Some(hi) = sm.current_committed_index() else { // Armed but nothing applied yet — leave it armed and retry once // an entry is applied and `hi` is known. - return; + return CatchUpDrain::Settled; }; if lo > hi { // Armed ahead of the committed watermark (e.g. spawn-armed from // the first available index before any entry applied on this // replica). Nothing to replay yet; stay armed. - return; + return CatchUpDrain::Settled; } (lo, hi) }; + // Last index this drain reads: the window end, capped at `hi`. + let window = self.config.catch_up_window.max(1); + let end = lo.saturating_add(window - 1).min(hi); - // 2. MultiRaft-lock scope: read the committed sequencer log range. No SM - // lock is held here (see the lock-discipline note above). + // 2. MultiRaft-lock scope: read the committed sequencer log window. No + // SM lock is held here (see the lock-discipline note above). let entries = { let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); - match mr.read_committed_entries(SEQUENCER_GROUP_ID, lo, hi) { + match mr.read_committed_entries(SEQUENCER_GROUP_ID, lo, end) { Ok(entries) => entries, Err(nodedb_cluster::error::ClusterError::Raft( nodedb_raft::RaftError::LogCompacted { .. }, @@ -93,7 +120,7 @@ impl Scheduler { .lock() .unwrap_or_else(|p| p.into_inner()) .clear_catch_up_up_to(self.vshard_id, hi); - return; + return CatchUpDrain::Settled; } Err(e) => { // Transient infra fault (e.g. group transiently absent). @@ -102,54 +129,101 @@ impl Scheduler { tracing::warn!( vshard = self.vshard_id, lo, - hi, + end, error = %e, "calvin catch-up: failed to read committed sequencer entries" ); - return; + return CatchUpDrain::Settled; } } }; // 3. SM-lock scope: decode the raw log entries into this vShard's // `SchedulerInput` stream (a pure `&self` read — no side effects). - let inputs = { + // Each entry decodes on its own, so every input keeps the Raft index + // it came from. Decoding holds no cross-entry state, so the stream is + // identical to a whole-range decode. + let inputs: Vec<(u64, SchedulerInput)> = { let sm = self .sequencer_state_machine .lock() .unwrap_or_else(|p| p.into_inner()); - sm.replay_epochs_for_vshard(&entries, self.vshard_id, 0, u64::MAX) + entries + .iter() + .flat_map(|entry| { + sm.replay_epochs_for_vshard( + std::slice::from_ref(entry), + self.vshard_id, + 0, + u64::MAX, + ) + .into_iter() + .map(move |input| (entry.index, input)) + }) + .collect() }; // 4. Feed each replayed input through the SAME live processing path — no // lock held. Determinism: identical inputs through identical code. // The in-flight guard makes an overlapping already-in-flight Txn a // no-op; Reserve/Release re-application is idempotent. - let replayed = inputs.len() as u64; - for input in inputs { + // + // An input after which the intake gate is closed stops the feed: a + // dispatch deferred at capacity, or a full in-flight backlog. The + // next drain resumes at the first input not yet processed. + let mut replayed: u64 = 0; + let mut resume_from: Option = None; + let mut feed = inputs.into_iter().peekable(); + while let Some((_, input)) = feed.next() { self.process_scheduler_input(input); + replayed += 1; + if self.intake_closure().is_some() { + resume_from = feed.peek().map(|(index, _)| *index); + break; + } } + // A window that ends before `hi` resumes at the first index past it. + let resume_from = resume_from.or(end.checked_add(1).filter(|&next| next <= hi)); - // Replay of `lo ..= hi` is complete: clear the armed catch-up, but only - // up to `hi` — a concurrent drop recorded at an index `> hi` while this - // replay ran is preserved for the next drain. This is the CONFIRM step - // the peek-not-take at the top defers to; a transient failure above - // returned early and left the entry armed. - self.sequencer_state_machine - .lock() - .unwrap_or_else(|p| p.into_inner()) - .clear_catch_up_up_to(self.vshard_id, hi); + { + let sm = self + .sequencer_state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()); + match resume_from { + // Stopped early: re-arm exactly at the first unprocessed input's + // index. Clearing below it first lets the min-collapse arm move + // the entry forward. Both run under one SM lock. + Some(next) => { + sm.clear_catch_up_up_to(self.vshard_id, next.saturating_sub(1)); + sm.arm_catch_up_from(self.vshard_id, next); + } + // Replay of `lo ..= hi` is complete (`end == hi` here): clear + // the armed catch-up, but only up to `hi` — a concurrent drop + // recorded at an index `> hi` while this replay ran is + // preserved for the next drain. This is the CONFIRM step the + // peek-not-take at the top defers to; a transient failure + // above returned early and left the entry armed. + None => sm.clear_catch_up_up_to(self.vshard_id, hi), + } + } if replayed > 0 { self.metrics.record_catch_up_replayed(replayed); tracing::info!( vshard = self.vshard_id, lo, + end, hi, replayed, "calvin catch-up: replayed dropped sequencer inputs from committed log" ); } + if resume_from.is_some() { + CatchUpDrain::Remaining + } else { + CatchUpDrain::Settled + } } } @@ -157,111 +231,19 @@ impl Scheduler { #[cfg(test)] mod tests { use super::*; - use std::collections::{BTreeSet, HashMap}; use std::sync::atomic::Ordering; - use std::sync::{Arc, Mutex}; use std::time::{Duration, Instant}; - use nodedb_cluster::MultiRaft; - use nodedb_cluster::RoutingTable; - use nodedb_cluster::calvin::types::{ - EngineKeySet, EpochBatch, ReadWriteSet, SchedulerInput, SequencedTxn, SortedVec, TxClass, - VersionedReadSet, - }; - use nodedb_cluster::calvin::{CalvinCompletionRegistry, SequencerEntry, SequencerStateMachine}; + use nodedb_cluster::calvin::types::{EpochBatch, SchedulerInput, SequencedTxn}; + use nodedb_cluster::calvin::{CalvinCompletionRegistry, SequencerEntry}; use nodedb_types::TenantId; - use nodedb_types::id::{DatabaseId, VShardId}; + use nodedb_types::id::DatabaseId; - use super::super::scheduler::SchedulerParams; - use crate::bridge::dispatch::Dispatcher; - use crate::control::cluster::calvin::scheduler::lock_manager::{ - AcquireOutcome, LockManager, TxnId, + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + build_test_scheduler, build_test_scheduler_with_data_side, fill_tenant_inflight, + make_sequenced_txn, make_validate_only_txn, test_coll_vshard, }; - use crate::control::cluster::calvin::scheduler::metrics::SchedulerMetrics; - use crate::control::cluster::calvin::scheduler::{NOT_YET_APPLIED_EPOCH, SchedulerConfig}; - use crate::control::state::SharedState; - use crate::wal::WalManager; - - /// Build a minimally-wired `Scheduler` for driver-level unit tests. The Data - /// Plane is NOT started — tests exercise Control-Plane routing, guards, and - /// request dispatch only, so no core loop is needed. The returned `TempDir` - /// must be kept alive for the scheduler's lifetime (backs the WAL and - /// Raft storage). - fn build_test_scheduler(vshard_id: u32) -> (Scheduler, tempfile::TempDir) { - let registry = CalvinCompletionRegistry::new_detached(); - let dir = tempfile::tempdir().unwrap(); - let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("test.wal")).unwrap()); - let (dispatcher, mut data_sides) = Dispatcher::new(1, 64); - let _data_side = data_sides - .pop() - .expect("one configured core has one data side"); - let shared = SharedState::new(dispatcher, wal).unwrap(); - - let rt = RoutingTable::uniform(1, &[1], 1); - let multi_raft = Arc::new(Mutex::new(MultiRaft::new(1, rt, dir.path().to_path_buf()))); - - let sequencer_state_machine = Arc::new(Mutex::new(SequencerStateMachine::new( - HashMap::new(), - Arc::clone(®istry), - ))); - - let (_tx, receiver) = tokio::sync::mpsc::channel(16); - let (_rr_tx, read_result_rx) = tokio::sync::mpsc::channel(16); - let (_prom_tx, promotion_rx) = tokio::sync::mpsc::unbounded_channel(); - let (verdict_tx, verdict_rx) = tokio::sync::mpsc::channel(16); - registry.register_verdict_signal_sender(vshard_id, verdict_tx); - - let lock_manager = Arc::new(Mutex::new(LockManager::new())); - - let scheduler = Scheduler::new(SchedulerParams { - vshard_id, - receiver, - shared, - multi_raft, - sequencer_state_machine, - // A freshly-built scheduler has applied nothing, so its watermark is the - // not-yet-applied sentinel (matching `read_applied_recovery` for a clean - // node). Hardcoding `0` here would instead claim epoch 0 is fully applied, - // making the exactly-once gate (`AppliedGate::is_applied`) short-circuit - // every epoch-0 replay before it reaches the lock table — silently - // defeating the end-to-end drain tests below. - fully_applied_epoch: NOT_YET_APPLIED_EPOCH, - applied_tail: BTreeSet::new(), - rebuild_target_epoch: 0, - config: SchedulerConfig::default(), - metrics: SchedulerMetrics::new(), - read_result_rx, - lock_manager, - promotion_rx, - registry, - verdict_rx, - }); - (scheduler, dir) - } - - fn make_sequenced_txn(epoch: u64, position: u32) -> SequencedTxn { - let write_set = ReadWriteSet::new(vec![EngineKeySet::Document { - collection: "test_coll".to_string(), - surrogates: SortedVec::new(vec![1]), - }]); - let tx_class = TxClass::new_single_vshard( - ReadWriteSet::new(vec![]), - write_set, - vec![], - TenantId::new(1), - None, - VersionedReadSet::default(), - ) - .expect("valid TxClass"); - SequencedTxn { - epoch, - position, - tx_class, - epoch_system_ms: 1_700_000_000_000, - epoch_vshard_txn_count: 1, - lock_owner: None, - } - } + use crate::control::cluster::calvin::scheduler::lock_manager::{AcquireOutcome, TxnId}; #[tokio::test] async fn drain_catch_up_is_noop_when_no_drop_recorded() { @@ -396,8 +378,9 @@ mod tests { async fn drain_replays_dropped_input_into_lock_table_end_to_end() { // Use the vShard that "test_coll" hashes to, so the batch's fan-out targets — // and its replay decodes for — this scheduler's vShard. - let vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "test_coll").as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "test_coll") + .vshard() + .as_u32(); let (mut scheduler, _dir) = build_test_scheduler(vshard); ensure_sequencer_leader(&scheduler); @@ -484,8 +467,9 @@ mod tests { /// epoch 1 (delivered live, in-flight) must be skipped. #[tokio::test] async fn drain_skips_in_flight_overlap_no_double_dispatch_end_to_end() { - let vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "test_coll").as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "test_coll") + .vshard() + .as_u32(); let (mut scheduler, _dir) = build_test_scheduler(vshard); ensure_sequencer_leader(&scheduler); @@ -551,4 +535,132 @@ mod tests { "the guarded overlap must not cause a second dispatch" ); } + + /// Commit two single-txn batches at epochs 0 and 1, each a validate-only + /// txn that stages on this scheduler, and drop both through a full + /// fan-out channel. Returns the two committed Raft indexes. + fn arm_two_dropped_stage_batches(scheduler: &Scheduler) -> (u64, u64) { + ensure_sequencer_leader(scheduler); + let txn0 = make_validate_only_txn(0, 0); + let txn1 = make_validate_only_txn(1, 0); + let (idx0, bytes0) = commit_epoch_batch(scheduler, make_batch(0, &txn0)); + let (idx1, bytes1) = commit_epoch_batch(scheduler, make_batch(1, &txn1)); + assert!(idx1 > idx0, "second batch commits at a later Raft index"); + apply_with_full_channel(scheduler, scheduler.vshard_id, idx0, &bytes0, &txn0); + apply_with_full_channel(scheduler, scheduler.vshard_id, idx1, &bytes1, &txn1); + (idx0, idx1) + } + + /// Draining a replay range against a dispatcher at tenant capacity marks + /// no replayed position applied. + #[tokio::test] + async fn drain_against_full_dispatcher_marks_no_replayed_position_applied() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + arm_two_dropped_stage_batches(&scheduler); + let shared = std::sync::Arc::clone(&scheduler.shared); + fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + + scheduler.drain_catch_up(); + + assert!( + !scheduler.applied.is_applied(0, 0), + "the refused replayed txn must stay unapplied" + ); + assert!( + !scheduler.applied.is_applied(1, 0), + "a replayed txn after the refusal must stay unapplied" + ); + } + + /// Draining against a dispatcher at tenant capacity stops at the first + /// refusal and leaves catch-up armed from the first input it did not + /// process. + #[tokio::test] + async fn drain_against_full_dispatcher_stays_armed_from_first_unprocessed_input() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let (_idx0, idx1) = arm_two_dropped_stage_batches(&scheduler); + let shared = std::sync::Arc::clone(&scheduler.shared); + fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + + scheduler.drain_catch_up(); + + let armed = scheduler + .sequencer_state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()) + .peek_catch_up_from(scheduler.vshard_id); + assert_eq!( + armed, + Some(idx1), + "catch-up must stay armed from the input after the refused one" + ); + } + + /// A drain over a range longer than its window replays exactly the + /// window and stays armed at the first index past it. The next drain + /// continues from there. + #[tokio::test] + async fn drain_replays_one_window_then_resumes_past_it() { + let vshard = test_coll_vshard(); + let (mut scheduler, _dir) = build_test_scheduler(vshard); + scheduler.config.catch_up_window = 1; + ensure_sequencer_leader(&scheduler); + + let txn0 = make_sequenced_txn(0, 0); + let txn1 = make_sequenced_txn(1, 0); + let (idx0, bytes0) = commit_epoch_batch(&scheduler, make_batch(0, &txn0)); + let (idx1, bytes1) = commit_epoch_batch(&scheduler, make_batch(1, &txn1)); + assert_eq!(idx1, idx0 + 1, "the two batches commit at adjacent indexes"); + apply_with_full_channel(&scheduler, vshard, idx0, &bytes0, &txn0); + apply_with_full_channel(&scheduler, vshard, idx1, &bytes1, &txn1); + + // A conflicting holder on the shared key makes each replayed txn + // block, so nothing dispatches. + let keys = + crate::control::cluster::calvin::scheduler::driver::helpers::expand_rw_set(&txn0); + { + let mut lm = scheduler + .lock_manager + .lock() + .unwrap_or_else(|p| p.into_inner()); + assert_eq!( + lm.acquire(TxnId::new(u64::MAX, 0), keys), + AcquireOutcome::Ready + ); + } + let armed = |scheduler: &Scheduler| { + scheduler + .sequencer_state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()) + .peek_catch_up_from(vshard) + }; + + let first = scheduler.drain_catch_up(); + + assert_eq!(first, CatchUpDrain::Remaining); + assert!(scheduler.blocked.contains_key(&TxnId::new(0, 0))); + assert!( + !scheduler.blocked.contains_key(&TxnId::new(1, 0)), + "the entry past the window must not be replayed" + ); + assert_eq!( + armed(&scheduler), + Some(idx1), + "catch-up stays armed at the first index past the window" + ); + + let second = scheduler.drain_catch_up(); + + assert_eq!(second, CatchUpDrain::Settled); + assert!( + scheduler.blocked.contains_key(&TxnId::new(1, 0)), + "the next drain replays the entry past the first window" + ); + assert_eq!(armed(&scheduler), None, "the armed range is fully replayed"); + } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs index d9181f9bb..11c773826 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs @@ -12,10 +12,13 @@ //! `finish_resolved_commit` / `commit_apply_tail` complete. use super::super::types::CommitState; +use super::commit_resolution_dispatch::CommitResolution; +use super::deferred::{DispatchOutcome, DispatchStep}; +use super::halt::{HaltReason, HaltStep, error_response_text}; use super::scheduler::Scheduler; use crate::bridge::envelope::{Response, Status}; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; -use crate::control::cluster::calvin::scheduler::metrics::infra_abort_reason; +use crate::control::server::dispatch_utils::MintedRecords; use crate::types::VShardId; use crate::wal::{CalvinStamp, RedoRecord}; use nodedb_physical::physical_plan::PhysicalPlan; @@ -24,46 +27,39 @@ use nodedb_physical::physical_plan::meta::MetaOp; impl Scheduler { /// Handle the `MetaOp::CalvinResolve` response: decode the resolved /// `RedoRecord`, WAL-append it (unless its op set is empty), then dispatch - /// the flush stamped with that record's LSN. + /// the flush that installs it, stamped with that record's LSN. /// - /// A non-`Ok` response, a decode failure, or a WAL-append failure is a - /// loud infra abort — never a silent fall-through to a non-durable flush. + /// The verdict is already COMMIT, so a skipped resolve would tear the + /// committed txn on this replica. A non-`Ok` response, a decode failure, + /// or a WAL-append failure halts the scheduler: the txn keeps its + /// `pending` entry and locks, and its position stays unapplied. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn finish_redo_resolve( &mut self, txn_id: TxnId, response: Response, ) { if response.status != Status::Ok { - tracing::warn!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - "calvin: CalvinResolve response was not Ok; locks NOT released (shard degraded)" + self.halt_apply( + txn_id, + HaltReason::ResolveFailed, + HaltStep::Resolve, + error_response_text("CalvinResolve", &response), ); - self.abort_redo_resolve_infra_error(txn_id); return; } let mut redo = match RedoRecord::from_bytes(response.payload.as_bytes()) { Ok(r) => r, Err(e) => { - tracing::error!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - error = %e, - "calvin: CalvinResolve redo record decode failed" + self.halt_apply( + txn_id, + HaltReason::ResolveFailed, + HaltStep::Resolve, + format!("CalvinResolve redo record decode failed: {e}"), ); - self.abort_redo_resolve_infra_error(txn_id); return; } }; - redo.calvin_stamp = Some(CalvinStamp { - epoch: txn_id.epoch, - position: txn_id.position, - vshard_id: self.vshard_id, - }); - let Some(pending) = self.pending.get(&txn_id) else { // Txn state was reclaimed out from under us (should not happen — // locks are held until `on_txn_complete`); complete defensively. @@ -73,34 +69,82 @@ impl Scheduler { }; let tenant_id = pending.txn.tx_class.tenant_id; let database_id = pending.txn.tx_class.database_id; + // The stamp carries what the slice folds, so the live install and + // restart replay fold at this record's LSN. + redo.calvin_stamp = Some(CalvinStamp { + epoch: txn_id.epoch, + position: txn_id.position, + vshard_id: self.vshard_id, + collections: pending.flush_scope.collections.clone(), + sum_targets: pending.flush_scope.sum_targets.clone(), + }); - let redo_lsn = if redo.ops.is_empty() { - None + // The flush installs these exact bytes, the payload of the record + // appended below. + let redo_bytes = if redo.ops.is_empty() { + Vec::new() } else { - match self.shared.wal.append_transaction_redo( - tenant_id, - VShardId::new(self.vshard_id), - database_id, - &redo, - ) { - Ok(lsn) => Some(lsn), + match redo.to_bytes() { + Ok(bytes) => bytes, + Err(e) => { + self.halt_apply( + txn_id, + HaltReason::ResolveFailed, + HaltStep::Resolve, + format!("CalvinResolve redo record encode failed: {e}"), + ); + return; + } + } + }; + + // The record's outcome-floor window opens before the append. It stays + // with the pending txn until the flush completes. + let (redo_lsn, redo_records) = if redo.ops.is_empty() { + (None, None) + } else { + let records = MintedRecords::open(&self.shared.outcome_floor); + let appended = records + .appender(&self.shared.wal, crate::wal::manager::NO_APPLY_KEY) + .with_event_source(super::request::CALVIN_EVENT_SOURCE) + .append_transaction_redo( + tenant_id, + VShardId::new(self.vshard_id), + database_id, + &redo, + ); + match appended { + Ok(lsn) => { + // The txn committed, so its redo record is never + // cancelled. The flush closes it from its outcome. + records.mark_sent(); + (Some(lsn), Some(records)) + } Err(e) => { - tracing::error!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - error = %e, - "calvin: TransactionRedo WAL append failed" + // The txn stays pending and unapplied, and a failed append + // leaves no record for restart replay to reach. + records.settle(); + self.halt_apply( + txn_id, + HaltReason::WalAppendFailed, + HaltStep::RedoAppend, + format!("TransactionRedo WAL append failed: {e}"), ); - self.abort_redo_resolve_infra_error(txn_id); return; } } }; + if let Some(pending) = self.pending.get_mut(&txn_id) { + pending.redo_records = redo_records; + pending.flush_scope.redo = redo_bytes; + } - if !self.dispatch_commit_resolution(txn_id, true, redo_lsn) { - // `dispatch_commit_resolution` already logged the dispatch failure. - self.abort_redo_resolve_infra_error(txn_id); + // A flush refused at capacity is parked for re-send. The txn awaits its + // flush response either way, so the state below is the same. + if let DispatchOutcome::Failed(error) = + self.dispatch_commit_resolution(txn_id, CommitResolution::Flush { redo_lsn }) + { + self.fail_dispatch_step(txn_id, DispatchStep::Flush, error); return; } @@ -112,17 +156,6 @@ impl Scheduler { } } - /// Complete `txn_id` as an infra error: releases its locks so the epoch - /// advances rather than stalling. Shared by every `finish_redo_resolve` - /// failure branch. - fn abort_redo_resolve_infra_error(&mut self, txn_id: TxnId) { - self.metrics.record_executor_error(); - self.metrics - .record_infra_abort(infra_abort_reason::IO_ERROR); - self.metrics.record_completed(); - self.on_txn_complete(txn_id); - } - /// Dispatch `MetaOp::CalvinResolve` to this vShard's core, registering a /// response bridge so the resolve response re-enters the completion loop /// under `CommitState::AwaitingRedoResolve`. @@ -130,14 +163,15 @@ impl Scheduler { /// Mirrors `dispatch_commit_resolution`'s exempt, no-WAL-LSN dispatch /// shape — a resolve reads the staged overlay and writes nothing. /// - /// Returns `false` if the dispatch failed (the caller then completes the - /// txn as an infra error). + /// A capacity refusal returns [`DispatchOutcome::Deferred`]: the resolve + /// is parked for re-send and the txn stays in flight. A txn with no + /// `pending` entry returns [`DispatchOutcome::Failed`]. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn dispatch_calvin_resolve( &mut self, txn_id: TxnId, - ) -> bool { + ) -> DispatchOutcome { let Some(pending) = self.pending.get(&txn_id) else { - return false; + return DispatchOutcome::Failed(missing_pending_error(txn_id)); }; let tenant_id = pending.txn.tx_class.tenant_id; let database_id = pending.txn.tx_class.database_id; @@ -150,26 +184,79 @@ impl Scheduler { // itself, so no committed LSN rides on this envelope. let request = self.build_exempt_request(request_id, tenant_id, database_id, plan, None); - let resp_rx = self.shared.tracker.register(request_id); - let dispatch_result = match self.shared.dispatcher.lock() { - Ok(mut d) => d.dispatch(request), - Err(poisoned) => poisoned.into_inner().dispatch(request), - }; - if let Err(e) = dispatch_result { - self.shared.tracker.cancel(&request_id); - tracing::error!( - vshard_id = self.vshard_id, - epoch, - position, - error = %e, - "calvin: CalvinResolve dispatch failed" - ); - return false; - } - // The resolve response re-enters the completion loop under the SAME // txn_id, now in `AwaitingRedoResolve`, where `finish_redo_resolve` runs. - self.spawn_response_bridge(txn_id, request_id, resp_rx); - true + self.dispatch_sequenced(txn_id, DispatchStep::Resolve, request) + } +} + +/// The terminal error for a commit-resolution dispatch whose txn has no +/// `pending` entry to build the request from. +pub(in crate::control::cluster::calvin::scheduler::driver::core) fn missing_pending_error( + txn_id: TxnId, +) -> crate::Error { + crate::Error::Internal { + detail: format!( + "calvin txn {}/{} has no pending entry to dispatch from", + txn_id.epoch, txn_id.position + ), + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::{ErrorCode, Payload}; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + error_response, scheduler_with_pending, staged_response, + }; + + /// A resolve that returns an error under a COMMIT verdict holds the txn + /// unapplied and halts: skipping it would tear the committed txn. + #[tokio::test] + async fn resolve_error_response_holds_committed_txn_unapplied() { + let txn_id = TxnId::new(8, 0); + let (mut scheduler, _dir) = + scheduler_with_pending(txn_id, CommitState::AwaitingRedoResolve); + + scheduler.finish_redo_resolve( + txn_id, + error_response(ErrorCode::Internal { + detail: "resolve failed".to_string(), + }), + ); + + assert!(!scheduler.applied.is_applied(8, 0)); + assert!(scheduler.pending.contains_key(&txn_id)); + assert_eq!( + scheduler.apply_halt().map(|h| h.reason), + Some(HaltReason::ResolveFailed) + ); + assert!(scheduler.shared.sequencer_halt.apply_halt().is_halted()); + } + + /// A resolve whose redo record does not decode holds the txn unapplied + /// and halts. + #[tokio::test] + async fn undecodable_resolve_payload_holds_committed_txn_unapplied() { + let txn_id = TxnId::new(8, 0); + let (mut scheduler, _dir) = + scheduler_with_pending(txn_id, CommitState::AwaitingRedoResolve); + let mut response = staged_response(Status::Ok, None); + response.payload = Payload::from_vec(vec![0xff, 0x00, 0x13]); + + scheduler.finish_redo_resolve(txn_id, response); + + assert!(!scheduler.applied.is_applied(8, 0)); + assert!(scheduler.pending.contains_key(&txn_id)); + assert_eq!( + scheduler.apply_halt().map(|h| h.reason), + Some(HaltReason::ResolveFailed) + ); + assert_eq!( + scheduler.pending.get(&txn_id).and_then(|p| p.commit_state), + Some(CommitState::AwaitingRedoResolve), + "no flush is dispatched" + ); } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolution_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolution_dispatch.rs index 42dc11229..3eacfeceb 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolution_dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolution_dispatch.rs @@ -5,49 +5,69 @@ use nodedb_physical::physical_plan::PhysicalPlan; use nodedb_physical::physical_plan::meta::MetaOp; +use super::commit_redo::missing_pending_error; +use super::deferred::{DispatchOutcome, DispatchStep}; use super::scheduler::Scheduler; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; +use crate::types::Lsn; + +/// How a staged transaction resolves on this vShard. +pub(in crate::control::cluster::calvin::scheduler::driver::core) enum CommitResolution { + /// Install the committed redo record the scheduler appended at + /// `redo_lsn`. The flush scope holds its bytes. Both are empty when the + /// transaction wrote nothing on this vShard. + Flush { redo_lsn: Option }, + /// Discard the staged state under an abort verdict. + Drop, +} impl Scheduler { /// Dispatch a flush or drop of a staged transaction's commit-pending buffer. + /// + /// A flush carries the redo record, the collections the local plans + /// write, and their materialized-sum targets, so the Data Plane installs + /// the record the way every committed transaction installs. + /// + /// A capacity refusal returns [`DispatchOutcome::Deferred`]: the flush or + /// drop is parked for re-send and the txn stays in flight. A txn with no + /// `pending` entry returns [`DispatchOutcome::Failed`]. The flush takes its + /// collections and sum targets from the scope derived at stage time. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn dispatch_commit_resolution( &mut self, txn_id: TxnId, - committed: bool, - wal_lsn: Option, - ) -> bool { - let Some(pending) = self.pending.get(&txn_id) else { - return false; + resolution: CommitResolution, + ) -> DispatchOutcome { + let Some(pending) = self.pending.get_mut(&txn_id) else { + return DispatchOutcome::Failed(missing_pending_error(txn_id)); }; let tenant_id = pending.txn.tx_class.tenant_id; let database_id = pending.txn.tx_class.database_id; let epoch = txn_id.epoch; let position = txn_id.position; - let plan = if committed { - PhysicalPlan::Meta(MetaOp::CalvinFlush { epoch, position }) - } else { - PhysicalPlan::Meta(MetaOp::CalvinDrop { epoch, position }) + let (plan, step, wal_lsn) = match resolution { + CommitResolution::Flush { redo_lsn } => { + pending.flush_scope.sends = pending.flush_scope.sends.saturating_add(1); + let scope = &pending.flush_scope; + ( + PhysicalPlan::Meta(MetaOp::CalvinFlush { + epoch, + position, + redo: scope.redo.clone(), + collections: scope.collections.clone(), + sum_targets: scope.sum_targets.clone(), + }), + DispatchStep::Flush, + redo_lsn, + ) + } + CommitResolution::Drop => ( + PhysicalPlan::Meta(MetaOp::CalvinDrop { epoch, position }), + DispatchStep::Drop, + None, + ), }; let request_id = self.next_request_id(); let request = self.build_exempt_request(request_id, tenant_id, database_id, plan, wal_lsn); - let resp_rx = self.shared.tracker.register(request_id); - let dispatch_result = match self.shared.dispatcher.lock() { - Ok(mut dispatcher) => dispatcher.dispatch(request), - Err(poisoned) => poisoned.into_inner().dispatch(request), - }; - if let Err(error) = dispatch_result { - self.shared.tracker.cancel(&request_id); - tracing::error!( - vshard_id = self.vshard_id, - epoch, - position, - committed, - %error, - "calvin: commit resolution dispatch failed" - ); - return false; - } - self.spawn_response_bridge(txn_id, request_id, resp_rx); - true + self.dispatch_sequenced(txn_id, step, request) } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve.rs deleted file mode 100644 index bdf2b6f5a..000000000 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve.rs +++ /dev/null @@ -1,736 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Verdict-driven commit resolution for staged static Calvin transactions. -//! -//! A static Calvin dispatch STAGES its transaction on the Data Plane (validate -//! the read-set + buffer the plans, no base mutation). Its executor response -//! carries the local commit vote on `read_set_valid`. This module drives the -//! final step: dispatch a flush (commit, after `commit_redo` has WAL-appended -//! the resolved `TransactionRedo`) or drop (abort) of the staged buffer, wait -//! for its response, then run the commit tail (deposit applied result, record -//! write versions — plus a `CalvinApplied` WAL fallback when no redo record -//! was appended — propose `CompletionAck`) for a flush, or ack-only for a -//! drop. - -use std::sync::atomic::Ordering; -use std::time::Instant; - -use nodedb_cluster::calvin::{SequencerEntry, VerdictSignal}; - -use super::super::types::CommitState; -use super::scheduler::Scheduler; -use super::staged_vote::{StagedVote, staged_commit_vote}; -use crate::bridge::envelope::{Response, Status}; -use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; -use crate::control::cluster::calvin::scheduler::metrics::infra_abort_reason; - -impl Scheduler { - /// Cast this participant's local commit vote for a staged transaction, then - /// PARK it on the cross-shard commit barrier awaiting the durable GLOBAL - /// verdict — it does NOT self-decide flush-or-drop on its local vote. - /// - /// The staged executor response is validate-only: its `read_set_valid` is - /// this shard's local commit vote (`Some(true)` => commit, `Some(false)` => - /// abort; a `None` from the active/dependent path is treated as commit). The - /// leader proposes that vote via the sequencer Raft group; the sequencer - /// aggregates all participants' votes into a single authoritative - /// `SequencerEntry::Verdict`, applied on every replica. - /// - /// This method moves the txn to [`CommitState::AwaitingVerdict`] WITHOUT - /// dispatching a resolve or drop, then immediately probes - /// `registry.verdict(txn)`: if the verdict is already durable (replay, or a - /// push we raced) it resumes at once via [`Self::resume_on_verdict`]; - /// otherwise it stays parked, holding locks and its staged buffer, until the - /// verdict push, a later probe, or the stall re-probe sweep delivers the - /// verdict. Resuming (in `resume_on_verdict`) is where the flush/drop is - /// dispatched and the flushed/dropped counters bump — using the GLOBAL - /// verdict, never the local vote. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn resolve_staged_commit( - &mut self, - txn_id: TxnId, - staged_response: &Response, - ) { - // A staged error is always an abort vote. Only successful staged - // responses may use `None` for the dependent-read path; accepting an - // error-plus-None as commit would let a failed participant flush after - // its peers received a global commit verdict. - let vote = staged_commit_vote(staged_response); - - // Durably propose this participant's commit vote via the sequencer - // Raft group, leader-guarded like `OllpMismatch`: only the data-group - // leader ran read-set validation, so only a leader's vote is - // authoritative. The sequencer aggregates every participant's vote into - // the single global verdict this txn parks on below. An abort travels as - // `AbortVote` so its cause survives to the coordinator. - if self.is_group_leader() { - let entry = match vote.abort_reason() { - Some(reason) => SequencerEntry::AbortVote { - epoch: txn_id.epoch, - position: txn_id.position, - vshard: self.vshard_id, - reason, - }, - None => SequencerEntry::Vote { - epoch: txn_id.epoch, - position: txn_id.position, - vshard: self.vshard_id, - commit: true, - }, - }; - self.propose_sequencer_entry(entry, txn_id, "commit vote"); - } - - if vote == StagedVote::SerializationConflict { - // The staged slice's read-set was no longer current: observe it, the - // same node-global signal the direct-apply path records. A - // participant error never validated a read-set, so it must not count - // here. - self.shared - .calvin_counters - .read_set_validation_failures - .fetch_add(1, Ordering::Relaxed); - } - - // PARK on the barrier: transition to `AwaitingVerdict` and arm the stall - // deadline. Do NOT dispatch resolve/drop here — the GLOBAL verdict, not - // this local vote, decides. If the txn already vanished (torn down - // elsewhere), there is nothing to park. - match self.pending.get_mut(&txn_id) { - Some(pending) => { - pending.commit_state = Some(CommitState::AwaitingVerdict); - // no-determinism: local stall-warning deadline only; the global replicated verdict, not this wall-clock, decides commit/abort. - pending.verdict_deadline = Some(Instant::now() + self.config.verdict_stall_warn()); - } - None => return, - } - - // PROBE on park (correctness backstop): the verdict may already be - // durable — on replay, or a push that raced ahead of this park. Resume - // immediately if so; the double-resume guard in `resume_on_verdict` - // makes a later duplicate push/probe a no-op. - if let Some(verdict) = self.registry.verdict(nodedb_cluster::calvin::TxnId::new( - txn_id.epoch, - txn_id.position, - )) { - self.resume_on_verdict(txn_id, verdict); - } - } - - /// Resume a txn parked in [`CommitState::AwaitingVerdict`] once the durable - /// GLOBAL verdict is known: dispatch its flush (commit) or drop (abort). - /// - /// `committed` is the authoritative cross-shard verdict — NOT this shard's - /// local vote. On commit, dispatches `MetaOp::CalvinResolve` and moves the - /// txn to [`CommitState::AwaitingRedoResolve`] (the resolved redo is - /// WAL-appended and the flush dispatched from [`Self::finish_redo_resolve`]). - /// On abort, dispatches the drop directly and moves the txn to - /// [`CommitState::AwaitingResolve`]. Bumps the flushed / dropped counter. The - /// commit tail runs later in [`Self::finish_resolved_commit`], once the - /// flush/drop response arrives. - /// - /// Double-resume guard: the verdict push and the probe-on-park (and the - /// stall re-probe sweep) can all fire for one txn, so this first confirms the - /// txn is still `Some(AwaitingVerdict)` — if it already transitioned out - /// (resolve/drop dispatched, or completed), this is a no-op. This guarantees - /// the flush/drop is dispatched exactly once. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn resume_on_verdict( - &mut self, - txn_id: TxnId, - committed: bool, - ) { - // Guard: only a still-parked txn resumes. Mirrors `handle_completion`'s - // state-match so a duplicate push/probe/timeout is idempotent. - if !matches!( - self.pending.get(&txn_id).and_then(|p| p.commit_state), - Some(CommitState::AwaitingVerdict) - ) { - return; - } - - let dispatched = if committed { - // Resolve the staged post-images into a replayable `RedoRecord` - // first; the redo is WAL-appended (in `finish_redo_resolve`) before - // the flush is dispatched, restoring restart durability for this - // vShard's slice of a multi-shard Calvin commit. - self.dispatch_calvin_resolve(txn_id) - } else { - self.dispatch_commit_resolution(txn_id, false, None) - }; - if !dispatched { - // Resolve/drop dispatch failed: complete the txn as an infra error so - // its locks release and the epoch advances rather than stalling. The - // staged buffer is reclaimed by a later drop or on core teardown. - self.metrics.record_executor_error(); - self.metrics - .record_infra_abort(infra_abort_reason::IO_ERROR); - self.metrics.record_completed(); - self.on_txn_complete(txn_id); - return; - } - - if let Some(pending) = self.pending.get_mut(&txn_id) { - pending.commit_state = Some(if committed { - CommitState::AwaitingRedoResolve - } else { - CommitState::AwaitingResolve { - committed: false, - redo_lsn: None, - } - }); - // No longer parked: clear the stall deadline. - pending.verdict_deadline = None; - } - - if committed { - self.shared - .calvin_counters - .commits_flushed - .fetch_add(1, Ordering::Relaxed); - } else { - self.shared - .calvin_counters - .commits_dropped - .fetch_add(1, Ordering::Relaxed); - } - } - - /// Handle a pushed [`VerdictSignal`] from this node's completion registry. - /// - /// Matches the signal to the parked txn by `(epoch, position)` and resumes - /// it. A signal for a txn this scheduler does not host, or one that already - /// resumed, is a harmless no-op (the double-resume guard covers the latter). - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn handle_verdict_signal( - &mut self, - signal: VerdictSignal, - ) { - let txn_id = TxnId::new(signal.epoch, signal.position); - self.resume_on_verdict(txn_id, signal.verdict.is_commit()); - } - - /// Sweep parked `AwaitingVerdict` txns whose stall deadline has passed. - /// - /// For each stalled txn, RE-PROBE the durable verdict: if it is now known, - /// resume (a push we dropped on a full channel, or a verdict that landed - /// after the last probe). If it is STILL unknown, KEEP WAITING — hold locks, - /// emit a stall metric + warning, and re-arm the deadline so the warning is - /// rate-limited rather than per-iteration. It NEVER releases locks and NEVER - /// unilaterally aborts: a participant cannot know whether a peer already - /// flushed a COMMIT, so aborting one side while a peer committed would tear - /// the transaction. The verdict is guaranteed to arrive eventually — a - /// post-failover leader re-aggregates the replicated votes (seeded on every - /// replica) into the same verdict — so waiting is always the safe action. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn check_awaiting_verdict_stalls( - &mut self, - ) { - // no-determinism: stall-detection clock drives warnings/metrics only; this path holds locks and never aborts, so it cannot affect the replicated outcome. - let now = Instant::now(); - let stalled: Vec = self - .pending - .iter() - .filter(|(_, p)| matches!(p.commit_state, Some(CommitState::AwaitingVerdict))) - .filter(|(_, p)| p.verdict_deadline.is_some_and(|d| now >= d)) - .map(|(id, _)| *id) - .collect(); - - for txn_id in stalled { - if let Some(verdict) = self.registry.verdict(nodedb_cluster::calvin::TxnId::new( - txn_id.epoch, - txn_id.position, - )) { - self.resume_on_verdict(txn_id, verdict); - continue; - } - - // Verdict still unknown: keep waiting, hold locks, never abort. - self.metrics.record_verdict_stall(); - tracing::warn!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - "calvin: staged txn still awaiting the cross-shard verdict past its stall \ - deadline; HOLDING locks and waiting (never aborting — a peer may have already \ - flushed a commit). The verdict is guaranteed to arrive." - ); - if let Some(pending) = self.pending.get_mut(&txn_id) { - pending.verdict_deadline = Some(now + self.config.verdict_stall_warn()); - } - } - } - - /// Run the commit tail once a flush/drop response has returned. - /// - /// On a successful flush the full commit tail runs (deposit applied result, - /// `CalvinApplied` WAL + write-version recording, `CompletionAck`). On a - /// successful drop only the `CompletionAck` is proposed — the coordinator's - /// completion waiter still fires and the epoch advances, but nothing was - /// written so there is no result to deposit, no apply LSN, and no versions - /// to record. A non-`Ok` resolve response is treated as an executor error. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn finish_resolved_commit( - &mut self, - txn_id: TxnId, - response: Response, - committed: bool, - redo_lsn: Option, - ) { - let completed = if response.status == Status::Ok { - if committed { - self.commit_apply_tail(txn_id, response, redo_lsn) - } else { - self.propose_sequencer_entry( - SequencerEntry::CompletionAck { - epoch: txn_id.epoch, - position: txn_id.position, - vshard_id: self.vshard_id, - }, - txn_id, - "completion ack (dropped)", - ); - true - } - } else { - tracing::error!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - committed, - "calvin: flush/drop response was not Ok while applying an already-committed \ - verdict; forcing infra-abort completion so locks release and the epoch advances" - ); - false - }; - - if completed { - self.metrics.record_completed(); - self.on_txn_complete(txn_id); - } else { - // The cross-shard verdict is already globally durable, and a commit's - // resolved redo was WAL-appended before this flush — so recovery - // re-applies the write. A local flush/apply or WAL-marker failure is - // therefore an infrastructure event, NOT an outcome change. It must - // never leave the txn parked: holding its locks forever wedges every - // txn queued behind those keys and freezes this vShard's epoch - // watermark (which anchors cross-shard BEGIN snapshots), and nothing - // re-drives a non-`AwaitingVerdict` pending entry. Surface the infra - // abort and force completion — the same forward-progress contract the - // resolve/drop dispatch-failure path in `resume_on_verdict` follows. - self.metrics.record_executor_error(); - self.metrics - .record_infra_abort(infra_abort_reason::IO_ERROR); - self.metrics.record_completed(); - self.on_txn_complete(txn_id); - } - } - - /// Deposit the applied result, durably mark the apply, record the apply's - /// write versions, and propose the `CompletionAck`. - /// - /// Shared by the flush-completion path and the direct-apply (dependent / - /// active) apply path. - /// - /// `redo_lsn` is `Some(lsn)` when a `TransactionRedo` record was already - /// WAL-appended for this commit's non-empty write set (`finish_redo_resolve`) - /// — that record already IS the durable applied marker, so only write - /// versions are recorded at it. `None` (a drop, an empty-ops staged commit, - /// or the direct-apply dependent/active path, which carries no redo record) - /// falls back to appending a `CalvinApplied` marker here, exactly as before - /// this record existed. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn commit_apply_tail( - &mut self, - txn_id: TxnId, - response: Response, - redo_lsn: Option, - ) -> bool { - // Deposit the FULL applied Response (affected-count + watermark + any - // RETURNING rows) into the local sidecar BEFORE proposing the replicated - // CompletionAck. The ack fires the coordinator's completion oneshot on - // every sequencer member, so depositing first guarantees the result is - // present by the time the coordinator drains it — no lost result, no - // race. - // - // Gated on the PRIMARY-WRITE participant: any participant whose slice - // carries the user's non-edge DML (Document/KV/Vector/etc.), as opposed - // to the implicit graph-edge cleanup that dual-homes alongside it. A - // multi-collection cross-shard COMMIT has MANY primary-write - // participants — each a plain affected-count write — and they coalesce: - // the first applied response stands for the coordinator (which discards - // it for a COMMIT tag anyway), and the plain-write siblings do not - // conflict. Only a genuine cross-shard RETURNING union — two - // participants each carrying RETURNING rows — records `Conflict`. - // Results travel via this in-process sidecar only — never the sequencer - // Raft log. - let (has_primary_write, has_returning) = self - .pending - .get(&txn_id) - .map(|p| (p.has_primary_write, p.has_returning)) - .unwrap_or((false, false)); - if has_primary_write { - use std::collections::hash_map::Entry; - - use crate::control::state::CalvinApplyResult; - - let key = nodedb_cluster::calvin::TxnId::new(txn_id.epoch, txn_id.position); - let mut results = self - .shared - .calvin_apply_results - .lock() - .unwrap_or_else(|p| p.into_inner()); - match results.entry(key) { - Entry::Vacant(slot) => { - slot.insert(CalvinApplyResult::Single { - response, - has_returning, - }); - } - Entry::Occupied(mut slot) => { - // Derive both facts from the existing entry BEFORE any - // insert, so the immutable borrow does not outlive the - // mutable one. - let existing_returning = matches!( - slot.get(), - CalvinApplyResult::Single { - has_returning: true, - .. - } - ); - let already_conflict = matches!(slot.get(), CalvinApplyResult::Conflict); - - if already_conflict { - // A RETURNING union was already recorded; stays Conflict. - } else if has_returning && existing_returning { - // Two RETURNING-bearing participants for one Calvin txn: - // a cross-shard RETURNING union, which is unsupported. - // Record Conflict so the coordinator fails the statement - // loudly rather than returning one shard's partial rows. - tracing::error!( - epoch = txn_id.epoch, - position = txn_id.position, - vshard = self.vshard_id, - "two RETURNING-bearing participants for one Calvin txn — cross-shard \ - RETURNING union unsupported" - ); - slot.insert(CalvinApplyResult::Conflict); - } else if has_returning { - // The incoming participant carries the rows; the existing - // entry was a plain affected-count sibling. Rows win. - slot.insert(CalvinApplyResult::Single { - response, - has_returning: true, - }); - } else { - // Incoming is a plain write; keep the existing entry — a - // multi-collection cross-shard COMMIT coalesces (the - // coordinator discards it for a COMMIT tag anyway). - } - } - } - } - let applied_lsn = match redo_lsn { - // The TransactionRedo record already durably marks this apply — the - // SAME shard-local WAL-LSN space fast-path writes and read - // watermarks use. Record the apply's per-key write versions at it; - // no second (CalvinApplied) marker is written. - Some(lsn) => { - self.record_calvin_write_versions(txn_id, lsn); - Some(lsn) - } - None => match self.shared.wal.append_calvin_applied( - crate::types::VShardId::new(self.vshard_id), - txn_id.epoch, - txn_id.position, - ) { - // The CalvinApplied WAL LSN is the committed write-LSN for this - // apply — the SAME shard-local WAL-LSN space fast-path writes and - // read watermarks use. Record the apply's per-key write versions - // at it once it exists; it does not exist yet at dispatch time. - Ok(applied_lsn) => { - self.record_calvin_write_versions(txn_id, applied_lsn); - Some(applied_lsn) - } - Err(e) => { - tracing::error!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - error = %e, - "calvin: failed to write CalvinApplied WAL record" - ); - None - } - }, - }; - let Some(lsn) = applied_lsn else { - // The apply cannot be acknowledged without a durable participant - // LSN: CDC and write-version consumers would otherwise observe a - // successful commit with no authoritative ordering point. - return false; - }; - // Control change-stream events are distinct from Data-Plane - // WriteEvents. Publish the participant-local logical manifests once, - // from the data-group leader, at the authoritative committed LSN. - if self.is_group_leader() - && let Some(pending) = self.pending.get_mut(&txn_id) - { - let tenant_id = pending.txn.tx_class.tenant_id; - let database_id = pending.txn.tx_class.database_id; - for change_set in std::mem::take(&mut pending.change_sets) { - crate::control::server::dispatch_utils::publish_change_set_with_lsn( - &self.shared, - tenant_id, - database_id, - change_set, - lsn, - ); - } - } - self.propose_sequencer_entry( - SequencerEntry::CompletionAck { - epoch: txn_id.epoch, - position: txn_id.position, - vshard_id: self.vshard_id, - }, - txn_id, - "completion ack", - ); - true - } -} - -#[cfg(test)] -mod tests { - use super::*; - use std::collections::HashMap; - use std::sync::{Arc, Mutex}; - - use nodedb_cluster::MultiRaft; - use nodedb_cluster::RoutingTable; - use nodedb_cluster::calvin::types::{ - EngineKeySet, ReadWriteSet, SequencedTxn, SortedVec, TxClass, VersionedReadSet, - }; - use nodedb_cluster::calvin::{ - AbortReason, CalvinCompletionRegistry, ParticipantVote, SequencerStateMachine, - VerdictOutcome, - }; - use nodedb_physical::physical_plan::PhysicalPlan; - use nodedb_physical::physical_plan::meta::MetaOp; - use nodedb_types::TenantId; - - use super::super::scheduler::{Scheduler, SchedulerParams}; - use crate::bridge::dispatch::Dispatcher; - use crate::bridge::envelope::Payload; - use crate::control::cluster::calvin::scheduler::driver::types::PendingTxn; - use crate::control::cluster::calvin::scheduler::lock_manager::LockManager; - use crate::control::cluster::calvin::scheduler::metrics::SchedulerMetrics; - use crate::control::cluster::calvin::scheduler::{NOT_YET_APPLIED_EPOCH, SchedulerConfig}; - use crate::control::state::SharedState; - use crate::types::{Lsn, RequestId}; - use crate::wal::WalManager; - - /// Same minimal scheduler fixture as the process/catch_up driver tests, - /// retaining its Data-Plane request receiver for tests that must observe - /// scheduler dispatches. - fn build_test_scheduler_with_data_side( - vshard_id: u32, - registry: Arc, - ) -> ( - Scheduler, - tempfile::TempDir, - crate::bridge::dispatch::CoreChannelDataSide, - ) { - let dir = tempfile::tempdir().unwrap(); - let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("test.wal")).unwrap()); - let (dispatcher, mut data_sides) = Dispatcher::new(1, 64); - let data_side = data_sides - .pop() - .expect("one configured core has one data side"); - let shared = SharedState::new(dispatcher, wal).unwrap(); - - let rt = RoutingTable::uniform(1, &[1], 1); - let multi_raft = Arc::new(Mutex::new(MultiRaft::new(1, rt, dir.path().to_path_buf()))); - - let sequencer_state_machine = Arc::new(Mutex::new(SequencerStateMachine::new( - HashMap::new(), - Arc::clone(®istry), - ))); - - let (_tx, receiver) = tokio::sync::mpsc::channel(16); - let (_rr_tx, read_result_rx) = tokio::sync::mpsc::channel(16); - let (_prom_tx, promotion_rx) = tokio::sync::mpsc::unbounded_channel(); - let (verdict_tx, verdict_rx) = tokio::sync::mpsc::channel(16); - registry.register_verdict_signal_sender(vshard_id, verdict_tx); - - let lock_manager = Arc::new(Mutex::new(LockManager::new())); - - let scheduler = Scheduler::new(SchedulerParams { - vshard_id, - receiver, - shared, - multi_raft, - sequencer_state_machine, - fully_applied_epoch: NOT_YET_APPLIED_EPOCH, - applied_tail: std::collections::BTreeSet::new(), - rebuild_target_epoch: 0, - config: SchedulerConfig::default(), - metrics: SchedulerMetrics::new(), - read_result_rx, - lock_manager, - promotion_rx, - registry, - verdict_rx, - }); - (scheduler, dir, data_side) - } - - /// Build a static-write `SequencedTxn` at `(epoch, position)`. - fn staged_pending(txn: SequencedTxn, txn_id: TxnId) -> PendingTxn { - PendingTxn { - txn, - lock_owner: txn_id, - dispatch_time: std::time::Instant::now(), - has_primary_write: true, - has_returning: false, - change_sets: Vec::new(), - commit_state: Some(CommitState::Staged), - verdict_deadline: None, - } - } - - fn staged_response(status: Status, read_set_valid: Option) -> Response { - Response { - request_id: RequestId::new(1), - status, - attempt: 1, - partial: false, - payload: Payload::empty(), - watermark_lsn: Lsn::ZERO, - error_code: None, - read_set_valid, - read_version_lsn: Lsn::ZERO, - write_set: Vec::new(), - } - } - - fn make_sequenced_txn(epoch: u64, position: u32) -> SequencedTxn { - let write_set = ReadWriteSet::new(vec![EngineKeySet::Document { - collection: "test_coll".to_string(), - surrogates: SortedVec::new(vec![1]), - }]); - let tx_class = TxClass::new_single_vshard( - ReadWriteSet::new(vec![]), - write_set, - vec![], - TenantId::new(1), - None, - VersionedReadSet::default(), - ) - .expect("valid TxClass"); - SequencedTxn { - epoch, - position, - tx_class, - epoch_system_ms: 1_700_000_000_000, - epoch_vshard_txn_count: 1, - lock_owner: None, - } - } - - /// A false vote from either participant makes the only global verdict abort; - /// applying that durable verdict broadcasts the abort to every parked local - /// participant. The scheduler's `resume_on_verdict(false)` then dispatches a - /// drop, never a resolve/flush, on each recipient. - #[tokio::test] - async fn two_participant_false_vote_broadcasts_global_abort_to_every_scheduler() { - let registry = CalvinCompletionRegistry::new_detached(); - let txn = nodedb_cluster::calvin::TxnId::new(14, 2); - let txn_id = TxnId::new(14, 2); - let (mut first_scheduler, _first_dir, mut first_data) = - build_test_scheduler_with_data_side(7, Arc::clone(®istry)); - let (mut second_scheduler, _second_dir, mut second_data) = - build_test_scheduler_with_data_side(9, Arc::clone(®istry)); - first_scheduler - .pending - .insert(txn_id, staged_pending(make_sequenced_txn(14, 2), txn_id)); - second_scheduler - .pending - .insert(txn_id, staged_pending(make_sequenced_txn(14, 2), txn_id)); - - // Local staging votes only park their own staged slices; neither the - // affirmative nor the failed participant may resolve or drop unilaterally. - first_scheduler.resolve_staged_commit(txn_id, &staged_response(Status::Ok, Some(true))); - second_scheduler.resolve_staged_commit(txn_id, &staged_response(Status::Error, None)); - for (scheduler, data_side) in [ - (&first_scheduler, &mut first_data), - (&second_scheduler, &mut second_data), - ] { - assert!(matches!( - scheduler - .pending - .get(&txn_id) - .and_then(|pending| pending.commit_state), - Some(CommitState::AwaitingVerdict) - )); - assert!(data_side.request_rx.try_pop().is_err()); - } - - // Model the replicated vote entries and their resulting durable verdict. - // The shared registry sends each scheduler's actual registered channel. - registry.seed_expected(txn, 2); - registry.note_vote(txn, 7, ParticipantVote::Commit); - assert!(registry.drain_unproposed_verdicts().is_empty()); - registry.note_vote( - txn, - 9, - ParticipantVote::Abort(Some(AbortReason::SerializationConflict)), - ); - assert_eq!( - registry.drain_unproposed_verdicts(), - vec![( - txn, - VerdictOutcome::Abort(Some(AbortReason::SerializationConflict)) - )] - ); - registry.note_verdict( - txn, - VerdictOutcome::Abort(Some(AbortReason::SerializationConflict)), - ); - assert_eq!(registry.verdict(txn), Some(false)); - - let first_signal = first_scheduler - .verdict_rx - .try_recv() - .expect("registry must signal the first registered scheduler"); - let second_signal = second_scheduler - .verdict_rx - .try_recv() - .expect("registry must signal the second registered scheduler"); - first_scheduler.handle_verdict_signal(first_signal); - second_scheduler.handle_verdict_signal(second_signal); - - for (scheduler, data_side) in [ - (&first_scheduler, &mut first_data), - (&second_scheduler, &mut second_data), - ] { - assert!(matches!( - scheduler - .pending - .get(&txn_id) - .and_then(|pending| pending.commit_state), - Some(CommitState::AwaitingResolve { - committed: false, - redo_lsn: None - }) - )); - let request = data_side - .request_rx - .try_pop() - .expect("global abort must dispatch a drop to every participant"); - assert!(matches!( - request.inner.plan, - PhysicalPlan::Meta(MetaOp::CalvinDrop { - epoch: 14, - position: 2 - }) - )); - assert!(data_side.request_rx.try_pop().is_err()); - } - } -} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs new file mode 100644 index 000000000..48a491c9a --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs @@ -0,0 +1,490 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Commit tail: runs once a flush/drop response has returned, depositing the +//! applied result, marking the apply durable, recording write versions, and +//! proposing the `CompletionAck`. + +use crate::bridge::envelope::{ErrorCode, Response, Status}; +use crate::control::cluster::calvin::scheduler::driver::core::commit_resolution_dispatch::CommitResolution; +use crate::control::cluster::calvin::scheduler::driver::core::deferred::{ + DispatchOutcome, DispatchStep, +}; +use crate::control::cluster::calvin::scheduler::driver::core::halt::{ + HaltReason, HaltStep, error_response_text, +}; +use crate::control::cluster::calvin::scheduler::driver::core::owed::SchedulerProposal; +use crate::control::cluster::calvin::scheduler::driver::core::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; +use crate::control::cluster::calvin::scheduler::metrics::infra_abort_reason; + +/// The most flushes one committed txn sends. Each refused install rolled +/// every write back, so a resend is safe. A refusal that outlasts the bound +/// halts the scheduler. +pub(in crate::control::cluster::calvin::scheduler::driver::core) const MAX_FLUSH_SENDS: u32 = 8; + +impl Scheduler { + /// Run the commit tail once a flush/drop response has returned. + /// + /// On a successful flush the full commit tail runs (deposit applied result, + /// `CalvinApplied` WAL + write-version recording, `CompletionAck`). On a + /// successful drop only the `CompletionAck` is proposed — the coordinator's + /// completion waiter still fires and the epoch advances, but nothing was + /// written so there is no result to deposit, no apply LSN, and no versions + /// to record. + /// + /// A flush refused with `RetryableRefusal` rolled its install back and + /// kept the staged buffer, so the scheduler sends the same flush again, up + /// to [`MAX_FLUSH_SENDS`] sends. Any other non-`Ok` flush, or a refusal + /// past the bound, halts the scheduler: a skipped flush tears the + /// committed txn on this replica. A non-`Ok` drop completes the txn: + /// under an abort verdict no replica writes anything. + pub(in crate::control::cluster::calvin::scheduler::driver::core) async fn finish_resolved_commit( + &mut self, + txn_id: TxnId, + response: Response, + committed: bool, + redo_lsn: Option, + ) { + if response.status != Status::Ok { + if committed && self.resend_refused_flush(txn_id, &response, redo_lsn) { + return; + } + if committed { + self.halt_apply( + txn_id, + HaltReason::FlushFailed, + HaltStep::Flush, + error_response_text("CalvinFlush", &response), + ); + return; + } + tracing::error!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + "calvin: drop response was not Ok under an abort verdict; completing the \ + aborted txn, since no replica writes it" + ); + self.metrics.record_executor_error(); + self.metrics + .record_infra_abort(infra_abort_reason::IO_ERROR); + self.metrics.record_completed(); + self.on_txn_complete(txn_id); + return; + } + + let completed = if committed { + self.commit_apply_tail(txn_id, response, redo_lsn).await + } else { + self.propose_sequencer_entry(txn_id, SchedulerProposal::CompletionAck); + true + }; + // `false` means the commit tail halted the scheduler: the txn stays + // pending and unapplied. + if completed { + self.metrics.record_completed(); + self.on_txn_complete(txn_id); + } + } + + /// Send the flush of `txn_id` again when the install refused it as + /// retryable and the send bound allows another. Returns whether the + /// refusal is handled: the flush went out again, or its dispatch failed + /// and the step's terminal handling ran. + fn resend_refused_flush( + &mut self, + txn_id: TxnId, + response: &Response, + redo_lsn: Option, + ) -> bool { + if !matches!( + response.error_code.as_deref(), + Some(ErrorCode::RetryableRefusal { .. }) + ) { + return false; + } + let sends = self + .pending + .get(&txn_id) + .map_or(MAX_FLUSH_SENDS, |pending| pending.flush_scope.sends); + if sends >= MAX_FLUSH_SENDS { + return false; + } + tracing::warn!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + sends, + "calvin: the flush install was refused as retryable; sending it again" + ); + if let DispatchOutcome::Failed(error) = + self.dispatch_commit_resolution(txn_id, CommitResolution::Flush { redo_lsn }) + { + self.fail_dispatch_step(txn_id, DispatchStep::Flush, error); + } + true + } + + /// Deposit the applied result, durably mark the apply, record the apply's + /// write versions, and propose the `CompletionAck`. + /// + /// Shared by the flush-completion path and the direct-apply (dependent / + /// active) apply path. + /// + /// Returns `false` once a failed `CalvinApplied` WAL append halted the + /// scheduler: the position must not be marked applied without its marker, + /// so the caller leaves the txn pending. + /// + /// `redo_lsn` is `Some(lsn)` when a `TransactionRedo` record was already + /// WAL-appended for this commit's non-empty write set (`finish_redo_resolve`) + /// — that record already IS the durable applied marker, so only write + /// versions are recorded at it. `None` (a drop, an empty-ops staged commit, + /// or the direct-apply dependent/active path, which carries no redo record) + /// falls back to appending a `CalvinApplied` marker here, exactly as before + /// this record existed. + pub(in crate::control::cluster::calvin::scheduler::driver::core) async fn commit_apply_tail( + &mut self, + txn_id: TxnId, + response: Response, + redo_lsn: Option, + ) -> bool { + // The install folded materialized sums into target rows no redo + // sub-record names. The redo record's stamp carries the sum targets, + // so restart replay folds at the same LSN. + // Deposit the FULL applied Response (affected-count + watermark + any + // RETURNING rows) into the local sidecar BEFORE proposing the replicated + // CompletionAck. The ack fires the coordinator's completion oneshot on + // every sequencer member, so depositing first guarantees the result is + // present by the time the coordinator drains it — no lost result, no + // race. + // + // Gated on the PRIMARY-WRITE participant: any participant whose slice + // carries the user's non-edge DML (Document/KV/Vector/etc.), as opposed + // to the implicit graph-edge cleanup that dual-homes alongside it. A + // multi-collection cross-shard COMMIT has MANY primary-write + // participants — each a plain affected-count write — and they coalesce: + // the first applied response stands for the coordinator (which discards + // it for a COMMIT tag anyway), and the plain-write siblings do not + // conflict. Only a genuine cross-shard RETURNING union — two + // participants each carrying RETURNING rows — records `Conflict`. + // Results travel via this in-process sidecar only — never the sequencer + // Raft log. + let (has_primary_write, has_returning) = self + .pending + .get(&txn_id) + .map(|p| (p.has_primary_write, p.has_returning)) + .unwrap_or((false, false)); + if has_primary_write { + use std::collections::hash_map::Entry; + + let response = statement_reply(response); + + use crate::control::state::CalvinApplyResult; + + let key = nodedb_cluster::calvin::TxnId::new(txn_id.epoch, txn_id.position); + let mut results = self + .shared + .calvin + .apply_results + .lock() + .unwrap_or_else(|p| p.into_inner()); + match results.entry(key) { + Entry::Vacant(slot) => { + slot.insert(CalvinApplyResult::Single { + response, + has_returning, + }); + } + Entry::Occupied(mut slot) => { + // Derive both facts from the existing entry BEFORE any + // insert, so the immutable borrow does not outlive the + // mutable one. + let existing_returning = matches!( + slot.get(), + CalvinApplyResult::Single { + has_returning: true, + .. + } + ); + let already_conflict = matches!(slot.get(), CalvinApplyResult::Conflict); + + if already_conflict { + // A RETURNING union was already recorded; stays Conflict. + } else if has_returning && existing_returning { + // Two RETURNING-bearing participants for one Calvin txn: + // a cross-shard RETURNING union, which is unsupported. + // Record Conflict so the coordinator fails the statement + // loudly rather than returning one shard's partial rows. + tracing::error!( + epoch = txn_id.epoch, + position = txn_id.position, + vshard = self.vshard_id, + "two RETURNING-bearing participants for one Calvin txn — cross-shard \ + RETURNING union unsupported" + ); + slot.insert(CalvinApplyResult::Conflict); + } else if has_returning { + // The incoming participant carries the rows; the existing + // entry was a plain affected-count sibling. Rows win. + slot.insert(CalvinApplyResult::Single { + response, + has_returning: true, + }); + } else { + // Incoming is a plain write; keep the existing entry — a + // multi-collection cross-shard COMMIT coalesces (the + // coordinator discards it for a COMMIT tag anyway). + } + } + } + } + let applied_lsn = match redo_lsn { + // The TransactionRedo record already durably marks this apply — the + // SAME shard-local WAL-LSN space fast-path writes and read + // watermarks use. Record the apply's per-key write versions at it; + // no second (CalvinApplied) marker is written. + Some(lsn) => { + self.record_calvin_write_versions(txn_id, lsn); + Some(lsn) + } + None => match self + .shared + .wal + .appender(crate::wal::manager::NO_APPLY_KEY) + .append_calvin_applied( + crate::types::VShardId::new(self.vshard_id), + txn_id.epoch, + txn_id.position, + ) { + // The CalvinApplied WAL LSN is the committed write-LSN for this + // apply — the SAME shard-local WAL-LSN space fast-path writes and + // read watermarks use. Record the apply's per-key write versions + // at it once it exists; it does not exist yet at dispatch time. + Ok(applied_lsn) => { + self.record_calvin_write_versions(txn_id, applied_lsn); + Some(applied_lsn) + } + Err(e) => { + self.halt_apply( + txn_id, + HaltReason::WalAppendFailed, + HaltStep::AppliedMarker, + format!("CalvinApplied WAL append failed: {e}"), + ); + None + } + }, + }; + let Some(lsn) = applied_lsn else { + // The apply cannot be acknowledged without a durable participant + // LSN: CDC and write-version consumers would otherwise observe a + // successful commit with no authoritative ordering point. The + // scheduler halted above. + return false; + }; + // The record at `lsn` is this position's only applied marker. An + // append only buffers it, so it is durable before the mark and the + // ack: a restart that lost it would take the position for unapplied, + // run the transaction again, and never settle its ack. The wait joins + // the WAL group commit, so concurrent Calvin commits share one fsync. + if let Err(e) = self.shared.wal.wait_durable(lsn).await { + self.halt_apply( + txn_id, + HaltReason::WalAppendFailed, + HaltStep::AppliedMarker, + format!("applied marker fsync at lsn {} failed: {e}", lsn.as_u64()), + ); + return false; + } + // Control change-stream events are distinct from Data-Plane + // WriteEvents. Publish the participant-local logical manifests once, + // from the data-group leader, at the authoritative committed LSN. + if self.is_group_leader() + && let Some(pending) = self.pending.get_mut(&txn_id) + { + let tenant_id = pending.txn.tx_class.tenant_id; + let database_id = pending.txn.tx_class.database_id; + for change_set in std::mem::take(&mut pending.change_sets) { + crate::control::server::dispatch_utils::publish_change_set_with_lsn( + &self.shared, + tenant_id, + database_id, + change_set, + lsn, + ); + } + } + // The commit's mark lands before the ack, as a write through the + // funnel records its mark before its response returns. + self.record_calvin_write_mark(txn_id); + self.propose_sequencer_entry(txn_id, SchedulerProposal::CompletionAck); + true + } +} + +/// The response the statement drains. A flush whose install succeeded but +/// whose reply failed to render answers `Ok` with the render error in +/// `error_code`. The statement reports that error. +fn statement_reply(mut response: Response) -> Response { + if response.status == Status::Ok && response.error_code.is_some() { + response.status = Status::Error; + response.payload = crate::bridge::envelope::Payload::empty(); + } + response +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + error_response, scheduler_with_pending, + }; + use crate::control::cluster::calvin::scheduler::driver::types::CommitState; + + fn internal_error() -> Response { + error_response(ErrorCode::Internal { + detail: "core failed".to_string(), + }) + } + + /// A flush that returns an error under a COMMIT verdict holds the txn + /// unapplied and halts: a second flush would apply nothing. + #[tokio::test] + async fn flush_error_response_holds_committed_txn_unapplied() { + let txn_id = TxnId::new(9, 2); + let (mut scheduler, _dir) = scheduler_with_pending( + txn_id, + CommitState::AwaitingResolve { + committed: true, + redo_lsn: None, + }, + ); + + scheduler + .finish_resolved_commit(txn_id, internal_error(), true, None) + .await; + + assert!(!scheduler.applied.is_applied(9, 2)); + assert!(scheduler.pending.contains_key(&txn_id)); + assert_eq!( + scheduler.apply_halt().map(|h| h.reason), + Some(HaltReason::FlushFailed) + ); + assert!(scheduler.shared.sequencer_halt.apply_halt().is_halted()); + } + + /// A drop that returns an error under an abort verdict still completes + /// the txn: no replica writes an aborted txn. + #[tokio::test] + async fn drop_error_response_under_abort_completes_txn() { + let txn_id = TxnId::new(9, 2); + let (mut scheduler, _dir) = scheduler_with_pending( + txn_id, + CommitState::AwaitingResolve { + committed: false, + redo_lsn: None, + }, + ); + + scheduler + .finish_resolved_commit(txn_id, internal_error(), false, None) + .await; + + assert!(scheduler.applied.is_applied(9, 2)); + assert!(!scheduler.pending.contains_key(&txn_id)); + assert!(!scheduler.is_apply_halted()); + } + fn retryable_refusal() -> Response { + error_response(ErrorCode::RetryableRefusal { + reason: "install rolled back".to_string(), + }) + } + + /// A flush refused as retryable reaches the Data Plane again with the + /// same redo record, and the txn stays pending and unhalted. + #[tokio::test] + async fn retryable_flush_refusal_resends_the_same_flush() { + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + await_data_plane_request, build_test_scheduler_with_data_side, make_sequenced_txn, + staged_pending, + }; + use nodedb_physical::physical_plan::PhysicalPlan; + use nodedb_physical::physical_plan::meta::MetaOp; + + let txn_id = TxnId::new(9, 2); + let registry = nodedb_cluster::calvin::CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = build_test_scheduler_with_data_side(7, registry); + let mut pending = staged_pending(make_sequenced_txn(9, 2), txn_id); + pending.commit_state = Some(CommitState::AwaitingResolve { + committed: true, + redo_lsn: None, + }); + pending.flush_scope.redo = vec![7, 7, 7]; + pending.flush_scope.sends = 1; + scheduler.pending.insert(txn_id, pending); + + scheduler + .finish_resolved_commit(txn_id, retryable_refusal(), true, None) + .await; + + assert!(!scheduler.is_apply_halted()); + assert!(!scheduler.applied.is_applied(9, 2)); + assert_eq!( + scheduler.pending.get(&txn_id).map(|p| p.flush_scope.sends), + Some(2) + ); + assert!( + await_data_plane_request(&mut data_side, |plan| matches!( + plan, + PhysicalPlan::Meta(MetaOp::CalvinFlush { epoch: 9, position: 2, redo, .. }) + if redo == &vec![7, 7, 7] + )) + .await, + "the resent flush carries the same redo record" + ); + } + + /// A retryable refusal past the send bound halts like any flush error. + #[tokio::test] + async fn retryable_flush_refusal_past_the_bound_halts() { + let txn_id = TxnId::new(9, 2); + let (mut scheduler, _dir) = scheduler_with_pending( + txn_id, + CommitState::AwaitingResolve { + committed: true, + redo_lsn: None, + }, + ); + if let Some(pending) = scheduler.pending.get_mut(&txn_id) { + pending.flush_scope.sends = MAX_FLUSH_SENDS; + } + + scheduler + .finish_resolved_commit(txn_id, retryable_refusal(), true, None) + .await; + + assert!(scheduler.pending.contains_key(&txn_id)); + assert_eq!( + scheduler.apply_halt().map(|h| h.reason), + Some(HaltReason::FlushFailed) + ); + } + + /// A render error on an installed flush reaches the statement as a typed + /// error, and the txn completes. + #[test] + fn a_render_error_on_an_installed_flush_becomes_the_statement_error() { + let mut response = error_response(ErrorCode::Internal { + detail: "render".to_string(), + }); + response.status = Status::Ok; + + let reply = statement_reply(response); + + assert_eq!(reply.status, Status::Error); + assert!(matches!( + reply.error_code.as_deref(), + Some(ErrorCode::Internal { .. }) + )); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/mod.rs new file mode 100644 index 000000000..be4a38649 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/mod.rs @@ -0,0 +1,17 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Verdict-driven commit resolution for staged static Calvin transactions. +//! +//! A static Calvin dispatch STAGES its transaction on the Data Plane (validate +//! the read-set + buffer the plans, no base mutation). Its executor response +//! carries the local commit vote on `read_set_valid`. This module drives the +//! final step: dispatch a flush (commit, after `commit_redo` has WAL-appended +//! the resolved `TransactionRedo`) or drop (abort) of the staged buffer, wait +//! for its response, then run the commit tail (deposit applied result, record +//! write versions — plus a `CalvinApplied` WAL fallback when no redo record +//! was appended — propose `CompletionAck`) for a flush, or ack-only for a +//! drop. + +mod apply_tail; +mod verdict; +mod vote; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs new file mode 100644 index 000000000..e396e1261 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs @@ -0,0 +1,558 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Resume-on-verdict, verdict-signal handling, and the stall re-probe sweep +//! for a staged Calvin transaction parked on the cross-shard commit barrier. + +use std::sync::atomic::Ordering; +use std::time::Instant; + +use nodedb_cluster::calvin::VerdictSignal; + +use crate::control::cluster::calvin::scheduler::driver::core::commit_resolution_dispatch::CommitResolution; +use crate::control::cluster::calvin::scheduler::driver::core::deferred::{ + DispatchOutcome, DispatchStep, +}; +use crate::control::cluster::calvin::scheduler::driver::core::halt::{HaltReason, HaltStep}; +use crate::control::cluster::calvin::scheduler::driver::core::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::driver::types::CommitState; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; + +impl Scheduler { + /// Resume a txn parked in [`CommitState::AwaitingVerdict`] once the durable + /// GLOBAL verdict is known: dispatch its flush (commit) or drop (abort). + /// + /// `committed` is the authoritative cross-shard verdict — NOT this shard's + /// local vote. On commit, dispatches `MetaOp::CalvinResolve` and moves the + /// txn to [`CommitState::AwaitingRedoResolve`] (the resolved redo is + /// WAL-appended and the flush dispatched from [`Self::finish_redo_resolve`]). + /// On abort, dispatches the drop directly and moves the txn to + /// [`CommitState::AwaitingResolve`]. Bumps the flushed / dropped counter. The + /// commit tail runs later in [`Self::finish_resolved_commit`], once the + /// flush/drop response arrives. + /// + /// Double-resume guard: the verdict push and the probe-on-park (and the + /// stall re-probe sweep) can all fire for one txn, so this first confirms the + /// txn is still `Some(AwaitingVerdict)` — if it already transitioned out + /// (resolve/drop dispatched, or completed), this is a no-op. This guarantees + /// the flush/drop is dispatched exactly once. + /// + /// A COMMIT verdict for a txn this replica failed to stage halts the + /// scheduler: the txn stays parked with its locks, its stall deadline + /// cleared, and its position unapplied. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn resume_on_verdict( + &mut self, + txn_id: TxnId, + committed: bool, + ) { + // Guard: only a still-parked txn resumes. Mirrors `handle_completion`'s + // state-match so a duplicate push/probe/timeout is idempotent. + if !matches!( + self.pending.get(&txn_id).and_then(|p| p.commit_state), + Some(CommitState::AwaitingVerdict) + ) { + return; + } + + if committed + && let Some(pending) = self.pending.get_mut(&txn_id) + && let Some(stage_error) = pending.stage_error.clone() + { + pending.verdict_deadline = None; + self.halt_apply( + txn_id, + HaltReason::LocalStageFailed, + HaltStep::Stage, + format!("COMMIT verdict for a txn this replica did not stage: {stage_error}"), + ); + return; + } + + let (outcome, step) = if committed { + // Resolve the staged post-images into a replayable `RedoRecord` + // first; the redo is WAL-appended (in `finish_redo_resolve`) before + // the flush is dispatched, restoring restart durability for this + // vShard's slice of a multi-shard Calvin commit. + (self.dispatch_calvin_resolve(txn_id), DispatchStep::Resolve) + } else { + ( + self.dispatch_commit_resolution(txn_id, CommitResolution::Drop), + DispatchStep::Drop, + ) + }; + if let DispatchOutcome::Failed(error) = outcome { + // Terminal resolve/drop refusal: the scheduler halts and holds + // the txn parked with its locks and staged buffer. The cleared + // deadline keeps the stall sweep from re-sending it. + if let Some(pending) = self.pending.get_mut(&txn_id) { + pending.verdict_deadline = None; + } + self.fail_dispatch_step(txn_id, step, error); + return; + } + + // Sent or parked for re-send at capacity: either way the txn awaits + // this step's response, so a duplicate verdict push or probe is a + // no-op under the guard above. + if let Some(pending) = self.pending.get_mut(&txn_id) { + pending.commit_state = Some(if committed { + CommitState::AwaitingRedoResolve + } else { + CommitState::AwaitingResolve { + committed: false, + redo_lsn: None, + } + }); + // No longer parked: clear the stall deadline. + pending.verdict_deadline = None; + } + + if committed { + self.shared + .calvin + .counters + .commits_flushed + .fetch_add(1, Ordering::Relaxed); + } else { + self.shared + .calvin + .counters + .commits_dropped + .fetch_add(1, Ordering::Relaxed); + } + } + + /// Handle a pushed [`VerdictSignal`] from this node's completion registry. + /// + /// Matches the signal to the parked txn by `(epoch, position)` and resumes + /// it. A signal for a txn this scheduler does not host, or one that already + /// resumed, is a harmless no-op (the double-resume guard covers the latter). + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn handle_verdict_signal( + &mut self, + signal: VerdictSignal, + ) { + let txn_id = TxnId::new(signal.epoch, signal.position); + self.resume_on_verdict(txn_id, signal.verdict.is_commit()); + } + + /// Sweep parked `AwaitingVerdict` txns whose stall deadline has passed. + /// + /// For each stalled txn, RE-PROBE the durable verdict: if it is now known, + /// resume (a push we dropped on a full channel, or a verdict that landed + /// after the last probe). If it is STILL unknown, KEEP WAITING — hold locks, + /// emit a stall metric + warning, and re-arm the deadline so the warning is + /// rate-limited rather than per-iteration. It NEVER releases locks and NEVER + /// unilaterally aborts: a participant cannot know whether a peer already + /// flushed a COMMIT, so aborting one side while a peer committed would tear + /// the transaction. The verdict is guaranteed to arrive eventually — a + /// post-failover leader re-aggregates the replicated votes (seeded on every + /// replica) into the same verdict — so waiting is always the safe action. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn check_awaiting_verdict_stalls( + &mut self, + ) { + // no-determinism: stall-detection clock drives warnings/metrics only; this path holds locks and never aborts, so it cannot affect the replicated outcome. + let now = Instant::now(); + let stalled: Vec = self + .pending + .iter() + .filter(|(_, p)| matches!(p.commit_state, Some(CommitState::AwaitingVerdict))) + .filter(|(_, p)| p.verdict_deadline.is_some_and(|d| now >= d)) + .map(|(id, _)| *id) + .collect(); + + for txn_id in stalled { + if let Some(verdict) = self.registry.verdict(nodedb_cluster::calvin::TxnId::new( + txn_id.epoch, + txn_id.position, + )) { + self.resume_on_verdict(txn_id, verdict); + continue; + } + + // Verdict still unknown: keep waiting, hold locks, never abort. + self.metrics.record_verdict_stall(); + tracing::warn!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + "calvin: staged txn still awaiting the cross-shard verdict past its stall \ + deadline; HOLDING locks and waiting (never aborting — a peer may have already \ + flushed a commit). The verdict is guaranteed to arrive." + ); + if let Some(pending) = self.pending.get_mut(&txn_id) { + pending.verdict_deadline = Some(now + self.config.verdict_stall_warn()); + } + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use nodedb_cluster::calvin::{ + AbortReason, CalvinCompletionRegistry, ParticipantVote, VerdictOutcome, + }; + use nodedb_physical::physical_plan::PhysicalPlan; + use nodedb_physical::physical_plan::meta::MetaOp; + use nodedb_types::TenantId; + + use super::*; + use crate::bridge::dispatch::CoreChannelDataSide; + use crate::bridge::envelope::ErrorCode; + use crate::bridge::envelope::{Payload, Status}; + use crate::control::cluster::calvin::scheduler::driver::core::halt::HaltReason; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + await_data_plane_request, build_test_scheduler_with_data_side, error_response, + fill_tenant_inflight, make_sequenced_txn, release_filler, spawn_scheduler_loop, + staged_pending, staged_response, + }; + use crate::control::state::SharedState; + use crate::types::RequestId; + use crate::wal::RedoRecord; + + /// A scheduler with one txn parked in a commit state while its tenant sits + /// at the dispatcher's in-flight cap. + struct ParkedAtCapacity { + scheduler: Scheduler, + _dir: tempfile::TempDir, + data_side: CoreChannelDataSide, + shared: Arc, + fillers: Vec, + } + + /// Park `txn_id` in `state`, then fill its tenant to the in-flight cap. + fn parked_at_capacity(txn_id: TxnId, state: CommitState) -> ParkedAtCapacity { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, dir, mut data_side) = build_test_scheduler_with_data_side(7, registry); + let mut pending = staged_pending(make_sequenced_txn(txn_id.epoch, txn_id.position), txn_id); + pending.commit_state = Some(state); + scheduler.pending.insert(txn_id, pending); + let shared = Arc::clone(&scheduler.shared); + let fillers = fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + ParkedAtCapacity { + scheduler, + _dir: dir, + data_side, + shared, + fillers, + } + } + + /// An Ok resolve response whose redo record carries no ops, so the flush + /// dispatch follows at once with no WAL append. + fn empty_redo_response() -> crate::bridge::envelope::Response { + let redo = RedoRecord { + version: 1, + ops: Vec::new(), + calvin_stamp: None, + }; + let mut response = staged_response(Status::Ok, None); + response.payload = Payload::from_vec(redo.to_bytes().expect("encode empty redo record")); + response + } + + /// Under a COMMIT verdict, a refused resolve dispatch does not complete + /// the txn: its position stays unapplied and its pending entry stays. + #[tokio::test] + async fn commit_verdict_resolve_refused_at_capacity_does_not_complete_txn() { + let txn_id = TxnId::new(14, 2); + let mut parked = parked_at_capacity(txn_id, CommitState::AwaitingVerdict); + let scheduler = &mut parked.scheduler; + + scheduler.resume_on_verdict(txn_id, true); + + assert!( + !scheduler.applied.is_applied(14, 2), + "a refused resolve must not mark the position applied" + ); + assert!( + scheduler.pending.contains_key(&txn_id), + "a refused resolve must keep the txn's pending entry" + ); + } + + /// A refused flush dispatch after the redo resolves does not complete the + /// txn: its position stays unapplied and its pending entry stays. + #[tokio::test] + async fn resolved_redo_flush_refused_at_capacity_does_not_complete_txn() { + let txn_id = TxnId::new(14, 2); + let mut parked = parked_at_capacity(txn_id, CommitState::AwaitingRedoResolve); + let scheduler = &mut parked.scheduler; + + scheduler.finish_redo_resolve(txn_id, empty_redo_response()); + + assert!( + !scheduler.applied.is_applied(14, 2), + "a refused flush must not mark the position applied" + ); + assert!( + scheduler.pending.contains_key(&txn_id), + "a refused flush must keep the txn's pending entry" + ); + } + + /// Under an ABORT verdict, a refused drop dispatch does not complete the + /// txn: its position stays unapplied and its pending entry stays. + #[tokio::test] + async fn abort_verdict_drop_refused_at_capacity_does_not_complete_txn() { + let txn_id = TxnId::new(14, 2); + let mut parked = parked_at_capacity(txn_id, CommitState::AwaitingVerdict); + let scheduler = &mut parked.scheduler; + + scheduler.resume_on_verdict(txn_id, false); + + assert!( + !scheduler.applied.is_applied(14, 2), + "a refused drop must not mark the position applied" + ); + assert!( + scheduler.pending.contains_key(&txn_id), + "a refused drop must keep the txn's pending entry" + ); + } + + /// Once a Data Plane response frees tenant capacity, a refused resolve + /// reaches the Data Plane. + #[tokio::test] + async fn refused_resolve_reaches_data_plane_after_capacity_frees() { + let txn_id = TxnId::new(14, 2); + let ParkedAtCapacity { + mut scheduler, + _dir, + mut data_side, + shared, + fillers, + } = parked_at_capacity(txn_id, CommitState::AwaitingVerdict); + + scheduler.resume_on_verdict(txn_id, true); + let running = spawn_scheduler_loop(scheduler); + release_filler(&shared, &mut data_side, fillers[0]); + + let arrived = await_data_plane_request(&mut data_side, |plan| { + matches!( + plan, + PhysicalPlan::Meta(MetaOp::CalvinResolve { + epoch: 14, + position: 2 + }) + ) + }) + .await; + running.stop().await; + + assert!( + arrived, + "the refused resolve must reach the Data Plane once capacity frees" + ); + } + + /// Once a Data Plane response frees tenant capacity, a refused flush + /// reaches the Data Plane. + #[tokio::test] + async fn refused_flush_reaches_data_plane_after_capacity_frees() { + let txn_id = TxnId::new(14, 2); + let ParkedAtCapacity { + mut scheduler, + _dir, + mut data_side, + shared, + fillers, + } = parked_at_capacity(txn_id, CommitState::AwaitingRedoResolve); + + scheduler.finish_redo_resolve(txn_id, empty_redo_response()); + let running = spawn_scheduler_loop(scheduler); + release_filler(&shared, &mut data_side, fillers[0]); + + let arrived = await_data_plane_request(&mut data_side, |plan| { + matches!( + plan, + PhysicalPlan::Meta(MetaOp::CalvinFlush { + epoch: 14, + position: 2, + .. + }) + ) + }) + .await; + running.stop().await; + + assert!( + arrived, + "the refused flush must reach the Data Plane once capacity frees" + ); + } + + /// A false vote from either participant makes the only global verdict abort; + /// applying that durable verdict broadcasts the abort to every parked local + /// participant. The scheduler's `resume_on_verdict(false)` then dispatches a + /// drop, never a resolve/flush, on each recipient. + #[tokio::test] + async fn two_participant_false_vote_broadcasts_global_abort_to_every_scheduler() { + let registry = CalvinCompletionRegistry::new_detached(); + let txn = nodedb_cluster::calvin::TxnId::new(14, 2); + let txn_id = TxnId::new(14, 2); + let (mut first_scheduler, _first_dir, mut first_data) = + build_test_scheduler_with_data_side(7, Arc::clone(®istry)); + let (mut second_scheduler, _second_dir, mut second_data) = + build_test_scheduler_with_data_side(9, Arc::clone(®istry)); + first_scheduler + .pending + .insert(txn_id, staged_pending(make_sequenced_txn(14, 2), txn_id)); + second_scheduler + .pending + .insert(txn_id, staged_pending(make_sequenced_txn(14, 2), txn_id)); + + // Local staging votes only park their own staged slices; neither the + // affirmative nor the failed participant may resolve or drop unilaterally. + first_scheduler.resolve_staged_commit(txn_id, &staged_response(Status::Ok, Some(true))); + second_scheduler.resolve_staged_commit(txn_id, &staged_response(Status::Error, None)); + for (scheduler, data_side) in [ + (&first_scheduler, &mut first_data), + (&second_scheduler, &mut second_data), + ] { + assert!(matches!( + scheduler + .pending + .get(&txn_id) + .and_then(|pending| pending.commit_state), + Some(CommitState::AwaitingVerdict) + )); + assert!(data_side.request_rx.try_pop().is_err()); + } + + // Model the replicated vote entries and their resulting durable verdict. + // The shared registry sends each scheduler's actual registered channel. + registry.seed_expected(txn, 2); + registry.note_vote(txn, 7, ParticipantVote::Commit); + assert!(registry.drain_unproposed_verdicts().is_empty()); + registry.note_vote( + txn, + 9, + ParticipantVote::Abort(Some(AbortReason::SerializationConflict)), + ); + assert_eq!( + registry.drain_unproposed_verdicts(), + vec![( + txn, + VerdictOutcome::Abort(Some(AbortReason::SerializationConflict)) + )] + ); + registry.note_verdict( + txn, + VerdictOutcome::Abort(Some(AbortReason::SerializationConflict)), + ); + assert_eq!(registry.verdict(txn), Some(false)); + + let first_signal = first_scheduler + .verdict_rx + .try_recv() + .expect("registry must signal the first registered scheduler"); + let second_signal = second_scheduler + .verdict_rx + .try_recv() + .expect("registry must signal the second registered scheduler"); + first_scheduler.handle_verdict_signal(first_signal); + second_scheduler.handle_verdict_signal(second_signal); + + for (scheduler, data_side) in [ + (&first_scheduler, &mut first_data), + (&second_scheduler, &mut second_data), + ] { + assert!(matches!( + scheduler + .pending + .get(&txn_id) + .and_then(|pending| pending.commit_state), + Some(CommitState::AwaitingResolve { + committed: false, + redo_lsn: None + }) + )); + let request = data_side + .request_rx + .try_pop() + .expect("global abort must dispatch a drop to every participant"); + assert!(matches!( + request.inner.plan, + PhysicalPlan::Meta(MetaOp::CalvinDrop { + epoch: 14, + position: 2 + }) + )); + assert!(data_side.request_rx.try_pop().is_err()); + } + } + + /// A follower scheduler whose stage failed, parked on the verdict barrier. + fn follower_with_failed_stage( + txn_id: TxnId, + ) -> (Scheduler, tempfile::TempDir, CoreChannelDataSide) { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, dir, data_side) = build_test_scheduler_with_data_side(7, registry); + assert!( + !scheduler.is_group_leader(), + "the fixture hosts no data group" + ); + scheduler.pending.insert( + txn_id, + staged_pending(make_sequenced_txn(txn_id.epoch, txn_id.position), txn_id), + ); + scheduler.resolve_staged_commit( + txn_id, + &error_response(ErrorCode::Internal { + detail: "stage failed".to_string(), + }), + ); + (scheduler, dir, data_side) + } + + /// A COMMIT verdict for a txn this follower failed to stage halts the + /// scheduler: the leader staged and voted commit, so this replica cannot + /// apply it. No resolve is dispatched, and the stall sweep stops. + #[tokio::test] + async fn commit_verdict_after_local_stage_error_halts_unapplied() { + let txn_id = TxnId::new(14, 2); + let (mut scheduler, _dir, mut data_side) = follower_with_failed_stage(txn_id); + + scheduler.resume_on_verdict(txn_id, true); + + assert!(!scheduler.applied.is_applied(14, 2)); + let pending = scheduler + .pending + .get(&txn_id) + .expect("the txn stays pending"); + assert_eq!(pending.commit_state, Some(CommitState::AwaitingVerdict)); + assert_eq!(pending.verdict_deadline, None); + assert_eq!( + scheduler.apply_halt().map(|h| h.reason), + Some(HaltReason::LocalStageFailed) + ); + assert!( + data_side.request_rx.try_pop().is_err(), + "no resolve reaches the Data Plane" + ); + } + + /// An abort verdict for a txn this replica failed to stage drops it as + /// usual: every replica reaches the same abort. + #[tokio::test] + async fn abort_verdict_after_local_stage_error_drops_without_halting() { + let txn_id = TxnId::new(14, 2); + let (mut scheduler, _dir, mut data_side) = follower_with_failed_stage(txn_id); + + scheduler.resume_on_verdict(txn_id, false); + + assert!(!scheduler.is_apply_halted()); + let request = data_side + .request_rx + .try_pop() + .expect("the abort dispatches a drop"); + assert!(matches!( + request.inner.plan, + PhysicalPlan::Meta(MetaOp::CalvinDrop { + epoch: 14, + position: 2 + }) + )); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs new file mode 100644 index 000000000..be5738a4e --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs @@ -0,0 +1,110 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Local commit-vote casting for a staged static Calvin transaction. + +use std::sync::atomic::Ordering; +use std::time::Instant; + +use crate::bridge::envelope::Response; +use crate::control::cluster::calvin::scheduler::driver::core::halt::error_response_text; +use crate::control::cluster::calvin::scheduler::driver::core::owed::SchedulerProposal; +use crate::control::cluster::calvin::scheduler::driver::core::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::driver::core::staged_vote::{ + StagedVote, staged_commit_vote, +}; +use crate::control::cluster::calvin::scheduler::driver::types::CommitState; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; + +impl Scheduler { + /// Cast this participant's local commit vote for a staged transaction, then + /// PARK it on the cross-shard commit barrier awaiting the durable GLOBAL + /// verdict — it does NOT self-decide flush-or-drop on its local vote. + /// + /// The staged executor response is validate-only: its `read_set_valid` is + /// this shard's local commit vote (`Some(true)` => commit, `Some(false)` => + /// abort; a `None` from the active/dependent path is treated as commit). The + /// leader proposes that vote via the sequencer Raft group; the sequencer + /// aggregates all participants' votes into a single authoritative + /// `SequencerEntry::Verdict`, applied on every replica. + /// + /// This method moves the txn to [`CommitState::AwaitingVerdict`] WITHOUT + /// dispatching a resolve or drop, then immediately probes + /// `registry.verdict(txn)`: if the verdict is already durable (replay, or a + /// push we raced) it resumes at once via [`Self::resume_on_verdict`]; + /// otherwise it stays parked, holding locks and its staged buffer, until the + /// verdict push, a later probe, or the stall re-probe sweep delivers the + /// verdict. Resuming (in `resume_on_verdict`) is where the flush/drop is + /// dispatched and the flushed/dropped counters bump — using the GLOBAL + /// verdict, never the local vote. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn resolve_staged_commit( + &mut self, + txn_id: TxnId, + staged_response: &Response, + ) { + // A staged error is always an abort vote. Only successful staged + // responses may use `None` for the dependent-read path; accepting an + // error-plus-None as commit would let a failed participant flush after + // its peers received a global commit verdict. + let vote = staged_commit_vote(staged_response); + + // Durably propose this participant's commit vote via the sequencer + // Raft group, leader-guarded like `OllpMismatch`: only the data-group + // leader ran read-set validation, so only a leader's vote is + // authoritative. The sequencer aggregates every participant's vote into + // the single global verdict this txn parks on below. An abort travels as + // `AbortVote` so its cause survives to the coordinator. The vote stays + // owed until the tally holds it, so a refused or dropped proposal is + // proposed again rather than lost. + if self.is_group_leader() { + self.propose_sequencer_entry( + txn_id, + SchedulerProposal::Vote { + abort: vote.abort_reason(), + }, + ); + } + + if vote == StagedVote::SerializationConflict { + // The staged slice's read-set was no longer current: observe it, the + // same node-global signal the direct-apply path records. A + // participant error never validated a read-set, so it must not count + // here. + self.shared + .calvin + .counters + .read_set_validation_failures + .fetch_add(1, Ordering::Relaxed); + } + + // PARK on the barrier: transition to `AwaitingVerdict` and arm the stall + // deadline. Do NOT dispatch resolve/drop here — the GLOBAL verdict, not + // this local vote, decides. If the txn already vanished (torn down + // elsewhere), there is nothing to park. + // + // A stage error parks too. A deterministic error fails on every + // replica, the leader votes abort, and every replica drops. A local + // error on a follower leaves the leader's commit vote standing: + // `resume_on_verdict` halts on that COMMIT verdict. + match self.pending.get_mut(&txn_id) { + Some(pending) => { + pending.commit_state = Some(CommitState::AwaitingVerdict); + pending.stage_error = (vote == StagedVote::ParticipantError) + .then(|| error_response_text("stage", staged_response)); + // no-determinism: local stall-warning deadline only; the global replicated verdict, not this wall-clock, decides commit/abort. + pending.verdict_deadline = Some(Instant::now() + self.config.verdict_stall_warn()); + } + None => return, + } + + // PROBE on park (correctness backstop): the verdict may already be + // durable — on replay, or a push that raced ahead of this park. Resume + // immediately if so; the double-resume guard in `resume_on_verdict` + // makes a later duplicate push/probe a no-op. + if let Some(verdict) = self.registry.verdict(nodedb_cluster::calvin::TxnId::new( + txn_id.epoch, + txn_id.position, + )) { + self.resume_on_verdict(txn_id, verdict); + } + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs index eb0d188d9..02f9eb0fb 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs @@ -10,9 +10,9 @@ //! [`CommitState::AwaitingVerdict`] has no outstanding bridge, so a completion //! for it is a no-op that keeps it parked. -use nodedb_cluster::calvin::SequencerEntry; - use super::super::types::CommitState; +use super::halt::{HaltReason, HaltStep}; +use super::owed::SchedulerProposal; use super::scheduler::Scheduler; use crate::bridge::envelope::Response; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; @@ -22,7 +22,7 @@ impl Scheduler { /// Process a completed executor response (or disconnected channel). /// /// Called from the `completion_rx` arm of the main `select!` loop. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn handle_completion( + pub(in crate::control::cluster::calvin::scheduler::driver::core) async fn handle_completion( &mut self, txn_id: TxnId, request_id: RequestId, @@ -31,20 +31,20 @@ impl Scheduler { let response = match resp_opt { Some(r) => r, None => { - // Bridge task observed a closed channel before any response. - tracing::warn!( - vshard_id = self.vshard_id, - request_id = request_id.as_u64(), - epoch = txn_id.epoch, - position = txn_id.position, - "calvin: executor response channel disconnected" - ); + // The bridge task saw the channel close before any response, + // so the request's outcome on this replica is unknown. Hold + // the txn unapplied and halt. self.metrics.record_executor_error(); - self.metrics.record_infra_abort( - crate::control::cluster::calvin::scheduler::metrics::infra_abort_reason::IO_ERROR, + let state = self.pending.get(&txn_id).and_then(|p| p.commit_state); + self.halt_apply( + txn_id, + HaltReason::ResponseDisconnected, + HaltStep::awaited_by(state), + format!( + "executor response channel for request {} closed before a response", + request_id.as_u64() + ), ); - self.metrics.record_completed(); - self.on_txn_complete(txn_id); return; } }; @@ -56,13 +56,20 @@ impl Scheduler { .unwrap_or(0); self.metrics.record_executor_txn_duration_ms(elapsed_ms); + // A staged transaction resolves through its commit-barrier state, OLLP + // answer included. The flush installs a resolved redo record and runs + // no OLLP check, so an OLLP answer under a commit state is a failed + // step that halts and holds. It never settles as a retry. + let commit_state = self.pending.get(&txn_id).and_then(|p| p.commit_state); + // OLLP mismatch: the active executor detected predicate drift and returned // OllpRetryRequired without writing. The retry loop is now COORDINATOR-owned // (`run_dependent_with_retry`): the scheduler must NOT re-submit a stale // prediction. Instead it (1) releases the aborted attempt's locks and // (2) signals the coordinator's completion waiter via the registry so it // can run a FRESH reconnaissance and resubmit. - if response.status == crate::bridge::envelope::Status::Error + if commit_state.is_none() + && response.status == crate::bridge::envelope::Status::Error && response.error_code.as_deref() == Some(&crate::bridge::envelope::ErrorCode::OllpRetryRequired) { @@ -101,14 +108,7 @@ impl Scheduler { // OllpMismatch calls note_ollp_mismatch, waking the coordinator's // retry-loop waiter wherever it is — mirrors how CompletionAck is // delivered to remote coordinators. - self.propose_sequencer_entry( - SequencerEntry::OllpMismatch { - epoch: txn_id.epoch, - position: txn_id.position, - }, - txn_id, - "OLLP mismatch signal", - ); + self.propose_sequencer_entry(txn_id, SchedulerProposal::OllpMismatch); return; } @@ -116,7 +116,6 @@ impl Scheduler { // vote; drive the vote-and-park and let the verdict resume the // flush-or-drop. Dependent / active txns carry no `commit_state` and apply // directly below. - let commit_state = self.pending.get(&txn_id).and_then(|p| p.commit_state); match commit_state { Some(CommitState::Staged) => { // Stage failures remain participants in the barrier: @@ -134,7 +133,8 @@ impl Scheduler { committed, redo_lsn, }) => { - self.finish_resolved_commit(txn_id, response, committed, redo_lsn); + self.finish_resolved_commit(txn_id, response, committed, redo_lsn) + .await; return; } Some(CommitState::AwaitingVerdict) => { @@ -157,20 +157,11 @@ impl Scheduler { None => {} } - let completed = if response.status == crate::bridge::envelope::Status::Ok { - // Observe whether the applying participant reported its slice of the - // transaction's reads as no longer current against the local write - // versions. Direct-apply (dependent/active) observation only: the - // staged path folds this into its commit vote instead. `None` means - // no read-set was checked. - if response.read_set_valid == Some(false) { - self.shared - .calvin_counters - .read_set_validation_failures - .fetch_add(1, std::sync::atomic::Ordering::Relaxed); - } - self.commit_apply_tail(txn_id, response, None) - } else { + if response.status != crate::bridge::envelope::Status::Ok { + // A failed direct apply must not leave the txn parked with its locks + // held: that wedges every txn queued behind those keys and freezes + // this vShard's epoch watermark, and no sweep re-drives the entry. + // Surface the infra abort and force completion. tracing::error!( vshard_id = self.vshard_id, epoch = txn_id.epoch, @@ -178,25 +169,144 @@ impl Scheduler { "calvin: executor response was not Ok; forcing infra-abort completion so locks \ release and the epoch advances" ); - false - }; - - if completed { - self.metrics.record_completed(); - self.on_txn_complete(txn_id); - } else { - // A failed direct apply must not leave the txn parked with its locks - // held: that wedges every txn queued behind those keys and freezes - // this vShard's epoch watermark, and no sweep re-drives the entry. - // Surface the infra abort and force completion — the same - // forward-progress contract the disconnected-channel path above - // follows. self.metrics.record_executor_error(); self.metrics.record_infra_abort( crate::control::cluster::calvin::scheduler::metrics::infra_abort_reason::IO_ERROR, ); self.metrics.record_completed(); self.on_txn_complete(txn_id); + return; + } + + // Observe whether the applying participant reported its slice of the + // transaction's reads as no longer current against the local write + // versions. Direct-apply (dependent/active) observation only: the + // staged path folds this into its commit vote instead. `None` means + // no read-set was checked. + if response.read_set_valid == Some(false) { + self.shared + .calvin + .counters + .read_set_validation_failures + .fetch_add(1, std::sync::atomic::Ordering::Relaxed); + } + // `false` means the commit tail halted the scheduler: the txn stays + // pending and unapplied. + if self.commit_apply_tail(txn_id, response, None).await { + self.metrics.record_completed(); + self.on_txn_complete(txn_id); } } } + +#[cfg(test)] +mod tests { + use std::sync::atomic::Ordering; + + use super::*; + use crate::control::cluster::calvin::scheduler::driver::core::halt::HaltReason; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + error_response, make_sequenced_txn, scheduler_with_pending, staged_pending, + }; + use crate::control::cluster::calvin::scheduler::metrics::apply_halt_reason; + + /// A response channel that closes before a response leaves the txn's + /// outcome unknown: the txn stays pending and unapplied, the scheduler + /// halts, and the node marker names the txn. + #[tokio::test] + async fn disconnected_response_holds_txn_unapplied_and_sets_node_marker() { + let txn_id = TxnId::new(5, 1); + let (mut scheduler, _dir) = scheduler_with_pending( + txn_id, + CommitState::AwaitingResolve { + committed: true, + redo_lsn: None, + }, + ); + + scheduler + .handle_completion(txn_id, RequestId::new(9), None) + .await; + + assert!( + !scheduler.applied.is_applied(5, 1), + "an unknown outcome must not mark the position applied" + ); + assert!(scheduler.pending.contains_key(&txn_id)); + assert_eq!( + scheduler.apply_halt().map(|h| h.reason), + Some(HaltReason::ResponseDisconnected) + ); + let marker = scheduler.shared.sequencer_halt.apply_halt().report(); + assert_eq!( + marker.map(|h| (h.vshard_id, h.epoch, h.position, h.step)), + Some((7, 5, 1, "flush")) + ); + assert_eq!(scheduler.metrics.apply_halted.load(Ordering::Relaxed), 1); + assert_eq!( + scheduler.metrics.apply_halt_reason.load(Ordering::Relaxed), + apply_halt_reason::RESPONSE_DISCONNECTED as u64 + ); + } + + /// A second disconnect keeps the first cause and holds its txn too. + #[tokio::test] + async fn second_halt_keeps_the_first_cause() { + let first = TxnId::new(5, 1); + let second = TxnId::new(6, 0); + let (mut scheduler, _dir) = scheduler_with_pending(first, CommitState::Staged); + let mut pending = staged_pending(make_sequenced_txn(6, 0), second); + pending.commit_state = Some(CommitState::AwaitingRedoResolve); + scheduler.pending.insert(second, pending); + + scheduler + .handle_completion(first, RequestId::new(9), None) + .await; + scheduler + .handle_completion(second, RequestId::new(10), None) + .await; + + assert!(!scheduler.applied.is_applied(6, 0)); + assert!(scheduler.pending.contains_key(&second)); + let marker = scheduler.shared.sequencer_halt.apply_halt().report(); + assert_eq!( + marker.map(|h| (h.epoch, h.position, h.step)), + Some((5, 1, "stage")) + ); + } + + /// A committed flush that answers `OllpRetryRequired` wrote nothing on + /// this replica. The txn stays pending and unapplied, its redo records + /// stay open, and the scheduler halts on the flush. + #[tokio::test] + async fn an_ollp_answer_to_a_committed_flush_halts_and_holds() { + let txn_id = TxnId::new(5, 1); + let (mut scheduler, _dir) = scheduler_with_pending( + txn_id, + CommitState::AwaitingResolve { + committed: true, + redo_lsn: None, + }, + ); + + scheduler + .handle_completion( + txn_id, + RequestId::new(9), + Some(error_response( + crate::bridge::envelope::ErrorCode::OllpRetryRequired, + )), + ) + .await; + + assert!(!scheduler.applied.is_applied(5, 1)); + assert!( + scheduler.pending.contains_key(&txn_id), + "the txn must stay pending" + ); + assert_eq!( + scheduler.apply_halt().map(|h| h.reason), + Some(HaltReason::FlushFailed) + ); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/cut_marker.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/cut_marker.rs new file mode 100644 index 000000000..d93be00c0 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/cut_marker.rs @@ -0,0 +1,106 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A backup's cut marker on this vShard's scheduler, and the tenant write +//! mark a committed Calvin transaction records. +//! +//! The scheduler reports a marker to the node's cut registry once every +//! transaction delivered before it finished: installed by its flush, or +//! dropped under an abort verdict. The backup waiting on the marker then +//! snapshots a vShard that holds every one of them. Every transaction +//! delivered after the marker records a commit HLC above the marker's +//! watermark, so a restore of that backup refuses it. + +use super::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; +use crate::control::security::auth_fence::cluster::group_of_vshard; +use crate::control::state::tenant_marks::MarkSite; + +impl Scheduler { + /// Receive a backup's cut marker carrying `hlc`, in input order. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn receive_cut_marker( + &mut self, + hlc: u64, + ) { + if self + .cut_floors + .receive(hlc, self.applied.highest_seen_epoch()) + { + self.shared.calvin.cuts.note_passed(self.vshard_id, hlc); + } + self.report_passed_cuts(); + } + + /// Report every marker whose earlier transactions all finished, and fold + /// the markers every later epoch is above. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn report_passed_cuts( + &mut self, + ) { + let applied = &self.applied; + let passed = self + .cut_floors + .take_passed(|through| applied.is_fully_applied_through(through)); + for hlc in passed { + self.shared.calvin.cuts.note_passed(self.vshard_id, hlc); + } + self.cut_floors.fold(self.applied.fully_applied_epoch()); + } + + /// Record the commit HLC of the committed transaction `txn_id` on its + /// tenant's observed write high-water, before its `CompletionAck` lets + /// the coordinator acknowledge the COMMIT. + /// + /// Only a slice that carries a primary user data write records it. The + /// implicit edge cleanup that dual-homes alongside one writes no user row + /// of its own. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn record_calvin_write_mark( + &self, + txn_id: TxnId, + ) { + let Some(pending) = self.pending.get(&txn_id) else { + return; + }; + if !pending.has_primary_write { + return; + } + let tx_class = &pending.txn.tx_class; + let tenant_id = tx_class.tenant_id.as_u64(); + let collection = tx_class.write_set.0.first().map(|set| set.collection()); + let commit_hlc = self + .cut_floors + .commit_hlc(pending.txn.epoch, pending.txn.epoch_system_ms); + self.shared + .advance_tenant_write_hlc(tenant_id, commit_hlc, "calvin flush", collection); + + // The durable mark, in the data group that homes this vShard, lands + // before the ack too: RESTORE's guard reads it on every node, after a + // restart as well. + let group_id = match group_of_vshard(&self.shared, self.vshard_id) { + Ok(group_id) => group_id, + Err(error) => { + tracing::error!( + vshard_id = self.vshard_id, + %error, + "calvin: no data group for this vShard; the commit's write mark stays \ + in memory only" + ); + return; + } + }; + let marks = &self.shared.tenant_marks; + marks.raise( + group_id, + tenant_id, + commit_hlc, + MarkSite::CalvinFlush, + collection, + ); + if let Err(error) = marks.persist(self.shared.credentials.catalog()) { + // The mark stays pending; the next persist on this node writes it. + tracing::error!( + vshard_id = self.vshard_id, + %error, + "calvin: the commit's write mark did not persist yet" + ); + } + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs new file mode 100644 index 000000000..4173b8b7f --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs @@ -0,0 +1,391 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Capacity-safe Data-Plane dispatch for sequenced Calvin work. +//! +//! A sequenced txn has no refusal outcome: every replica must apply it. So +//! every scheduler dispatch goes through [`Scheduler::dispatch_sequenced`]. +//! A capacity refusal parks the request in a FIFO. The txn keeps its locks +//! and its `pending` entry, and never reaches `on_txn_complete`. The run loop +//! re-sends parked requests once a routed response frees capacity. A terminal +//! refusal halts the scheduler (see [`super::halt`]). + +use std::collections::VecDeque; +use std::sync::atomic::Ordering; +use std::time::Instant; + +use nodedb_physical::physical_plan::PhysicalPlan; +use nodedb_physical::physical_plan::meta::MetaOp; + +use super::halt::{HaltReason, HaltStep}; +use super::scheduler::Scheduler; +use crate::bridge::dispatch::DispatchRefusal; +use crate::bridge::envelope::Request; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; +use crate::types::RequestId; + +/// The Calvin sub-operation one scheduler dispatch carries. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) enum DispatchStep { + /// `CalvinExecuteStatic` stage of a static txn. + StageStatic, + /// `CalvinExecuteActive` stage of a dependent-read txn. + StageActive, + /// `CalvinResolve` of a committed staged txn. + Resolve, + /// `CalvinFlush` of a committed staged txn. + Flush, + /// `CalvinDrop` of an aborted staged txn. + Drop, + /// One-way `RecordCalvinWriteVersions` of a committed txn. + WriteVersionRecord, +} + +/// Result of [`Scheduler::dispatch_sequenced`]. +#[derive(Debug)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) enum DispatchOutcome { + /// The dispatcher accepted the request. + Sent, + /// The dispatcher refused at capacity. The request waits in the deferred + /// FIFO, and the txn stays in flight. + Deferred, + /// The dispatcher refused terminally. Nothing is parked. + Failed(crate::Error), +} + +/// A request refused at capacity, waiting to be re-sent. +pub(in crate::control::cluster::calvin::scheduler::driver::core) struct DeferredDispatch { + txn_id: TxnId, + step: DispatchStep, + request: Request, +} + +/// FIFO of requests refused at capacity, in refusal order. +pub(in crate::control::cluster::calvin::scheduler::driver::core) type DeferredQueue = + VecDeque; + +/// Result of one send attempt, before the caller decides where a refused +/// request goes in the FIFO. +enum Attempt { + Sent, + Capacity(Box), + Failed(crate::Error), +} + +impl Scheduler { + /// Register `request` in the tracker, dispatch it, and cancel the + /// registration on refusal. + /// + /// A capacity refusal appends the request to the deferred FIFO and + /// returns [`DispatchOutcome::Deferred`]. The caller keeps the txn in + /// flight. Any other refusal returns [`DispatchOutcome::Failed`], and the + /// caller runs its terminal handling. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn dispatch_sequenced( + &mut self, + txn_id: TxnId, + step: DispatchStep, + request: Request, + ) -> DispatchOutcome { + match self.send_once(txn_id, step, request) { + Attempt::Sent => DispatchOutcome::Sent, + Attempt::Capacity(parked) => { + self.deferred.push_back(*parked); + self.metrics + .set_dispatch_deferred_depth(self.deferred.len()); + DispatchOutcome::Deferred + } + Attempt::Failed(error) => DispatchOutcome::Failed(error), + } + } + + /// Whether any refused request waits for capacity. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn has_deferred_dispatch( + &self, + ) -> bool { + !self.deferred.is_empty() + } + + /// Whether the run loop re-sends parked requests: some wait, and the + /// scheduler has not halted. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn resends_deferred( + &self, + ) -> bool { + self.has_deferred_dispatch() && !self.is_apply_halted() + } + + /// Number of refused requests waiting for capacity. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn deferred_dispatch_len( + &self, + ) -> usize { + self.deferred.len() + } + + /// Re-send parked requests in FIFO order. + /// + /// Stops at the first capacity refusal, which goes back to the FIFO + /// front. A terminal refusal runs the step's terminal handling, and stops + /// the pass once the scheduler halted. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn redispatch_deferred( + &mut self, + ) { + while let Some(parked) = self.deferred.pop_front() { + let DeferredDispatch { + txn_id, + step, + mut request, + } = parked; + if step != DispatchStep::WriteVersionRecord && !self.pending.contains_key(&txn_id) { + tracing::error!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + ?step, + "calvin: parked dispatch for a txn no longer pending; one step per txn is broken, discarding" + ); + continue; + } + self.refresh_deferred_request(step, &mut request); + match self.send_once(txn_id, step, request) { + Attempt::Sent => {} + Attempt::Capacity(parked) => { + self.deferred.push_front(*parked); + break; + } + Attempt::Failed(error) => { + self.fail_dispatch_step(txn_id, step, error); + if self.is_apply_halted() { + break; + } + } + } + } + self.metrics + .set_dispatch_deferred_depth(self.deferred.len()); + } + + /// Run the terminal handling of a dispatch the dispatcher refused for a + /// reason other than capacity. + /// + /// A stage, resolve, flush, or drop refusal halts the scheduler: the txn + /// keeps its `pending` entry and locks, and its position stays unapplied. + /// A refusal during shutdown holds the txn the same way. A write-version + /// record is dropped with a warning, because the commit does not depend on + /// it. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn fail_dispatch_step( + &mut self, + txn_id: TxnId, + step: DispatchStep, + error: crate::Error, + ) { + let halt_step = match step { + DispatchStep::StageStatic | DispatchStep::StageActive => HaltStep::Stage, + DispatchStep::Resolve => HaltStep::Resolve, + DispatchStep::Flush => HaltStep::Flush, + DispatchStep::Drop => HaltStep::Drop, + DispatchStep::WriteVersionRecord => { + tracing::warn!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + %error, + "calvin: write-version record dispatch failed" + ); + return; + } + }; + self.halt_apply( + txn_id, + HaltReason::DispatchRefused, + halt_step, + error.to_string(), + ); + } + + /// One send attempt: register, dispatch, and cancel on refusal. + fn send_once(&mut self, txn_id: TxnId, step: DispatchStep, request: Request) -> Attempt { + // A crash test holds one collection's flush here: the redo record is + // appended and no core holds the flush. The flush waits in the + // re-send queue, as at capacity, so the scheduler keeps running. + #[cfg(feature = "failpoints")] + if step == DispatchStep::Flush && crate::control::fail_gate::holds_flush(&request.plan) { + tracing::info!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + "calvin: flush held at a fail point" + ); + return Attempt::Capacity(Box::new(DeferredDispatch { + txn_id, + step, + request, + })); + } + let request_id = request.request_id; + let resp_rx = self.shared.tracker.register(request_id); + let result = match self.shared.dispatcher.lock() { + Ok(mut dispatcher) => dispatcher.try_dispatch(request), + Err(poisoned) => poisoned.into_inner().try_dispatch(request), + }; + let refusal = match result { + Ok(()) => { + self.on_dispatch_sent(txn_id, step, request_id, resp_rx); + return Attempt::Sent; + } + Err(refusal) => refusal, + }; + self.shared.tracker.cancel(&request_id); + let DispatchRefusal { error, request } = *refusal; + match error { + crate::Error::DispatchCapacity { scope } => { + self.metrics.record_dispatch_deferred(); + tracing::debug!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + ?step, + %scope, + "calvin: dispatch refused at capacity; deferred until capacity frees" + ); + Attempt::Capacity(Box::new(DeferredDispatch { + txn_id, + step, + request, + })) + } + other => Attempt::Failed(other), + } + } + + /// Per-step bookkeeping once the dispatcher accepts a request. + fn on_dispatch_sent( + &mut self, + txn_id: TxnId, + step: DispatchStep, + request_id: RequestId, + resp_rx: crate::control::ResponseReceiver, + ) { + match step { + DispatchStep::StageStatic | DispatchStep::StageActive => { + self.metrics.record_dispatch(); + if let Some(pending) = self.pending.get_mut(&txn_id) { + // no-determinism: dispatch_time is executor-latency observability, off-WAL + pending.dispatch_time = Instant::now(); + } + self.spawn_response_bridge(txn_id, request_id, resp_rx); + } + DispatchStep::Resolve | DispatchStep::Flush | DispatchStep::Drop => { + self.spawn_response_bridge(txn_id, request_id, resp_rx); + } + DispatchStep::WriteVersionRecord => { + // One-way record: drain the response so it routes to a live + // receiver, then discard it. + tokio::spawn(async move { + let mut rx = resp_rx; + let _ = rx.recv().await; + }); + self.shared + .calvin + .counters + .write_versions_recorded + .fetch_add(1, Ordering::Relaxed); + } + } + } + + /// Refresh the dispatch-time fields of a parked request before a re-send. + /// + /// The deadline restarts from now. A stage request re-reads group + /// leadership, which the Data Plane uses to gate OLLP verification. + fn refresh_deferred_request(&self, step: DispatchStep, request: &mut Request) { + request.deadline = self.request_deadline(); + if !matches!(step, DispatchStep::StageStatic | DispatchStep::StageActive) { + return; + } + if let PhysicalPlan::Meta( + MetaOp::CalvinExecuteStatic { + is_group_leader, .. + } + | MetaOp::CalvinExecuteActive { + is_group_leader, .. + }, + ) = &mut request.plan + { + *is_group_leader = self.is_group_leader(); + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use nodedb_cluster::calvin::CalvinCompletionRegistry; + use nodedb_cluster::calvin::types::SchedulerInput; + use nodedb_types::TenantId; + + use super::*; + use crate::control::cluster::calvin::scheduler::driver::core::halt::HaltReason; + use crate::control::cluster::calvin::scheduler::driver::core::intake::IntakeClosure; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + begin_data_plane_drain, build_test_scheduler_with_data_side, fill_tenant_inflight, + make_validate_only_txn, test_coll_vshard, + }; + + /// A stage refused because the Data Plane drains stays pending and + /// unapplied, keeps its locks, and closes intake. Shutdown sets no node + /// marker. + #[tokio::test] + async fn stage_refused_while_draining_is_held_unapplied_without_node_marker() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, _data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + begin_data_plane_drain(&scheduler.shared); + let txn_id = TxnId::new(3, 0); + + scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(4, 0))); + + assert!( + !scheduler.applied.is_applied(3, 0), + "a terminal refusal must not mark the position applied" + ); + assert!( + scheduler.pending.contains_key(&txn_id), + "the refused txn keeps its pending entry" + ); + assert!( + scheduler.blocked.contains_key(&TxnId::new(4, 0)), + "the refused txn keeps its key locks" + ); + assert_eq!( + scheduler.apply_halt().map(|h| h.reason), + Some(HaltReason::Draining) + ); + assert_eq!(scheduler.intake_closure(), Some(IntakeClosure::ApplyHalted)); + assert!( + !scheduler.shared.sequencer_halt.apply_halt().is_halted(), + "a shutdown halt sets no node marker" + ); + } + + /// A parked stage whose re-send is refused terminally stays pending and + /// unapplied, and the scheduler stops re-sending. + #[tokio::test] + async fn parked_stage_refused_terminally_on_resend_is_held_unapplied() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let shared = Arc::clone(&scheduler.shared); + fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + let txn_id = TxnId::new(3, 0); + scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + assert!(scheduler.has_deferred_dispatch(), "the stage parks"); + + begin_data_plane_drain(&shared); + scheduler.redispatch_deferred(); + + assert!(!scheduler.applied.is_applied(3, 0)); + assert!(scheduler.pending.contains_key(&txn_id)); + assert!(scheduler.is_apply_halted()); + assert!(!scheduler.resends_deferred()); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs index 206d24982..2726df689 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs @@ -11,6 +11,7 @@ use nodedb_cluster::calvin::types::SequencedTxn; use nodedb_physical::physical_plan::PhysicalPlan; use nodedb_physical::physical_plan::meta::MetaOp; +use super::super::deferred::{DispatchOutcome, DispatchStep}; use super::super::scheduler::Scheduler; use super::primary_write::{ participant_change_sets, plans_have_primary_write, plans_have_returning, @@ -45,7 +46,7 @@ impl Scheduler { error = %e, "calvin scheduler: active plan decode failed; releasing locks" ); - self.on_txn_complete(txn_id); + self.on_unpending_txn_complete(txn_id, lock_owner); return; } }; @@ -74,8 +75,8 @@ impl Scheduler { error = %e, "calvin scheduler: active txn homes no local writes; releasing locks" ); - self.propose_routing_failure(epoch, position, txn_id, &e); - self.on_txn_complete(txn_id); + self.propose_routing_failure(txn_id, &e); + self.on_unpending_txn_complete(txn_id, lock_owner); return; } Err(e) => { @@ -86,8 +87,8 @@ impl Scheduler { error = %e, "calvin scheduler: active txn routing failed; releasing locks" ); - self.propose_routing_failure(epoch, position, txn_id, &e); - self.on_txn_complete(txn_id); + self.propose_routing_failure(txn_id, &e); + self.on_unpending_txn_complete(txn_id, lock_owner); return; } }; @@ -97,6 +98,7 @@ impl Scheduler { let has_primary_write = plans_have_primary_write(&plans, has_non_derived_write); let has_returning = plans_have_returning(&plans); let change_sets = participant_change_sets(&plans, tenant_id, self.vshard_id); + let flush_scope = super::super::super::types::FlushScope::of_plans(&plans); let plan = PhysicalPlan::Meta(MetaOp::CalvinExecuteActive { epoch, position, @@ -110,42 +112,18 @@ impl Scheduler { // Calvin allocates the CalvinApplied WAL LSN post-apply (in the // scheduler's response handler), so no committed LSN is known at // dispatch time to stamp here. - let request = - self.build_exempt_request(request_id, tenant_id, txn.tx_class.database_id, plan, None); - - let resp_rx = self.shared.tracker.register(request_id); - - let dispatch_result = match self.shared.dispatcher.lock() { - Ok(mut d) => d.dispatch(request), - Err(poisoned) => poisoned.into_inner().dispatch(request), - }; - - if let Err(e) = dispatch_result { - error!( - vshard_id = self.vshard_id, - epoch, - position, - error = %e, - "calvin scheduler: active dispatch failed; releasing locks" - ); - self.on_txn_complete(txn_id); - return; - } - - self.metrics.record_dispatch(); - - // no-determinism: executor latency observability, off-WAL path - let dispatch_instant = Instant::now(); - - self.spawn_response_bridge(txn_id, request_id, resp_rx); + let database_id = txn.tx_class.database_id; + let request = self.build_exempt_request(request_id, tenant_id, database_id, plan, None); + // The txn enters `pending` before the dispatch, so a stage refused at + // capacity stays in flight with its locks until the re-send. self.pending.insert( txn_id, super::super::super::types::PendingTxn { txn, lock_owner, // no-determinism: dispatch_time is scheduler observability, not Calvin WAL data - dispatch_time: dispatch_instant, + dispatch_time: Instant::now(), has_primary_write, has_returning, change_sets, @@ -157,7 +135,89 @@ impl Scheduler { commit_state: Some(super::super::super::types::CommitState::Staged), // Set only once the txn parks in `AwaitingVerdict`. verdict_deadline: None, + stage_error: None, + // Set once a committed txn appends its redo record. + redo_records: None, + flush_scope, }, ); + + if let DispatchOutcome::Failed(error) = + self.dispatch_sequenced(txn_id, DispatchStep::StageActive, request) + { + self.fail_dispatch_step(txn_id, DispatchStep::StageActive, error); + } + } +} + +#[cfg(test)] +mod tests { + use std::collections::BTreeMap; + use std::sync::Arc; + + use nodedb_cluster::calvin::CalvinCompletionRegistry; + use nodedb_types::TenantId; + + use super::*; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + await_data_plane_request, build_test_scheduler_with_data_side, fill_tenant_inflight, + make_local_write_txn, release_filler, spawn_scheduler_loop, test_coll_vshard, + }; + + /// A refused active dispatch keeps the txn in flight: its position stays + /// unapplied and its pending entry stays. + #[tokio::test] + async fn active_dispatch_refused_at_capacity_leaves_txn_unapplied() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let shared = Arc::clone(&scheduler.shared); + fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + let txn_id = TxnId::new(5, 0); + + scheduler.dispatch_active_txn(make_local_write_txn(5, 0), txn_id, txn_id, BTreeMap::new()); + + assert!( + !scheduler.applied.is_applied(5, 0), + "a refused active dispatch must not mark the position applied" + ); + assert!( + scheduler.pending.contains_key(&txn_id), + "a refused active dispatch must keep the txn's pending entry" + ); + } + + /// Once a Data Plane response frees tenant capacity, the refused active + /// request reaches the Data Plane. + #[tokio::test] + async fn active_dispatch_refused_at_capacity_is_retried_after_capacity_frees() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let shared = Arc::clone(&scheduler.shared); + let fillers = fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + let txn_id = TxnId::new(5, 0); + + scheduler.dispatch_active_txn(make_local_write_txn(5, 0), txn_id, txn_id, BTreeMap::new()); + let running = spawn_scheduler_loop(scheduler); + release_filler(&shared, &mut data_side, fillers[0]); + + let arrived = await_data_plane_request(&mut data_side, |plan| { + matches!( + plan, + PhysicalPlan::Meta(MetaOp::CalvinExecuteActive { + epoch: 5, + position: 0, + .. + }) + ) + }) + .await; + running.stop().await; + + assert!( + arrived, + "the refused active request must reach the Data Plane once capacity frees" + ); } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/bind_identities.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/bind_identities.rs index e8ffd9451..9481dc9f2 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/bind_identities.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/bind_identities.rs @@ -8,8 +8,8 @@ //! or a later point read by primary key resolves nothing on this node. use nodedb_physical::physical_plan::PhysicalPlan; -use tracing::error; +use super::super::halt::{HaltReason, HaltStep}; use super::super::scheduler::Scheduler; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; use crate::control::surrogate::bind_plan_identities; @@ -18,10 +18,12 @@ use crate::types::{DatabaseId, TenantId}; impl Scheduler { /// Bind every identity in `plans` first-wins and rewrite each surrogate /// slot with the authoritative value, the same walk the replicated-write - /// decoder runs. On a catalog error the txn is terminated as a routing - /// failure: applying rows nobody can resolve by key is worse than aborting. + /// decoder runs. /// - /// Returns `false` after terminating the txn; the caller returns at once. + /// A catalog error is local to this replica, and its peers apply the + /// slice. So the scheduler halts: the txn is not yet in `pending`, its + /// locks stay held under its lock owner, and its position stays + /// unapplied. Returns `false` after the halt; the caller returns at once. pub(super) fn bind_local_identities( &mut self, plans: &mut [PhysicalPlan], @@ -36,17 +38,12 @@ impl Scheduler { match bound { Ok(()) => true, Err(e) => { - let epoch = txn_id.epoch; - let position = txn_id.position; - error!( - vshard_id = self.vshard_id, - epoch, - position, - error = %e, - "calvin scheduler: surrogate binding failed; releasing locks" + self.halt_apply( + txn_id, + HaltReason::IdentityBindFailed, + HaltStep::IdentityBind, + format!("surrogate binding failed: {e}"), ); - self.propose_routing_failure(epoch, position, txn_id, &e); - self.on_txn_complete(txn_id); false } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs index e8443cf30..db07d33dd 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs @@ -11,6 +11,8 @@ use nodedb_cluster::calvin::types::SequencedTxn; use nodedb_physical::physical_plan::PhysicalPlan; use nodedb_physical::physical_plan::meta::MetaOp; +use super::super::deferred::{DispatchOutcome, DispatchStep}; +use super::super::owed::SchedulerProposal; use super::super::routing::PlanRouting; use super::super::scheduler::Scheduler; use super::primary_write::{ @@ -50,21 +52,12 @@ impl Scheduler { /// full deadline and report a generic timeout. Mirrors the OllpMismatch /// broadcast in `handle_executor_response`. Shared by `dispatch_txn` and /// `dispatch_active_txn`. - pub(super) fn propose_routing_failure( - &self, - epoch: u64, - position: u32, - txn_id: TxnId, - err: &crate::Error, - ) { + pub(super) fn propose_routing_failure(&mut self, txn_id: TxnId, err: &crate::Error) { self.propose_sequencer_entry( - nodedb_cluster::calvin::SequencerEntry::TxnRoutingFailed { - epoch, - position, + txn_id, + SchedulerProposal::RoutingFailed { detail: err.to_string(), }, - txn_id, - "txn routing-failure signal", ); } @@ -148,7 +141,7 @@ impl Scheduler { error = %e, "calvin scheduler: plan decode failed; releasing locks and skipping txn" ); - self.on_txn_complete(txn_id); + self.on_unpending_txn_complete(txn_id, lock_owner); return; } }; @@ -164,8 +157,8 @@ impl Scheduler { error = %e, "calvin scheduler: static txn routing failed; releasing locks" ); - self.propose_routing_failure(epoch, position, txn_id, &e); - self.on_txn_complete(txn_id); + self.propose_routing_failure(txn_id, &e); + self.on_unpending_txn_complete(txn_id, lock_owner); return; } }; @@ -199,8 +192,8 @@ impl Scheduler { error = %e, "calvin scheduler: static txn homes no local work; releasing locks" ); - self.propose_routing_failure(epoch, position, txn_id, &e); - self.on_txn_complete(txn_id); + self.propose_routing_failure(txn_id, &e); + self.on_unpending_txn_complete(txn_id, lock_owner); return; } @@ -219,8 +212,8 @@ impl Scheduler { ); } - /// Build and dispatch a `CalvinExecuteStatic` task, then park the txn in - /// `pending` as `Staged`. + /// Park the txn in `pending` as `Staged`, then build and dispatch its + /// `CalvinExecuteStatic` task. /// /// Shared by the write path (`plans` = this vShard's local write slice) and /// the validate-only read path (`plans` empty). Both carry the txn's FULL @@ -228,6 +221,9 @@ impl Scheduler { /// the read-set — whether or not `plans` is empty — and returns the commit /// vote on `read_set_valid`. A validate-only task has `has_primary_write == /// false`, so it deposits no result sidecar entry, exactly as intended. + /// + /// The txn enters `pending` before the dispatch, so a stage refused at + /// capacity stays in flight with its locks until the re-send. fn dispatch_calvin_static( &mut self, txn: SequencedTxn, @@ -246,6 +242,8 @@ impl Scheduler { let has_primary_write = plans_have_primary_write(&plans, has_non_derived_write); let has_returning = plans_have_returning(&plans); let change_sets = participant_change_sets(&plans, tenant_id, self.vshard_id); + let flush_scope = super::super::super::types::FlushScope::of_plans(&plans); + let database_id = txn.tx_class.database_id; let plan = PhysicalPlan::Meta(MetaOp::CalvinExecuteStatic { epoch, position, @@ -262,34 +260,7 @@ impl Scheduler { // Calvin allocates the CalvinApplied WAL LSN post-apply (in the // scheduler's response handler), so no committed LSN is known at // dispatch time to stamp here. - let request = - self.build_exempt_request(request_id, tenant_id, txn.tx_class.database_id, plan, None); - - let resp_rx = self.shared.tracker.register(request_id); - - let dispatch_result = match self.shared.dispatcher.lock() { - Ok(mut d) => d.dispatch(request), - Err(poisoned) => poisoned.into_inner().dispatch(request), - }; - - if let Err(e) = dispatch_result { - error!( - vshard_id = self.vshard_id, - epoch, - position, - error = %e, - "calvin scheduler: dispatch failed; releasing locks" - ); - self.on_txn_complete(txn_id); - return; - } - - self.metrics.record_dispatch(); - - // no-determinism: executor latency observability, off-WAL path - let dispatch_instant = Instant::now(); - - self.spawn_response_bridge(txn_id, request_id, resp_rx); + let request = self.build_exempt_request(request_id, tenant_id, database_id, plan, None); self.pending.insert( txn_id, @@ -297,17 +268,27 @@ impl Scheduler { txn, lock_owner, // no-determinism: dispatch_time is scheduler observability, not Calvin WAL data - dispatch_time: dispatch_instant, + dispatch_time: Instant::now(), has_primary_write, has_returning, change_sets, - // This dispatch STAGED the txn (validate + buffer, no apply); + // This dispatch STAGES the txn (validate + buffer, no apply); // its response carries the local commit vote that drives the // subsequent flush-or-drop. commit_state: Some(super::super::super::types::CommitState::Staged), // Set only once the txn parks in `AwaitingVerdict`. verdict_deadline: None, + stage_error: None, + // Set once a committed txn appends its redo record. + redo_records: None, + flush_scope, }, ); + + if let DispatchOutcome::Failed(error) = + self.dispatch_sequenced(txn_id, DispatchStep::StageStatic, request) + { + self.fail_dispatch_step(txn_id, DispatchStep::StageStatic, error); + } } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/halt.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/halt.rs new file mode 100644 index 000000000..aaeb8e559 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/halt.rs @@ -0,0 +1,286 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Apply halt for one Calvin scheduler. +//! +//! Every replica applies every sequenced txn. A position is marked applied +//! only after it applied on this replica, or after an abort that every +//! replica reaches identically. A replica-local infrastructure error is +//! neither: the txn's effect on this replica is unknown or missing. So the +//! scheduler halts and never marks that position applied. +//! +//! A halted scheduler: +//! - keeps the stuck txn unapplied, with its locks and its `pending` entry; +//! - closes intake with [`IntakeClosure::ApplyHalted`]. The fan-out then drops +//! new input and arms catch-up, which keeps the sequencer log retained; +//! - stops the deferred re-send and the catch-up drain; +//! - keeps routing responses, verdicts, and promotions for other in-flight +//! txns. Each one holds locks disjoint from the stuck txn, so its outcome +//! does not depend on it. Peer vShards wait on its votes, and the per-position +//! applied gate keeps restart replay exact. A later infrastructure error on +//! another txn holds that txn the same way. +//! +//! The first cause wins. It logs one `error!` line, sets the +//! `nodedb_calvin_apply_halted` gauge, and records the node-wide +//! [`CalvinApplyHaltMarker`] that `/healthz` and the native status report. +//! While the node shuts down (the Data Plane drains), the scheduler holds the +//! same way, logs at info, and sets no node marker: the process is exiting. +//! +//! [`IntakeClosure::ApplyHalted`]: super::intake::IntakeClosure::ApplyHalted +//! [`CalvinApplyHaltMarker`]: crate::control::cluster::CalvinApplyHaltMarker + +use tracing::{debug, error, info, warn}; + +use super::super::types::CommitState; +use super::scheduler::Scheduler; +use crate::bridge::envelope::Response; +use crate::control::cluster::CalvinApplyHalt; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; +use crate::control::cluster::calvin::scheduler::metrics::apply_halt_reason; + +/// Why a scheduler halted. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) enum HaltReason { + /// The node shuts down and its Data Plane drains. + Draining, + /// The dispatcher refused a request for a reason other than capacity. + DispatchRefused, + /// The executor response channel closed before a response arrived. + ResponseDisconnected, + /// The resolve of a committed txn failed or returned an undecodable record. + ResolveFailed, + /// The flush of a committed txn returned an error. + FlushFailed, + /// This replica failed to stage a txn whose verdict is COMMIT. + LocalStageFailed, + /// The surrogate catalog refused a coordinator-assigned identity. + IdentityBindFailed, + /// A redo or `CalvinApplied` WAL append failed. + WalAppendFailed, +} + +impl HaltReason { + /// The `nodedb_calvin_apply_halted` reason index. + fn metric_reason(self) -> usize { + match self { + Self::Draining => apply_halt_reason::DRAINING, + Self::DispatchRefused => apply_halt_reason::DISPATCH_REFUSED, + Self::ResponseDisconnected => apply_halt_reason::RESPONSE_DISCONNECTED, + Self::ResolveFailed => apply_halt_reason::RESOLVE_FAILED, + Self::FlushFailed => apply_halt_reason::FLUSH_FAILED, + Self::LocalStageFailed => apply_halt_reason::LOCAL_STAGE_FAILED, + Self::IdentityBindFailed => apply_halt_reason::IDENTITY_BIND_FAILED, + Self::WalAppendFailed => apply_halt_reason::WAL_APPEND_FAILED, + } + } + + /// The reason label on the gauge and the node marker. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn label( + self, + ) -> &'static str { + apply_halt_reason::LABELS[self.metric_reason()] + } +} + +/// The sub-operation of the stuck txn that failed. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) enum HaltStep { + /// The stage request, or the stage result under a COMMIT verdict. + Stage, + /// The resolve request of a committed txn. + Resolve, + /// The flush request of a committed txn. + Flush, + /// The drop request of an aborted txn. + Drop, + /// The direct apply of a txn with no commit state. + Apply, + /// The `TransactionRedo` WAL append. + RedoAppend, + /// The `CalvinApplied` WAL append. + AppliedMarker, + /// The surrogate identity binding before the stage dispatch. + IdentityBind, +} + +impl HaltStep { + /// The step label on the node marker and the log line. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn label( + self, + ) -> &'static str { + match self { + Self::Stage => "stage", + Self::Resolve => "resolve", + Self::Flush => "flush", + Self::Drop => "drop", + Self::Apply => "apply", + Self::RedoAppend => "redo_append", + Self::AppliedMarker => "applied_marker", + Self::IdentityBind => "identity_bind", + } + } + + /// The request a txn in `state` awaits a response for. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn awaited_by( + state: Option, + ) -> Self { + match state { + Some(CommitState::Staged | CommitState::AwaitingVerdict) => Self::Stage, + Some(CommitState::AwaitingRedoResolve) => Self::Resolve, + Some(CommitState::AwaitingResolve { + committed: true, .. + }) => Self::Flush, + Some(CommitState::AwaitingResolve { + committed: false, .. + }) => Self::Drop, + None => Self::Apply, + } + } +} + +/// The first halt of one scheduler. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) struct ApplyHalt { + pub reason: HaltReason, + /// The stuck txn, step, and error text, in the node marker's shape. + pub report: CalvinApplyHalt, +} + +/// First-cause-wins halt latch of one scheduler. Never clears. +#[derive(Debug, Default)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) struct HaltLatch { + first: Option, +} + +/// Error text for an executor response that was not `Ok`. +pub(in crate::control::cluster::calvin::scheduler::driver::core) fn error_response_text( + request: &str, + response: &Response, +) -> String { + format!( + "{request} returned {:?} with error code {:?}", + response.status, + response.error_code.as_deref() + ) +} + +impl Scheduler { + /// Whether this scheduler halted. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn is_apply_halted( + &self, + ) -> bool { + self.halt.first.is_some() + } + + /// The first halt, if this scheduler halted. + #[cfg(test)] + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn apply_halt( + &self, + ) -> Option<&ApplyHalt> { + self.halt.first.as_ref() + } + + /// Hold `txn_id` unapplied and halt this scheduler. + /// + /// The caller leaves the txn's `pending` entry and locks in place and + /// never calls `on_txn_complete` for it. A halt while the node shuts down + /// records [`HaltReason::Draining`] whatever `reason` says. A second halt + /// only logs: the first cause stays. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn halt_apply( + &mut self, + txn_id: TxnId, + reason: HaltReason, + step: HaltStep, + error: String, + ) { + // The txn stays unapplied here, so restart replay must reach its redo + // record. + self.hold_redo_records(txn_id); + let reason = if self.node_shutting_down() { + HaltReason::Draining + } else { + reason + }; + if let Some(first) = &self.halt.first { + if first.reason == HaltReason::Draining || reason == HaltReason::Draining { + debug!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + step = step.label(), + %error, + "calvin scheduler: shutting down; holding txn unapplied" + ); + } else { + warn!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + reason = reason.label(), + step = step.label(), + %error, + halted_epoch = first.report.epoch, + halted_position = first.report.position, + "calvin scheduler: apply halted; holding another txn unapplied" + ); + } + return; + } + + let report = CalvinApplyHalt { + vshard_id: self.vshard_id, + epoch: txn_id.epoch, + position: txn_id.position, + reason: reason.label(), + step: step.label(), + error, + }; + self.metrics.set_apply_halted(reason.metric_reason()); + if reason == HaltReason::Draining { + info!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + step = report.step, + error = %report.error, + "calvin scheduler: shutting down; holding txn unapplied and closing intake" + ); + } else { + error!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + reason = report.reason, + step = report.step, + error = %report.error, + "calvin scheduler: apply halted; txn held unapplied with its locks, \ + intake closed until restart" + ); + crate::diag::calvin_apply_halted( + report.vshard_id, + report.epoch, + report.position, + report.reason, + report.step, + &report.error, + ); + self.shared + .sequencer_halt + .apply_halt() + .record(report.clone()); + } + self.halt.first = Some(ApplyHalt { reason, report }); + } + + /// Whether the node shuts down: the shutdown watch fired, or the + /// dispatcher closed its Data Plane enqueue gate. + fn node_shutting_down(&self) -> bool { + if self.shared.shutdown.is_shutdown() { + return true; + } + self.shared + .dispatcher + .lock() + .unwrap_or_else(|p| p.into_inner()) + .is_data_plane_draining() + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs new file mode 100644 index 000000000..f87b16717 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs @@ -0,0 +1,332 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Intake gate for the Calvin scheduler. +//! +//! The scheduler takes new sequenced input only while it can make progress. +//! The gate closes for good once the scheduler halts (see [`super::halt`]). +//! It closes while a capacity-refused dispatch waits for re-send, or while +//! the in-flight backlog sits at [`SchedulerConfig::max_inflight_backlog`]. +//! A closed gate disables the run loop's receiver arm and skips the catch-up +//! drain. Completions, verdicts, read results, promotions, and capacity +//! wakeups stay active, because they drain the backlog. +//! +//! Input left unread fills the bounded fan-out channel. The sequencer apply +//! path then drops the next input and arms catch-up for this vShard, and the +//! drain replays it from the committed log once the gate opens. +//! +//! The gate changes only when inputs are read, never the order they are +//! processed in, so replicas stay deterministic. +//! +//! [`SchedulerConfig::max_inflight_backlog`]: super::super::config::SchedulerConfig::max_inflight_backlog + +use tracing::debug; + +use super::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::metrics::intake_closure_reason; + +/// Why the intake gate is closed. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) enum IntakeClosure { + /// The scheduler halted on a txn it cannot mark applied. + ApplyHalted, + /// A capacity-refused dispatch waits in the deferred FIFO. + DeferredDispatch, + /// The in-flight backlog is at its bound, and some of it progresses + /// without new input. + BacklogFull, +} + +impl IntakeClosure { + /// The `nodedb_calvin_intake_gate_closed_total` reason index. + fn metric_reason(self) -> usize { + match self { + Self::ApplyHalted => intake_closure_reason::APPLY_HALTED, + Self::DeferredDispatch => intake_closure_reason::DEFERRED_DISPATCH, + Self::BacklogFull => intake_closure_reason::BACKLOG_FULL, + } + } +} + +/// Last observed intake gate state, kept to detect transitions. +#[derive(Debug, Default)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) struct IntakeGate { + closed: Option, +} + +impl Scheduler { + /// Pending, blocked, and dependent-barrier txns. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn inflight_backlog( + &self, + ) -> usize { + self.pending.len() + self.blocked.len() + self.dependent_barrier.len() + } + + /// Why intake must stay closed, or `None` when the gate is open. + /// + /// The backlog bound applies only while a pending txn or a dependent + /// barrier exists. Those finish on executor responses, verdicts, and read + /// results. A backlog of blocked txns alone can wait on a reservation + /// release, which arrives as input, so it keeps the gate open. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn intake_closure( + &self, + ) -> Option { + if self.is_apply_halted() { + return Some(IntakeClosure::ApplyHalted); + } + if self.has_deferred_dispatch() { + return Some(IntakeClosure::DeferredDispatch); + } + let drains_without_input = !self.pending.is_empty() || !self.dependent_barrier.is_empty(); + if drains_without_input && self.inflight_backlog() >= self.config.max_inflight_backlog { + return Some(IntakeClosure::BacklogFull); + } + None + } + + /// Re-evaluate the intake gate and return whether it is open. + /// + /// Publishes the backlog gauge on every call. Logs each transition at + /// debug, and counts each open-to-closed transition by reason. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn refresh_intake_gate( + &mut self, + ) -> bool { + let backlog = self.inflight_backlog(); + let closure = self.intake_closure(); + self.metrics.set_intake_backlog(backlog); + let previous = self.intake.closed; + if closure == previous { + return closure.is_none(); + } + match closure { + Some(reason) => { + if previous.is_none() { + self.metrics + .record_intake_gate_closed(reason.metric_reason()); + } + debug!( + vshard_id = self.vshard_id, + ?reason, + backlog, + deferred = self.deferred_dispatch_len(), + "calvin scheduler: intake gate closed" + ); + } + None => { + debug!( + vshard_id = self.vshard_id, + backlog, "calvin scheduler: intake gate opened" + ); + } + } + self.metrics.set_intake_gate_closed(closure.is_some()); + self.intake.closed = closure; + closure.is_none() + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::collections::BTreeSet; + use std::sync::Arc; + use std::sync::atomic::Ordering; + use std::time::{Duration, Instant}; + + use nodedb_cluster::calvin::CalvinCompletionRegistry; + use nodedb_cluster::calvin::types::SchedulerInput; + use nodedb_types::TenantId; + use tokio::sync::mpsc; + + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + build_test_scheduler, build_test_scheduler_with_data_side, fill_tenant_inflight, + make_sequenced_txn, make_validate_only_txn, release_filler, spawn_scheduler_loop, + staged_pending, test_coll_vshard, + }; + use crate::control::cluster::calvin::scheduler::driver::types::BlockedTxn; + use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; + use crate::types::RequestId; + + /// How long a held input must stay unread. Several liveness ticks of the + /// spawned loop fit in it. + const HOLD_WAIT: Duration = Duration::from_millis(300); + + /// How long a test waits for the loop to read an input. + const CONSUME_WAIT: Duration = Duration::from_secs(5); + + /// Whether the loop reads every input sent on `input_tx` within `wait`. + async fn inputs_consumed_within( + input_tx: &mpsc::Sender, + wait: Duration, + ) -> bool { + let drained = async { + while input_tx.capacity() < input_tx.max_capacity() { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }; + tokio::time::timeout(wait, drained).await.is_ok() + } + + fn blocked_fixture(epoch: u64) -> BlockedTxn { + BlockedTxn { + txn: make_sequenced_txn(epoch, 0), + keys: BTreeSet::new(), + // no-determinism: test-only blocked_at timestamp for a fabricated BlockedTxn fixture. + blocked_at: Instant::now(), + } + } + + /// A backlog of blocked txns alone keeps intake open at the bound: a + /// reservation release they wait on arrives as input. + #[tokio::test] + async fn blocked_only_backlog_at_bound_keeps_intake_open() { + let (mut scheduler, _dir) = build_test_scheduler(0); + scheduler.config.max_inflight_backlog = 1; + scheduler + .blocked + .insert(TxnId::new(3, 0), blocked_fixture(3)); + + assert_eq!(scheduler.intake_closure(), None); + } + + /// A pending txn that fills the backlog bound closes intake. + #[tokio::test] + async fn pending_backlog_at_bound_closes_intake() { + let (mut scheduler, _dir) = build_test_scheduler(0); + scheduler.config.max_inflight_backlog = 2; + let held = TxnId::new(3, 0); + scheduler + .pending + .insert(held, staged_pending(make_sequenced_txn(3, 0), held)); + assert_eq!(scheduler.intake_closure(), None, "below the bound"); + + scheduler + .blocked + .insert(TxnId::new(4, 0), blocked_fixture(4)); + assert_eq!( + scheduler.intake_closure(), + Some(IntakeClosure::BacklogFull), + "at the bound with a pending txn" + ); + } + + /// With a dispatch deferred at capacity, the run loop leaves a new input + /// unread. It reads the input once capacity frees and the deferral drains. + #[tokio::test] + async fn deferred_dispatch_holds_new_input_until_capacity_frees() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let shared = Arc::clone(&scheduler.shared); + let fillers = fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + assert!( + scheduler.has_deferred_dispatch(), + "the stage dispatch defers" + ); + let metrics = Arc::clone(&scheduler.metrics); + + let running = spawn_scheduler_loop(scheduler); + running + .input_tx() + .send(SchedulerInput::Txn(make_validate_only_txn(4, 0))) + .await + .expect("the loop's input channel is open"); + + let read_while_deferred = inputs_consumed_within(running.input_tx(), HOLD_WAIT).await; + let gate_closed = metrics.intake_gate_closed.load(Ordering::Relaxed); + let deferred_closures = metrics.intake_gate_closed_counts + [intake_closure_reason::DEFERRED_DISPATCH] + .load(Ordering::Relaxed); + + release_filler(&shared, &mut data_side, fillers[0]); + let read_after_capacity = inputs_consumed_within(running.input_tx(), CONSUME_WAIT).await; + running.stop().await; + + assert!( + !read_while_deferred, + "the loop must not read input while a dispatch is deferred" + ); + assert_eq!(gate_closed, 1, "the gate gauge reports closed"); + assert_eq!(deferred_closures, 1, "one closure for a deferred dispatch"); + assert!( + read_after_capacity, + "the loop must read the input once the deferral drains" + ); + } + + /// With the backlog at the configured bound, the run loop leaves a new + /// input unread. + #[tokio::test] + async fn full_backlog_holds_new_input() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, _data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + scheduler.config.max_inflight_backlog = 1; + let held = TxnId::new(3, 0); + scheduler + .pending + .insert(held, staged_pending(make_validate_only_txn(3, 0), held)); + let metrics = Arc::clone(&scheduler.metrics); + + let running = spawn_scheduler_loop(scheduler); + running + .input_tx() + .send(SchedulerInput::Txn(make_validate_only_txn(4, 0))) + .await + .expect("the loop's input channel is open"); + + let read_at_bound = inputs_consumed_within(running.input_tx(), HOLD_WAIT).await; + running.stop().await; + + assert!( + !read_at_bound, + "the loop must not read input while the backlog is at its bound" + ); + assert_eq!(metrics.intake_gate_closed.load(Ordering::Relaxed), 1); + assert_eq!(metrics.intake_backlog.load(Ordering::Relaxed), 1); + assert_eq!( + metrics.intake_gate_closed_counts[intake_closure_reason::BACKLOG_FULL] + .load(Ordering::Relaxed), + 1 + ); + } + + /// A halted scheduler closes intake with `ApplyHalted`, and the run loop + /// leaves new input unread. + #[tokio::test] + async fn halted_scheduler_closes_intake_and_reads_no_input() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, _data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let held = TxnId::new(3, 0); + scheduler + .pending + .insert(held, staged_pending(make_validate_only_txn(3, 0), held)); + scheduler + .handle_completion(held, RequestId::new(9), None) + .await; + assert_eq!(scheduler.intake_closure(), Some(IntakeClosure::ApplyHalted)); + let metrics = Arc::clone(&scheduler.metrics); + + let running = spawn_scheduler_loop(scheduler); + running + .input_tx() + .send(SchedulerInput::Txn(make_validate_only_txn(4, 0))) + .await + .expect("the loop's input channel is open"); + + let read_while_halted = inputs_consumed_within(running.input_tx(), HOLD_WAIT).await; + running.stop().await; + + assert!( + !read_while_halted, + "the loop must not read input once the scheduler halted" + ); + assert_eq!(metrics.intake_gate_closed.load(Ordering::Relaxed), 1); + assert_eq!( + metrics.intake_gate_closed_counts[intake_closure_reason::APPLY_HALTED] + .load(Ordering::Relaxed), + 1 + ); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs index e0f25ee2b..8fd8be522 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs @@ -3,62 +3,43 @@ //! Calvin scheduler driver core. //! //! One [`Scheduler`] task runs per vshard hosted on this node. It receives -//! [`SequencedTxn`]s from the sequencer, acquires deterministic locks, -//! dispatches static / dependent-read transactions to the Data Plane, -//! waits for executor responses, and writes `CalvinApplied` WAL records. -//! -//! Sub-modules (one concern per file): -//! -//! - [`scheduler`] — `Scheduler` struct, ctor, run loop. -//! - [`completion_route`] — routes each executor response (disconnect, OLLP -//! mismatch, staged commit-resolution state, or direct apply) to its handler. -//! - [`process`] — new-txn processing, dependent-read barrier setup, -//! txn-completion bookkeeping. -//! - [`catch_up`] — sequencer-fan-out catch-up drain: replays inputs dropped on -//! this replica (channel Full/Closed) from the committed sequencer Raft log. -//! - [`dispatch`] — static / active dispatch to the Data Plane executor. -//! - [`routing`] — exhaustive `PhysicalPlan` → vshard routing oracle used by -//! `dispatch`'s local-plan filtering. -//! - [`commit_resolve`] — verdict-driven flush-or-drop of a staged static -//! transaction, plus the shared commit tail. -//! - [`staged_vote`] — derives a participant's local commit vote from its -//! staged executor response, keeping the two abort causes apart. -//! - [`commit_redo`] — resolves a committed staged transaction's post-images -//! into a replayable `TransactionRedo` WAL record ahead of the flush. -//! - [`read_result`] — `CalvinReadResult` handling and barrier timeouts. -//! - [`propose`] — propose `CalvinReadResult` Raft entries. -//! - [`request`] — shared `Request` construction for already-sequenced Calvin -//! sub-operations. -//! - [`write_version_record`] — post-apply write-version recording for -//! committed Calvin transactions (at the CalvinApplied WAL LSN). -//! -//! # Determinism -//! -//! All bookkeeping uses `BTreeMap`/`BTreeSet` — never `HashMap`/`HashSet`. -//! Dispatch order is `(epoch, position)` order. -//! -//! # Timing / `Instant::now()` -//! -//! `Instant::now()` is used for: -//! - Lock-wait latency metrics (observability only). -//! - Dependent-read barrier `timeout_at` (off-WAL path only). -//! -//! Never used for WAL-influencing values. +//! sequenced txns from the sequencer, acquires deterministic locks, dispatches +//! static / dependent-read transactions to the Data Plane, waits for executor +//! responses, and writes `CalvinApplied` WAL records. Each sub-module owns one +//! concern; see that file's own doc comment for what it does. +//! +//! All bookkeeping uses `BTreeMap`/`BTreeSet` — never `HashMap`/`HashSet` — +//! and dispatch order is `(epoch, position)` order. `Instant::now()` is used +//! only for lock-wait latency metrics and the dependent-read barrier +//! `timeout_at`, both off the WAL-influencing path; every call site carries a +//! `// no-determinism:` marker. pub mod catch_up; pub mod commit_redo; pub mod commit_resolution_dispatch; pub mod commit_resolve; pub mod completion_route; +mod cut_marker; +pub mod deferred; pub mod dispatch; +pub mod halt; +pub mod intake; +pub mod owed; pub mod process; pub mod propose; pub mod read_result; +mod redo_window; pub mod request; pub mod routing; pub mod scheduler; +pub mod sequencer_proposer; pub mod staged_vote; +#[cfg(test)] +mod test_proposer; +#[cfg(test)] +mod test_support; pub mod write_version_record; pub use propose::{CalvinReadResultProposal, propose_calvin_read_result}; pub use scheduler::{Scheduler, SchedulerParams}; +pub use sequencer_proposer::{RaftSequencerProposer, SequencerProposer}; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/entry.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/entry.rs new file mode 100644 index 000000000..31453d838 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/entry.rs @@ -0,0 +1,243 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Owed sequencer entries: what the scheduler proposes, and how it tells +//! that an entry is applied. +//! +//! A proposal can be lost: a node that does not lead the sequencer group +//! refuses it, and a leader change can drop an appended entry. A lost vote +//! leaves every participant waiting for a verdict forever. So the scheduler +//! keeps each proposal in [`OwedEntries`] and proposes it again on the stall +//! tick until this node's completion registry shows its effect. + +use std::collections::BTreeMap; + +use nodedb_cluster::calvin::{AbortReason, ParticipantProgress, SequencerEntry}; + +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; +use crate::control::cluster::calvin::scheduler::metrics::sequencer_propose_kind; + +/// A sequencer entry the scheduler proposes for one txn on its vShard. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum SchedulerProposal { + /// This vShard's commit vote. `Some(reason)` is an abort vote. + Vote { abort: Option }, + /// This vShard applied or dropped the txn. + CompletionAck, + /// The active executor saw its OLLP prediction drift. + OllpMismatch, + /// The txn's local plan routing failed for good. + RoutingFailed { detail: String }, +} + +impl SchedulerProposal { + /// The owed-entry kind this proposal fills. + pub fn kind(&self) -> OwedKind { + match self { + Self::Vote { .. } => OwedKind::Vote, + Self::CompletionAck => OwedKind::CompletionAck, + Self::OllpMismatch => OwedKind::OllpMismatch, + Self::RoutingFailed { .. } => OwedKind::RoutingFailed, + } + } + + /// The sequencer entry for `txn_id` proposed by `vshard`. + pub fn entry(&self, txn_id: TxnId, vshard: u32) -> SequencerEntry { + let (epoch, position) = (txn_id.epoch, txn_id.position); + match self { + Self::Vote { abort: None } => SequencerEntry::Vote { + epoch, + position, + vshard, + commit: true, + }, + Self::Vote { + abort: Some(reason), + } => SequencerEntry::AbortVote { + epoch, + position, + vshard, + reason: *reason, + }, + Self::CompletionAck => SequencerEntry::CompletionAck { + epoch, + position, + vshard_id: vshard, + }, + Self::OllpMismatch => SequencerEntry::OllpMismatch { epoch, position }, + Self::RoutingFailed { detail } => SequencerEntry::TxnRoutingFailed { + epoch, + position, + detail: detail.clone(), + }, + } + } +} + +/// The kind of an owed entry. A txn owes at most one entry of each kind. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +pub enum OwedKind { + Vote, + CompletionAck, + OllpMismatch, + RoutingFailed, +} + +impl OwedKind { + /// Short name for logs. + pub fn label(self) -> &'static str { + sequencer_propose_kind::LABELS[self.metric_index()] + } + + /// Index into the `sequencer_propose_kind` metric labels. + pub fn metric_index(self) -> usize { + match self { + Self::Vote => sequencer_propose_kind::VOTE, + Self::CompletionAck => sequencer_propose_kind::COMPLETION_ACK, + Self::OllpMismatch => sequencer_propose_kind::OLLP_MISMATCH, + Self::RoutingFailed => sequencer_propose_kind::ROUTING_FAILED, + } + } + + /// Whether this node applied the entry of this kind. + /// + /// `progress` is the registry's view of the txn for the scheduler's + /// vShard. `entry_seen` is whether the registry held an entry for the + /// txn at an earlier check. The registry removes an entry only once the + /// txn's outcome fired, and a txn with a fired outcome needs no entry of + /// any kind. A missing entry that was never seen means this node has not + /// seeded the txn, so the entry is still owed. + /// + /// A stored verdict also settles a vote: the verdict forms only once + /// every participant's vote is in the tally. + pub fn is_applied(self, progress: Option, entry_seen: bool) -> bool { + let Some(p) = progress else { + return entry_seen; + }; + match self { + Self::Vote => p.voted || p.has_verdict, + Self::CompletionAck => p.acked, + Self::OllpMismatch => p.mismatched, + Self::RoutingFailed => p.routing_failed, + } + } +} + +/// One owed sequencer entry. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct OwedEntry { + /// The msgpack-encoded `SequencerEntry`. + pub bytes: Vec, + /// The registry held an entry for the txn at some check. + pub entry_seen: bool, + /// The last proposal left this node after the previous sweep. The next + /// sweep skips the entry once, which gives it one tick to apply. + pub in_flight: bool, +} + +/// Owed entries keyed by txn and kind, in deterministic order. +/// +/// Bounded: it holds at most one entry per (txn, kind), and a txn owes at +/// most two kinds (a vote and a completion ack, or a single terminal +/// signal). An entry leaves once this node applies it. Entries pile up only +/// while no sequencer leader is reachable, and then the sequencer admits no +/// new txns either. +pub type OwedEntries = BTreeMap<(TxnId, OwedKind), OwedEntry>; + +#[cfg(test)] +mod tests { + use super::*; + + fn progress() -> ParticipantProgress { + ParticipantProgress { + voted: false, + acked: false, + has_verdict: false, + mismatched: false, + routing_failed: false, + } + } + + const ALL_KINDS: [OwedKind; 4] = [ + OwedKind::Vote, + OwedKind::CompletionAck, + OwedKind::OllpMismatch, + OwedKind::RoutingFailed, + ]; + + #[test] + fn nothing_is_applied_while_the_registry_shows_no_effect() { + for kind in ALL_KINDS { + assert!(!kind.is_applied(Some(progress()), true), "{kind:?}"); + } + } + + #[test] + fn each_kind_is_applied_once_its_own_effect_shows() { + let voted = ParticipantProgress { + voted: true, + ..progress() + }; + let acked = ParticipantProgress { + acked: true, + ..progress() + }; + let mismatched = ParticipantProgress { + mismatched: true, + ..progress() + }; + let routing_failed = ParticipantProgress { + routing_failed: true, + ..progress() + }; + assert!(OwedKind::Vote.is_applied(Some(voted), false)); + assert!(!OwedKind::CompletionAck.is_applied(Some(voted), false)); + assert!(OwedKind::CompletionAck.is_applied(Some(acked), false)); + assert!(!OwedKind::Vote.is_applied(Some(acked), false)); + assert!(OwedKind::OllpMismatch.is_applied(Some(mismatched), false)); + assert!(OwedKind::RoutingFailed.is_applied(Some(routing_failed), false)); + } + + #[test] + fn a_stored_verdict_settles_the_vote_only() { + let decided = ParticipantProgress { + has_verdict: true, + ..progress() + }; + assert!(OwedKind::Vote.is_applied(Some(decided), false)); + assert!(!OwedKind::CompletionAck.is_applied(Some(decided), false)); + } + + #[test] + fn a_removed_entry_settles_every_kind_only_after_it_was_seen() { + for kind in ALL_KINDS { + assert!(kind.is_applied(None, true), "{kind:?}"); + assert!(!kind.is_applied(None, false), "{kind:?}"); + } + } + + #[test] + fn a_vote_proposal_carries_its_abort_reason() { + let txn = TxnId::new(3, 4); + assert!(matches!( + SchedulerProposal::Vote { abort: None }.entry(txn, 9), + SequencerEntry::Vote { + epoch: 3, + position: 4, + vshard: 9, + commit: true + } + )); + assert!(matches!( + SchedulerProposal::Vote { + abort: Some(AbortReason::SerializationConflict) + } + .entry(txn, 9), + SequencerEntry::AbortVote { + epoch: 3, + position: 4, + vshard: 9, + reason: AbortReason::SerializationConflict + } + )); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/mod.rs new file mode 100644 index 000000000..60e264f97 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/mod.rs @@ -0,0 +1,8 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Sequencer entries the scheduler owes until this node applies them. + +pub mod entry; +pub mod retry; + +pub use entry::{OwedEntries, OwedEntry, OwedKind, SchedulerProposal}; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/retry.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/retry.rs new file mode 100644 index 000000000..cf31a06c3 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/retry.rs @@ -0,0 +1,335 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Owing, proposing, and re-proposing the scheduler's sequencer entries. +//! +//! Re-proposing is safe because every entry kind applies idempotently in the +//! completion registry. A vote is stored per vShard, so a repeat overwrites +//! it with the same value, and the verdict it completes is emitted once. An +//! ack is a set insert per vShard, and the completion fires once. The +//! mismatch and routing-failure signals set a flag, and each fires its +//! waiter once. + +use tracing::{debug, error}; + +use nodedb_cluster::calvin::ParticipantProgress; + +use super::entry::{OwedEntry, OwedKind, SchedulerProposal}; +use crate::control::cluster::calvin::scheduler::driver::core::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::driver::core::sequencer_proposer::SequencerProposer; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; + +impl Scheduler { + /// Owe `proposal` for `txn_id` and propose it once. + /// + /// The entry stays owed until this node's completion registry shows it + /// applied. [`Self::retry_owed_sequencer_entries`] proposes it again on + /// the stall tick until then. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn propose_sequencer_entry( + &mut self, + txn_id: TxnId, + proposal: SchedulerProposal, + ) { + let kind = proposal.kind(); + let bytes = match zerompk::to_msgpack_vec(&proposal.entry(txn_id, self.vshard_id)) { + Ok(bytes) => bytes, + Err(e) => { + error!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + kind = kind.label(), + error = %e, + "calvin: failed to encode a sequencer entry; it cannot be proposed", + ); + return; + } + }; + let entry_seen = self.participant_progress(txn_id).is_some(); + let in_flight = propose_owed( + self.sequencer_proposer.as_ref(), + self.vshard_id, + txn_id, + kind, + bytes.clone(), + ); + self.owed.insert( + (txn_id, kind), + OwedEntry { + bytes, + entry_seen, + in_flight, + }, + ); + } + + /// Drop every owed entry this node applied, and propose the rest again. + /// + /// Runs on the stall tick. An entry proposed since the previous sweep is + /// skipped once, so an entry on its way through Raft is not proposed a + /// second time before it can apply. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn retry_owed_sequencer_entries( + &mut self, + ) { + let registry = &self.registry; + let vshard_id = self.vshard_id; + self.owed.retain(|(txn_id, kind), owed| { + let progress = registry.participant_progress(cluster_txn_id(*txn_id), vshard_id); + owed.entry_seen |= progress.is_some(); + !kind.is_applied(progress, owed.entry_seen) + }); + + for ((txn_id, kind), owed) in self.owed.iter_mut() { + if owed.in_flight { + owed.in_flight = false; + continue; + } + self.metrics + .record_sequencer_propose_retry(kind.metric_index()); + owed.in_flight = propose_owed( + self.sequencer_proposer.as_ref(), + vshard_id, + *txn_id, + *kind, + owed.bytes.clone(), + ); + } + } + + /// The registry's view of `txn_id` for this scheduler's vShard. + fn participant_progress(&self, txn_id: TxnId) -> Option { + self.registry + .participant_progress(cluster_txn_id(txn_id), self.vshard_id) + } +} + +/// The completion registry's key for `txn_id`. +fn cluster_txn_id(txn_id: TxnId) -> nodedb_cluster::calvin::TxnId { + nodedb_cluster::calvin::TxnId::new(txn_id.epoch, txn_id.position) +} + +/// Propose `bytes`. Returns `true` when the entry left this node. +fn propose_owed( + proposer: &dyn SequencerProposer, + vshard_id: u32, + txn_id: TxnId, + kind: OwedKind, + bytes: Vec, +) -> bool { + match proposer.propose(bytes) { + Ok(_) => true, + Err(e) => { + debug!( + vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + kind = kind.label(), + error = %e, + "calvin: sequencer entry not proposed; the stall tick proposes it again", + ); + false + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::atomic::Ordering; + use std::time::Duration; + + use nodedb_cluster::calvin::{ParticipantVote, SequencerEntry}; + + use super::*; + use crate::bridge::envelope::Status; + use crate::control::cluster::calvin::scheduler::driver::core::test_proposer::{ + CapturingProposer, elect_data_group_leader, + }; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + build_test_scheduler, make_sequenced_txn, scheduler_with_pending, spawn_scheduler_loop, + staged_pending, staged_response, + }; + use crate::control::cluster::calvin::scheduler::driver::types::CommitState; + use crate::control::cluster::calvin::scheduler::metrics::sequencer_propose_kind; + + const VSHARD: u32 = 7; + + fn retries(scheduler: &Scheduler, kind: usize) -> u64 { + scheduler.metrics.sequencer_propose_retry_counts[kind].load(Ordering::Relaxed) + } + + /// The leader's vote survives two refused proposals, then stops being + /// proposed once the tally holds it. + #[tokio::test] + async fn staged_leader_vote_is_reproposed_until_the_tally_shows_it() { + let (mut scheduler, _dir) = build_test_scheduler(VSHARD); + let proposer = CapturingProposer::failing_first(2); + scheduler.sequencer_proposer = proposer.clone(); + elect_data_group_leader(&scheduler); + let txn_id = TxnId::new(20, 1); + scheduler.registry.seed_expected(cluster_txn_id(txn_id), 2); + scheduler + .pending + .insert(txn_id, staged_pending(make_sequenced_txn(20, 1), txn_id)); + + scheduler.resolve_staged_commit(txn_id, &staged_response(Status::Ok, Some(true))); + assert_eq!(proposer.attempt_count(), 1, "the first proposal is refused"); + + scheduler.retry_owed_sequencer_entries(); + assert_eq!(proposer.attempt_count(), 2, "the second is refused too"); + assert!(proposer.accepted().is_empty()); + + scheduler.retry_owed_sequencer_entries(); + assert_eq!( + proposer.accepted(), + vec![SequencerEntry::Vote { + epoch: 20, + position: 1, + vshard: VSHARD, + commit: true, + }] + ); + + scheduler.retry_owed_sequencer_entries(); + assert_eq!( + proposer.attempt_count(), + 3, + "an accepted entry gets one tick to apply" + ); + scheduler.retry_owed_sequencer_entries(); + assert_eq!( + proposer.attempt_count(), + 4, + "an accepted entry that did not apply is proposed again" + ); + + scheduler + .registry + .note_vote(cluster_txn_id(txn_id), VSHARD, ParticipantVote::Commit); + scheduler.retry_owed_sequencer_entries(); + scheduler.retry_owed_sequencer_entries(); + assert_eq!( + proposer.attempt_count(), + 4, + "an applied vote is not proposed" + ); + assert!(scheduler.owed.is_empty()); + assert_eq!(retries(&scheduler, sequencer_propose_kind::VOTE), 3); + } + + /// The completion ack stays owed after the txn leaves `pending`, and + /// stops being proposed once the registry records it. + #[tokio::test] + async fn completion_ack_is_reproposed_until_the_registry_records_it() { + let txn_id = TxnId::new(21, 0); + let (mut scheduler, _dir) = scheduler_with_pending( + txn_id, + CommitState::AwaitingResolve { + committed: false, + redo_lsn: None, + }, + ); + let proposer = CapturingProposer::failing_first(1); + scheduler.sequencer_proposer = proposer.clone(); + scheduler.registry.seed_expected(cluster_txn_id(txn_id), 2); + + scheduler + .finish_resolved_commit(txn_id, staged_response(Status::Ok, None), false, None) + .await; + assert!(!scheduler.pending.contains_key(&txn_id)); + assert_eq!(proposer.attempt_count(), 1, "the first proposal is refused"); + + scheduler.retry_owed_sequencer_entries(); + assert_eq!( + proposer.accepted(), + vec![SequencerEntry::CompletionAck { + epoch: 21, + position: 0, + vshard_id: VSHARD, + }] + ); + scheduler.retry_owed_sequencer_entries(); + scheduler.retry_owed_sequencer_entries(); + assert_eq!(proposer.attempt_count(), 3); + + scheduler + .registry + .note_completion_ack(cluster_txn_id(txn_id), VSHARD); + scheduler.retry_owed_sequencer_entries(); + scheduler.retry_owed_sequencer_entries(); + assert_eq!( + proposer.attempt_count(), + 3, + "an applied ack is not proposed" + ); + assert!(scheduler.owed.is_empty()); + assert_eq!( + retries(&scheduler, sequencer_propose_kind::COMPLETION_ACK), + 2 + ); + } + + /// The registry removes a txn's entry once its outcome fired. An owed + /// entry for such a txn is settled, not proposed again. + #[tokio::test] + async fn entry_for_a_txn_whose_outcome_fired_is_settled() { + let (mut scheduler, _dir) = build_test_scheduler(VSHARD); + let proposer = CapturingProposer::failing_first(1); + scheduler.sequencer_proposer = proposer.clone(); + let txn_id = TxnId::new(22, 0); + let _outcome = scheduler + .registry + .register_completion(cluster_txn_id(txn_id), 1); + + scheduler.propose_sequencer_entry(txn_id, SchedulerProposal::CompletionAck); + scheduler + .registry + .note_completion_ack(cluster_txn_id(txn_id), VSHARD); + assert_eq!(scheduler.participant_progress(txn_id), None); + + scheduler.retry_owed_sequencer_entries(); + assert_eq!(proposer.attempt_count(), 1); + assert!(scheduler.owed.is_empty()); + } + + /// A txn this node never seeded keeps its entry owed: a missing registry + /// entry that was never seen does not settle it. + #[tokio::test] + async fn entry_for_an_unseeded_txn_stays_owed() { + let (mut scheduler, _dir) = build_test_scheduler(VSHARD); + let proposer = CapturingProposer::accepting(); + scheduler.sequencer_proposer = proposer.clone(); + let txn_id = TxnId::new(23, 0); + + scheduler.propose_sequencer_entry(txn_id, SchedulerProposal::OllpMismatch); + scheduler.retry_owed_sequencer_entries(); + scheduler.retry_owed_sequencer_entries(); + + assert_eq!(proposer.attempt_count(), 2); + assert_eq!(scheduler.owed.len(), 1); + } + + /// The run loop's stall tick drives the retry. + #[tokio::test] + async fn run_loop_reproposes_a_refused_entry_on_the_stall_tick() { + let (mut scheduler, _dir) = build_test_scheduler(VSHARD); + let proposer = CapturingProposer::failing_first(1); + scheduler.sequencer_proposer = proposer.clone(); + scheduler.propose_sequencer_entry( + TxnId::new(24, 0), + SchedulerProposal::RoutingFailed { + detail: "unroutable".to_string(), + }, + ); + assert!(proposer.accepted().is_empty()); + + let running = spawn_scheduler_loop(scheduler); + let accepted = tokio::time::timeout(Duration::from_secs(5), async { + while proposer.accepted().is_empty() { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await; + running.stop().await; + + assert!(accepted.is_ok(), "the stall tick proposes the entry again"); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs index 6eb7c58da..9392b872e 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs @@ -53,16 +53,19 @@ impl Scheduler { SchedulerInput::Txn(txn) => self.process_new_txn(txn), SchedulerInput::Reserve { owner, key } => self.install_reservation(owner, key), SchedulerInput::Release { owner, reason } => self.release_reservation(owner, reason), + SchedulerInput::CutMarker { hlc } => self.receive_cut_marker(hlc), } } /// The replicated epoch an input is stamped with — the monotonic logical - /// clock the lease reap advances on. + /// clock the lease reap advances on. A cut marker carries no epoch, so it + /// reports `0`, which never advances the clock. fn input_epoch(input: &SchedulerInput) -> u64 { match input { SchedulerInput::Txn(txn) => txn.epoch, SchedulerInput::Reserve { owner, .. } => owner.epoch, SchedulerInput::Release { owner, .. } => owner.epoch, + SchedulerInput::CutMarker { .. } => 0, } } @@ -243,21 +246,45 @@ impl Scheduler { self.dependent_barrier.insert(txn_id, barrier); } - /// Called when a transaction completes (success or infrastructure error). + /// Complete an in-flight txn (success or infrastructure error). + /// + /// Releases the lock-table owner recorded in its `pending` entry. A txn + /// with no `pending` entry has already completed, so this logs and + /// releases nothing. A txn that fails before it enters `pending` uses + /// [`Self::on_unpending_txn_complete`] instead. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn on_txn_complete( &mut self, txn_id: TxnId, ) { - let epoch = txn_id.epoch; - // Recover the lock-table owner (equals `txn_id` unless a reservation - // owned the lock). Blocked txns never reach here, so `pending` always - // holds the entry by the time a txn completes. - let lock_owner = self - .pending - .get(&txn_id) - .map(|p| p.lock_owner) - .unwrap_or(txn_id); + let Some(pending) = self.pending.remove(&txn_id) else { + tracing::error!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + "calvin: completion for a txn with no pending entry; nothing to release" + ); + return; + }; + // The flush applied the txn's redo record. + if let Some(records) = pending.redo_records { + records.settle(); + } + self.release_and_mark_applied(txn_id, pending.lock_owner); + } + + /// Complete a txn that failed before it entered `pending`, releasing the + /// locks held under `lock_owner`. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn on_unpending_txn_complete( + &mut self, + txn_id: TxnId, + lock_owner: TxnId, + ) { + self.release_and_mark_applied(txn_id, lock_owner); + } + /// Release `lock_owner`'s locks, dispatch the promoted waiters, and mark + /// `txn_id`'s position applied. + fn release_and_mark_applied(&mut self, txn_id: TxnId, lock_owner: TxnId) { // Release this txn's locks. `release` promotes any waiter queued behind // each freed key to holder (moving it pending -> held) and returns the // fully-promoted ids. Those ids are already holders in the table the @@ -275,11 +302,11 @@ impl Scheduler { // once ALL of its positions for this vShard have terminally completed, // so any advertised watermark reflects a FULLY-applied epoch — the value // `BEGIN` needs for a torn-free cross-shard snapshot anchor. - if let Some(watermark) = self.applied.mark_applied(epoch, txn_id.position) { + let folded = self.applied.mark_applied(txn_id.epoch, txn_id.position); + self.applied_mirror.mark(txn_id.epoch, txn_id.position); + if let Some(watermark) = folded { self.publish_watermark(watermark); } - - self.pending.remove(&txn_id); } /// Dispatch transactions that a `LockManager::release` promoted to holder. @@ -361,105 +388,113 @@ impl Scheduler { #[cfg(test)] mod tests { use super::*; - use std::collections::{BTreeSet, HashMap}; - use std::sync::Mutex; + use std::collections::BTreeSet; use std::sync::atomic::Ordering; - use nodedb_cluster::MultiRaft; - use nodedb_cluster::RoutingTable; - use nodedb_cluster::calvin::types::{ - EngineKeySet, ReadWriteSet, SortedVec, TxClass, VersionedReadSet, - }; - use nodedb_cluster::calvin::{CalvinCompletionRegistry, SequencerStateMachine}; + use nodedb_cluster::calvin::CalvinCompletionRegistry; + use nodedb_physical::physical_plan::PhysicalPlan; + use nodedb_physical::physical_plan::meta::MetaOp; use nodedb_types::TenantId; - use super::super::scheduler::SchedulerParams; - use crate::bridge::dispatch::Dispatcher; - use crate::control::cluster::calvin::scheduler::lock_manager::LockManager; - use crate::control::cluster::calvin::scheduler::metrics::SchedulerMetrics; - use crate::control::cluster::calvin::scheduler::{NOT_YET_APPLIED_EPOCH, SchedulerConfig}; - use crate::control::state::SharedState; - use crate::wal::WalManager; - - /// Build a minimally-wired `Scheduler` for driver-level unit tests. The Data - /// Plane is NOT started — tests exercise Control-Plane routing, guards, and - /// request dispatch only, so no core loop is needed. The returned `TempDir` - /// must be kept alive for the scheduler's lifetime (backs the WAL and - /// Raft storage). - fn build_test_scheduler(vshard_id: u32) -> (Scheduler, tempfile::TempDir) { + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + await_data_plane_request, build_test_scheduler, build_test_scheduler_with_data_side, + fill_tenant_inflight, make_sequenced_txn, make_validate_only_txn, release_filler, + spawn_scheduler_loop, test_coll_vshard, + }; + + /// A refused stage dispatch leaves the txn unapplied and publishes no + /// watermark for its epoch. + #[tokio::test] + async fn stage_dispatch_refused_at_capacity_leaves_txn_unapplied() { let registry = CalvinCompletionRegistry::new_detached(); - let dir = tempfile::tempdir().unwrap(); - let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("test.wal")).unwrap()); - let (dispatcher, mut data_sides) = Dispatcher::new(1, 64); - let _data_side = data_sides - .pop() - .expect("one configured core has one data side"); - let shared = SharedState::new(dispatcher, wal).unwrap(); - - let rt = RoutingTable::uniform(1, &[1], 1); - let multi_raft = Arc::new(Mutex::new(MultiRaft::new(1, rt, dir.path().to_path_buf()))); - - let sequencer_state_machine = Arc::new(Mutex::new(SequencerStateMachine::new( - HashMap::new(), - Arc::clone(®istry), - ))); - - let (_tx, receiver) = tokio::sync::mpsc::channel(16); - let (_rr_tx, read_result_rx) = tokio::sync::mpsc::channel(16); - let (_prom_tx, promotion_rx) = tokio::sync::mpsc::unbounded_channel(); - let (verdict_tx, verdict_rx) = tokio::sync::mpsc::channel(16); - registry.register_verdict_signal_sender(vshard_id, verdict_tx); - - let lock_manager = Arc::new(Mutex::new(LockManager::new())); - - let scheduler = Scheduler::new(SchedulerParams { - vshard_id, - receiver, - shared, - multi_raft, - sequencer_state_machine, - // A freshly-built scheduler has applied nothing, so its watermark is the - // not-yet-applied sentinel (matching `read_applied_recovery` for a clean - // node). Hardcoding `0` here would instead claim epoch 0 is fully applied, - // making the exactly-once gate (`AppliedGate::is_applied`) short-circuit - // every epoch-0 replay before it reaches the lock table — silently - // defeating the end-to-end drain tests below. - fully_applied_epoch: NOT_YET_APPLIED_EPOCH, - applied_tail: BTreeSet::new(), - rebuild_target_epoch: 0, - config: SchedulerConfig::default(), - metrics: SchedulerMetrics::new(), - read_result_rx, - lock_manager, - promotion_rx, - registry, - verdict_rx, - }); - (scheduler, dir) + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let shared = Arc::clone(&scheduler.shared); + fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + let watermark_before = shared.calvin.last_applied_epoch.load(Ordering::Acquire); + + scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + + assert!( + !scheduler.applied.is_applied(3, 0), + "a capacity refusal must not mark the position applied" + ); + assert_eq!( + shared.calvin.last_applied_epoch.load(Ordering::Acquire), + watermark_before, + "a capacity refusal must not publish a watermark for the txn's epoch" + ); } - fn make_sequenced_txn(epoch: u64, position: u32) -> SequencedTxn { - let write_set = ReadWriteSet::new(vec![EngineKeySet::Document { - collection: "test_coll".to_string(), - surrogates: SortedVec::new(vec![1]), - }]); - let tx_class = TxClass::new_single_vshard( - ReadWriteSet::new(vec![]), - write_set, - vec![], - TenantId::new(1), - None, - VersionedReadSet::default(), - ) - .expect("valid TxClass"); - SequencedTxn { - epoch, - position, - tx_class, - epoch_system_ms: 1_700_000_000_000, - epoch_vshard_txn_count: 1, - lock_owner: None, - } + /// A refused stage dispatch keeps the txn's key locks: a later txn on the + /// same key queues behind it. + #[tokio::test] + async fn stage_dispatch_refused_at_capacity_keeps_key_locks() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let shared = Arc::clone(&scheduler.shared); + fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + + scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(4, 0))); + + assert!( + scheduler.blocked.contains_key(&TxnId::new(4, 0)), + "a txn on the same key must block behind the refused txn's held locks" + ); + } + + /// A refused stage dispatch leaves no request-tracker entry behind. + #[tokio::test] + async fn stage_dispatch_refused_at_capacity_leaves_no_tracker_entry() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let shared = Arc::clone(&scheduler.shared); + fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + let tracked_before = shared.tracker.in_flight(); + + scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + + assert_eq!( + shared.tracker.in_flight(), + tracked_before, + "a refused dispatch must not leave a registered tracker entry" + ); + } + + /// Once a Data Plane response frees tenant capacity, the refused txn's + /// stage request reaches the Data Plane. + #[tokio::test] + async fn stage_dispatch_refused_at_capacity_is_retried_after_capacity_frees() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let shared = Arc::clone(&scheduler.shared); + let fillers = fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + + scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + let running = spawn_scheduler_loop(scheduler); + release_filler(&shared, &mut data_side, fillers[0]); + + let arrived = await_data_plane_request(&mut data_side, |plan| { + matches!( + plan, + PhysicalPlan::Meta(MetaOp::CalvinExecuteStatic { + epoch: 3, + position: 0, + .. + }) + ) + }) + .await; + running.stop().await; + + assert!( + arrived, + "the refused stage request must reach the Data Plane once capacity frees" + ); } #[tokio::test] diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/read_result.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/read_result.rs index 55bcfd619..4a5a11539 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/read_result.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/read_result.rs @@ -77,7 +77,9 @@ impl Scheduler { self.metrics.record_infra_abort( crate::control::cluster::calvin::scheduler::metrics::infra_abort_reason::PASSIVE_PARTICIPANT_TIMEOUT, ); - self.on_txn_complete(txn_id); + // A barrier txn never entered `pending`: release its locks + // under the owner the barrier recorded. + self.on_unpending_txn_complete(txn_id, barrier.lock_owner); } } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/redo_window.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/redo_window.rs new file mode 100644 index 000000000..04dab5e8b --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/redo_window.rs @@ -0,0 +1,124 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Close the outcome-floor window of a committed txn's `TransactionRedo` +//! record. +//! +//! The record joins the pending txn when it is appended. The window settles +//! when the txn completes: the flush applied the record. It holds when the +//! txn stays unapplied here, because restart replay must reach the record. + +use super::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; + +impl Scheduler { + /// Hold the redo record of `txn_id`: the scheduler halted with the txn + /// unapplied. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn hold_redo_records( + &mut self, + txn_id: TxnId, + ) { + if let Some(records) = self + .pending + .get_mut(&txn_id) + .and_then(|pending| pending.redo_records.take()) + { + records.hold(); + } + } + + /// Hold the redo record of every pending txn: the scheduler stops with + /// them unapplied. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn hold_all_redo_records( + &mut self, + ) { + for pending in self.pending.values_mut() { + if let Some(records) = pending.redo_records.take() { + records.hold(); + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::cluster::calvin::scheduler::driver::core::halt::{HaltReason, HaltStep}; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::scheduler_with_pending; + use crate::control::cluster::calvin::scheduler::driver::types::CommitState; + use crate::control::server::dispatch_utils::MintedRecords; + use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; + + /// Attach an appended redo record to the pending txn, as a committed + /// resolve does. + fn attach_redo_record(scheduler: &mut Scheduler, txn_id: TxnId) -> Lsn { + let records = MintedRecords::open(&scheduler.shared.outcome_floor); + let lsn = records + .appender(&scheduler.shared.wal, crate::wal::manager::NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"redo", + ) + .expect("append"); + if let Some(pending) = scheduler.pending.get_mut(&txn_id) { + pending.redo_records = Some(records); + } + lsn + } + + fn awaiting_flush() -> CommitState { + CommitState::AwaitingResolve { + committed: true, + redo_lsn: None, + } + } + + #[tokio::test] + async fn a_halted_txn_holds_its_redo_record() { + let txn_id = TxnId::new(3, 1); + let (mut scheduler, _dir) = scheduler_with_pending(txn_id, awaiting_flush()); + let lsn = attach_redo_record(&mut scheduler, txn_id); + + scheduler.halt_apply( + txn_id, + HaltReason::FlushFailed, + HaltStep::Flush, + "flush refused".to_string(), + ); + + let floor = &scheduler.shared.outcome_floor; + assert!( + floor.floor() < lsn, + "the held record keeps the floor below it" + ); + assert_eq!(floor.leaked_windows(), 0); + } + + #[tokio::test] + async fn a_completed_txn_settles_its_redo_record() { + let txn_id = TxnId::new(3, 2); + let (mut scheduler, _dir) = scheduler_with_pending(txn_id, awaiting_flush()); + let lsn = attach_redo_record(&mut scheduler, txn_id); + + scheduler.on_txn_complete(txn_id); + + let floor = &scheduler.shared.outcome_floor; + assert_eq!(floor.floor(), lsn); + assert_eq!(floor.leaked_windows(), 0); + } + + #[tokio::test] + async fn a_stopping_scheduler_holds_every_pending_redo_record() { + let txn_id = TxnId::new(3, 3); + let (mut scheduler, _dir) = scheduler_with_pending(txn_id, awaiting_flush()); + let lsn = attach_redo_record(&mut scheduler, txn_id); + + scheduler.hold_all_redo_records(); + + let floor = &scheduler.shared.outcome_floor; + assert!(floor.floor() < lsn); + assert_eq!(floor.leaked_windows(), 0); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/request.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/request.rs index 7648c3586..44bd7b658 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/request.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/request.rs @@ -9,6 +9,12 @@ use crate::bridge::envelope::{Admission, ExemptReason, Priority, Request}; use crate::types::{DatabaseId, Lsn, ReadConsistency, RequestId, TenantId, VShardId}; use nodedb_physical::physical_plan::PhysicalPlan; +/// The event source every Calvin sub-operation runs with. The redo record a +/// committed Calvin transaction appends carries the same source, so WAL +/// replay rebuilds the events its flush emits. +pub(in crate::control::cluster::calvin::scheduler::driver::core) const CALVIN_EVENT_SOURCE: + crate::event::EventSource = crate::event::EventSource::User; + impl Scheduler { /// Builds a `Request` for an already-sequenced Calvin sub-operation. /// @@ -32,16 +38,12 @@ impl Scheduler { database_id, vshard_id: VShardId::new(self.vshard_id), plan, - // no-determinism: scheduler deadline controls waiting, not ordered state. - deadline: Instant::now() - + Duration::from_millis( - self.config.epoch_duration_ms * u64::from(self.config.txn_deadline_multiplier), - ), + deadline: self.request_deadline(), priority: Priority::Normal, trace_id: nodedb_types::TraceId([0u8; 16]), consistency: ReadConsistency::Strong, idempotency_key: None, - event_source: crate::event::EventSource::User, + event_source: CALVIN_EVENT_SOURCE, user_roles: Vec::new(), user_id: None, statement_digest: None, @@ -51,4 +53,16 @@ impl Scheduler { admission: Admission::Exempt(ExemptReason::AlreadyOrdered), } } + + /// The deadline for a Calvin sub-operation sent now: one epoch duration + /// times the configured deadline multiplier. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn request_deadline( + &self, + ) -> Instant { + // no-determinism: scheduler deadline controls waiting, not ordered state. + Instant::now() + + Duration::from_millis( + self.config.epoch_duration_ms * u64::from(self.config.txn_deadline_multiplier), + ) + } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs index af8e2f59b..9a7fa84f0 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs @@ -14,8 +14,7 @@ use nodedb_physical::physical_plan::{ }; use crate::types::{DatabaseId, VShardId}; -#[cfg(test)] -use nodedb_types::QualifiedCollection; +use nodedb_types::{CollectionKey, QualifiedCollection}; /// Where a `PhysicalPlan` routes for Calvin cross-shard scheduling purposes. /// @@ -38,23 +37,35 @@ pub(crate) enum PlanRouting { Unroutable(&'static str), } +/// Whether any read in `reads` homes to `vshard_id`. A read entry carries the +/// plan's database-qualified name. An entry that does not de-qualify homes +/// nowhere, so a txn whose only work is such a read fails loudly as one that +/// homes no local work. pub(crate) fn homes_versioned_read( reads: &nodedb_types::calvin::VersionedReadSet, database_id: DatabaseId, vshard_id: u32, ) -> bool { reads.iter().any(|entry| { - VShardId::from_collection_in_database(database_id, &entry.collection).as_u32() == vshard_id + CollectionKey::from_qualified_str(database_id, &entry.collection) + .is_ok_and(|key| key.vshard().as_u32() == vshard_id) }) } -fn collection_vshard_in_database(database_id: DatabaseId, collection: &str) -> VShardId { - VShardId::from_collection_in_database(database_id, collection) +/// Route a collection-homed write to the vShard of its canonical key. The +/// plan carries the database-qualified name, de-qualified here. +fn collection_routing(database_id: DatabaseId, collection: &QualifiedCollection) -> PlanRouting { + match CollectionKey::from_qualified(database_id, collection) { + Ok(key) => PlanRouting::Vshards(vec![key.vshard()]), + Err(_) => PlanRouting::Unroutable( + "collection name lacks the qualifier of the transaction's database", + ), + } } #[cfg(test)] fn collection_vshard(collection: &str) -> VShardId { - collection_vshard_in_database(DatabaseId::DEFAULT, collection) + CollectionKey::from_bare(DatabaseId::DEFAULT, collection).vshard() } /// Returns the routing decision for `plan`. Exhaustive over every @@ -105,10 +116,7 @@ fn document_routing(op: &DocumentOp, database_id: DatabaseId) -> PlanRouting { // was derived from homes elsewhere, and the pair is dual-homed by the // two tasks' own vshards rather than by one plan claiming both. | DocumentOp::ApplyBalanceDelta { collection, .. } => { - PlanRouting::Vshards(vec![collection_vshard_in_database( - database_id, - collection.as_str(), - )]) + collection_routing(database_id, collection) } // Never scheduled: the write-resolve orchestrator proposes it through // Raft directly, on the vshard of the collection it resolved. @@ -117,10 +125,7 @@ fn document_routing(op: &DocumentOp, database_id: DatabaseId) -> PlanRouting { ), DocumentOp::InsertSelect { target_collection, .. - } => PlanRouting::Vshards(vec![collection_vshard_in_database( - database_id, - target_collection.as_str(), - )]), + } => collection_routing(database_id, target_collection), // Both join the target with a DIFFERENT source collection; nothing on // the plan enforces the two live on the same vshard. DocumentOp::Merge { .. } | DocumentOp::UpdateFromJoin { .. } => PlanRouting::Unroutable( @@ -165,10 +170,7 @@ fn kv_routing(op: &KvOp, database_id: DatabaseId) -> PlanRouting { // that collection's vshard like every other single-collection write. | KvOp::PredicateUpdate { collection, .. } | KvOp::PredicateDelete { collection, .. } => { - PlanRouting::Vshards(vec![collection_vshard_in_database( - database_id, - collection.as_str(), - )]) + collection_routing(database_id, collection) } // Source and dest are DIFFERENT collections; no co-location guarantee. KvOp::TransferItem { .. } => PlanRouting::Unroutable( @@ -196,7 +198,8 @@ fn kv_routing(op: &KvOp, database_id: DatabaseId) -> PlanRouting { | KvOp::SortedIndexTopK { .. } | KvOp::SortedIndexRange { .. } | KvOp::SortedIndexCount { .. } - | KvOp::SortedIndexScore { .. } => PlanRouting::NotAWrite, + | KvOp::SortedIndexScore { .. } + | KvOp::SortedIndexTxnRead { .. } => PlanRouting::NotAWrite, } } @@ -215,12 +218,7 @@ fn vector_routing(op: &VectorOp, database_id: DatabaseId) -> PlanRouting { | VectorOp::DirectInsertIfAbsent { collection, .. } | VectorOp::DirectDelete { collection, .. } | VectorOp::DirectTruncate { collection, .. } - | VectorOp::DirectUpdate { collection, .. } => { - PlanRouting::Vshards(vec![collection_vshard_in_database( - database_id, - collection.as_str(), - )]) - } + | VectorOp::DirectUpdate { collection, .. } => collection_routing(database_id, collection), // Never scheduled: the write-resolve orchestrator proposes it through // Raft directly, on the vshard of the collection it resolved. VectorOp::ResolvedDirectWrite { .. } => PlanRouting::Unroutable( @@ -304,10 +302,7 @@ fn graph_routing(op: &GraphOp) -> PlanRouting { fn timeseries_routing(op: &TimeseriesOp, database_id: DatabaseId) -> PlanRouting { match op { TimeseriesOp::Ingest { collection, .. } | TimeseriesOp::Truncate { collection, .. } => { - PlanRouting::Vshards(vec![collection_vshard_in_database( - database_id, - collection.as_str(), - )]) + collection_routing(database_id, collection) } // Read-only: it reports the lines the wrapped ingest would store and // mutates nothing. @@ -322,12 +317,7 @@ fn columnar_routing(op: &ColumnarOp, database_id: DatabaseId) -> PlanRouting { | ColumnarOp::Delete { collection, .. } | ColumnarOp::ResolvedUpdate { collection, .. } | ColumnarOp::ResolvedDelete { collection, .. } - | ColumnarOp::Truncate { collection, .. } => { - PlanRouting::Vshards(vec![collection_vshard_in_database( - database_id, - collection.as_str(), - )]) - } + | ColumnarOp::Truncate { collection, .. } => collection_routing(database_id, collection), ColumnarOp::Scan { .. } | ColumnarOp::MaterializeScan { .. } | ColumnarOp::ResolveDml { .. } => PlanRouting::NotAWrite, @@ -346,12 +336,7 @@ fn crdt_routing(op: &CrdtOp, database_id: DatabaseId) -> PlanRouting { | CrdtOp::SetConstraints { collection, .. } | CrdtOp::DropConstraints { collection, .. } | CrdtOp::RestoreToVersion { collection, .. } - | CrdtOp::ImportSnapshot { collection, .. } => { - PlanRouting::Vshards(vec![collection_vshard_in_database( - database_id, - collection.as_str(), - )]) - } + | CrdtOp::ImportSnapshot { collection, .. } => collection_routing(database_id, collection), CrdtOp::Read { .. } | CrdtOp::PreviewApply { .. } | CrdtOp::ReadConstraints { .. } @@ -569,21 +554,23 @@ mod tests { #[test] fn collection_routing_preserves_database_scope() { + let db = DatabaseId::new(7); let collection = (0..2048) .map(|i| format!("db_scoped_{i}")) .find(|name| { - collection_vshard_in_database(DatabaseId::DEFAULT, name) - != collection_vshard_in_database(DatabaseId::new(7), name) + CollectionKey::from_bare(DatabaseId::DEFAULT, name).vshard() + != CollectionKey::from_bare(db, name).vshard() }) .expect("collection whose home differs by database"); let plan = PhysicalPlan::Document(DocumentOp::Truncate { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, &collection), + collection: QualifiedCollection::new(db, &collection), restart_identity: false, resolved_sum_targets: Vec::new(), declared_primary_key: None, }); - let expected = collection_vshard_in_database(DatabaseId::new(7), &collection); - match plan_vshard_in_database(&plan, DatabaseId::new(7)) { + // The plan carries the qualified name; the home is the bare key's. + let expected = CollectionKey::from_bare(db, &collection).vshard(); + match plan_vshard_in_database(&plan, db) { PlanRouting::Vshards(actual) => assert_eq!(actual, vec![expected]), PlanRouting::ControlPlaneOnly | PlanRouting::NotAWrite | PlanRouting::Unroutable(_) => { panic!("document truncate must be database-scoped") diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs index 40aecbc67..b9c66e849 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs @@ -5,19 +5,22 @@ use std::collections::BTreeMap; use std::sync::{Arc, Mutex}; -use tokio::sync::mpsc; +use tokio::sync::{Notify, mpsc}; use tracing::info; use nodedb_cluster::MultiRaft; use nodedb_cluster::calvin::types::SchedulerInput; -use nodedb_cluster::calvin::{ - CalvinCompletionRegistry, SEQUENCER_GROUP_ID, SequencerEntry, SequencerStateMachine, - VerdictSignal, -}; +use nodedb_cluster::calvin::{CalvinCompletionRegistry, SequencerStateMachine, VerdictSignal}; use super::super::barrier::{PendingDependentBarrier, ReadResultEvent}; use super::super::config::SchedulerConfig; use super::super::types::{BlockedTxn, PendingTxn}; +use super::catch_up::CatchUpDrain; +use super::deferred::DeferredQueue; +use super::halt::HaltLatch; +use super::intake::IntakeGate; +use super::owed::OwedEntries; +use super::sequencer_proposer::SequencerProposer; use crate::bridge::envelope::Response; use crate::control::cluster::calvin::scheduler::lock_manager::{LockManager, TxnId}; use crate::control::cluster::calvin::scheduler::metrics::SchedulerMetrics; @@ -49,10 +52,17 @@ pub struct Scheduler { /// Shared control-plane state used for dispatch, response tracking, WAL, /// and request-id allocation. pub(in crate::control::cluster::calvin::scheduler::driver::core) shared: Arc, - /// Handle to MultiRaft so completion acknowledgements can be proposed to - /// the sequencer group. + /// Handle to MultiRaft for the data-group leader check and the catch-up + /// read of the sequencer log. pub(in crate::control::cluster::calvin::scheduler::driver::core) multi_raft: Arc>, + /// Hands sequencer entries to the sequencer group, locally on its leader + /// and by forward from any other node. + pub(in crate::control::cluster::calvin::scheduler::driver::core) sequencer_proposer: + Arc, + /// Sequencer entries proposed and not yet seen applied. See + /// [`super::owed`]. + pub(in crate::control::cluster::calvin::scheduler::driver::core) owed: OwedEntries, /// Shared handle to the sequencer state machine. The state machine records, /// per vShard, the earliest Raft index whose fan-out `try_send` was DROPPED /// (channel Full/Closed) so a dropped `SchedulerInput` never permanently @@ -66,13 +76,14 @@ pub struct Scheduler { Arc>, /// Deterministic lock manager for this vshard. Shared (via `Arc>`) /// with the Control-Plane write-admission gate through - /// `SharedState.calvin_lock_managers`, so a fast-path point write contends + /// `SharedState.calvin.lock_managers`, so a fast-path point write contends /// on the SAME lock table this scheduler validates against. The scheduler /// still runs single-threaded per vShard, so the mutex is uncontended except /// for the brief probe the gate takes. pub(in crate::control::cluster::calvin::scheduler::driver::core) lock_manager: Arc>, - /// In-flight static/active transactions awaiting executor response. + /// In-flight static/active transactions awaiting executor response, + /// including those whose request waits in `deferred` for capacity. /// `BTreeMap` ensures deterministic iteration order. pub(in crate::control::cluster::calvin::scheduler::driver::core) pending: BTreeMap, @@ -101,6 +112,13 @@ pub struct Scheduler { /// deterministic threshold below which an orphaned shared reservation is /// released. Purely a function of replicated input order — no wall clock. pub(in crate::control::cluster::calvin::scheduler::driver::core) max_input_epoch: u64, + /// Backup cut markers this scheduler received: the commit HLC floors they + /// set and the markers not yet reported. + pub(in crate::control::cluster::calvin::scheduler::driver::core) cut_floors: + crate::control::cluster::calvin::scheduler::cut_floor::CutFloors, + /// Shared mirror of `applied`, read by authorization coverage. + pub(in crate::control::cluster::calvin::scheduler::driver::core) applied_mirror: + Arc, /// Scheduler configuration. pub(in crate::control::cluster::calvin::scheduler::driver::core) config: SchedulerConfig, /// Metrics. @@ -108,7 +126,7 @@ pub struct Scheduler { /// Fan-in receiver for executor responses. /// /// Each dispatched transaction spawns a lightweight bridge task that - /// awaits the per-request `mpsc::Receiver` and forwards the + /// awaits the per-request `ResponseReceiver` and forwards the /// result here as a [`CompletionItem`]. The scheduler's `select!` loop /// includes this channel as a first-class arm so it wakes the moment /// any executor response is ready — no polling, no sleep. @@ -143,6 +161,18 @@ pub struct Scheduler { /// push, so a full/closed channel is never a correctness hazard. pub(in crate::control::cluster::calvin::scheduler::driver::core) verdict_rx: mpsc::Receiver, + /// Requests the bridge dispatcher refused at capacity, in refusal order. + /// Each txn stays in flight and holds its locks until its request is + /// re-sent. Holds at most one step per in-flight txn, plus one + /// write-version record per committed txn. + pub(in crate::control::cluster::calvin::scheduler::driver::core) deferred: DeferredQueue, + /// The bridge dispatcher's capacity-freed signal, cloned once at + /// construction. The run loop waits on it while requests are deferred. + pub(in crate::control::cluster::calvin::scheduler::driver::core) capacity_freed: Arc, + /// Last observed intake gate state. See [`super::intake`]. + pub(in crate::control::cluster::calvin::scheduler::driver::core) intake: IntakeGate, + /// First halt cause, once set. See [`super::halt`]. + pub(in crate::control::cluster::calvin::scheduler::driver::core) halt: HaltLatch, } /// Parameters for [`Scheduler::new`]. @@ -151,6 +181,9 @@ pub struct SchedulerParams { pub receiver: mpsc::Receiver, pub shared: Arc, pub multi_raft: Arc>, + /// The node's sequencer proposer. Production passes one + /// `RaftSequencerProposer` shared by every scheduler on the node. + pub sequencer_proposer: Arc, /// Shared sequencer state machine, source of the per-vShard catch-up index /// the drain replays from. Same `Arc` the Raft apply loop drives. pub sequencer_state_machine: Arc>, @@ -164,11 +197,11 @@ pub struct SchedulerParams { pub read_result_rx: mpsc::Receiver, /// The shared lock table for this vShard. Constructed by /// `reconcile_vshard_schedulers` and registered in - /// `SharedState.calvin_lock_managers` under the SAME `Arc` passed here. + /// `SharedState.calvin.lock_managers` under the SAME `Arc` passed here. pub lock_manager: Arc>, /// Receiver for gate-side lock promotions. Constructed by /// `reconcile_vshard_schedulers`; its `UnboundedSender` is registered in - /// `SharedState.calvin_promotion_senders` for this same vShard so a fast-path + /// `SharedState.calvin.promotion_senders` for this same vShard so a fast-path /// guard drop can hand promoted waiters back to this scheduler. pub promotion_rx: mpsc::UnboundedReceiver>, /// Shared completion registry for verdict probes on the commit barrier. @@ -187,6 +220,7 @@ impl Scheduler { receiver, shared, multi_raft, + sequencer_proposer, sequencer_state_machine, fully_applied_epoch, applied_tail, @@ -205,11 +239,28 @@ impl Scheduler { let completion_cap = config.channel_capacity; let (completion_tx, completion_rx) = mpsc::channel(completion_cap); + let applied_mirror = shared.authorization_fence.calvin_mirrors().register( + vshard_id, + fully_applied_epoch, + &applied_tail, + ); + + // A backup's cut waits on every scheduler this node runs. + shared.calvin.cuts.register(vshard_id); + + let capacity_freed = shared + .dispatcher + .lock() + .unwrap_or_else(|p| p.into_inner()) + .capacity_freed(); + Self { vshard_id, receiver, shared, multi_raft, + sequencer_proposer, + owed: OwedEntries::new(), sequencer_state_machine, lock_manager, pending: BTreeMap::new(), @@ -217,6 +268,8 @@ impl Scheduler { dependent_barrier: BTreeMap::new(), read_result_rx, applied: AppliedGate::new(fully_applied_epoch, applied_tail), + cut_floors: Default::default(), + applied_mirror, rebuild_target_epoch, max_input_epoch: 0, config, @@ -226,6 +279,10 @@ impl Scheduler { promotion_rx, registry, verdict_rx, + deferred: DeferredQueue::new(), + capacity_freed, + intake: IntakeGate::default(), + halt: HaltLatch::default(), } } @@ -262,19 +319,23 @@ impl Scheduler { /// Publish an advanced fully-applied watermark to the metrics gauge and the /// shared cross-shard snapshot anchor. /// - /// `BEGIN` reads `SharedState::last_applied_calvin_epoch` to anchor a + /// `BEGIN` reads `CalvinLocalState::last_applied_epoch` to anchor a /// session's cross-shard snapshot version, so it MUST reflect the /// FULLY-applied epoch — never an epoch that has only some of its positions /// committed, which would let a session anchor on a torn epoch. `fetch_max` /// keeps it monotonic across all per-vShard schedulers writing the counter. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn publish_watermark( - &self, + &mut self, watermark: u64, ) { self.metrics.update_last_applied_epoch(watermark); + self.applied_mirror.fold(watermark); self.shared - .last_applied_calvin_epoch + .calvin + .last_applied_epoch .fetch_max(watermark, std::sync::atomic::Ordering::Release); + // A marker passes once every epoch delivered before it folded. + self.report_passed_cuts(); } /// Spawn a bridge task that awaits a single executor response and forwards @@ -287,7 +348,7 @@ impl Scheduler { &self, txn_id: TxnId, request_id: RequestId, - mut response_rx: mpsc::Receiver, + mut response_rx: crate::control::ResponseReceiver, ) { let tx = self.completion_tx.clone(); tokio::spawn(async move { @@ -314,10 +375,32 @@ impl Scheduler { let mut stall_tick = tokio::time::interval(self.config.verdict_stall_warn() / 4); stall_tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); + // Woken when a routed Data-Plane response frees dispatcher capacity. + let capacity_freed = Arc::clone(&self.capacity_freed); + // Set when a tick left armed catch-up unreplayed. The next open-gate + // pass fires the tick at once to resume it. + let mut catch_up_resume = false; + loop { + // Register for the capacity wake BEFORE the re-send pass. A + // response routed after a refusal but before this point is + // covered by the pass itself. One routed after it wakes the arm. + let capacity_notified = capacity_freed.notified(); + tokio::pin!(capacity_notified); + capacity_notified.as_mut().enable(); + if self.resends_deferred() { + self.redispatch_deferred(); + } + self.check_dependent_barrier_timeouts(); self.check_awaiting_verdict_stalls(); + let intake_open = self.refresh_intake_gate(); + if intake_open && catch_up_resume { + catch_up_resume = false; + stall_tick.reset_immediately(); + } + tokio::select! { biased; @@ -328,7 +411,10 @@ impl Scheduler { maybe_completion = self.completion_rx.recv() => { if let Some((txn_id, request_id, resp_opt)) = maybe_completion { - self.handle_completion(txn_id, request_id, resp_opt); + // Awaited in the arm: the loop takes no other input + // until this completion, its durability wait included, + // is fully handled. + self.handle_completion(txn_id, request_id, resp_opt).await; } } @@ -357,7 +443,12 @@ impl Scheduler { } } - maybe_txn = self.receiver.recv() => { + _ = &mut capacity_notified, if self.resends_deferred() => { + // Capacity freed: the next loop pass re-sends deferred + // requests in FIFO order. + } + + maybe_txn = self.receiver.recv(), if intake_open => { match maybe_txn { Some(input) => self.process_scheduler_input(input), None => { @@ -375,54 +466,20 @@ impl Scheduler { // (channel Full/Closed) so a missed `SchedulerInput` never // permanently diverges this vShard's lock table from its peers. // O(1) common case (no pending catch-up). See `drain_catch_up`. - self.drain_catch_up(); + // A closed intake gate skips the drain until it opens. + catch_up_resume = + !intake_open || self.drain_catch_up() == CatchUpDrain::Remaining; + // Propose again every owed sequencer entry not yet applied. + self.retry_owed_sequencer_entries(); // The top-of-loop check_awaiting_verdict_stalls / - // check_dependent_barrier_timeouts do the stall work on every - // wake; this arm guarantees the loop wakes to run them (and the - // drain) when no other event arrives. + // check_dependent_barrier_timeouts and the deferred re-send + // pass run on every wake; this arm guarantees the loop wakes + // to run them (and the drain) when no other event arrives. } } } - } - - /// Encode `entry` as MessagePack and propose it to the sequencer Raft group. - /// - /// Logs a warning on encode failure or propose failure; never panics. - /// `op_name` is a short human-readable label used in warning messages - /// (e.g. `"completion ack"`, `"OLLP mismatch signal"`). - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn propose_sequencer_entry( - &self, - entry: SequencerEntry, - txn_id: TxnId, - op_name: &str, - ) { - match zerompk::to_msgpack_vec(&entry) { - Ok(bytes) => { - if let Err(e) = self - .multi_raft - .lock() - .unwrap_or_else(|p| p.into_inner()) - .propose_to_group(SEQUENCER_GROUP_ID, bytes) - { - tracing::warn!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - error = %e, - "calvin: failed to propose {op_name}", - ); - } - } - Err(e) => { - tracing::warn!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - error = %e, - "calvin: failed to encode {op_name}", - ); - } - } + // Every txn still pending stays unapplied on this replica. + self.hold_all_redo_records(); } /// Allocate a fresh request ID for a dispatch. @@ -437,70 +494,9 @@ impl Scheduler { #[cfg(test)] mod tests { use super::*; - use std::collections::{BTreeSet, HashMap}; - - use nodedb_cluster::RoutingTable; - - use crate::bridge::dispatch::Dispatcher; - - /// Build a minimally-wired `Scheduler` for driver-level unit tests. The Data - /// Plane is NOT started — tests exercise Control-Plane routing, guards, and - /// request dispatch only, so no core loop is needed. The returned `TempDir` - /// must be kept alive for the scheduler's lifetime (backs the WAL and - /// Raft storage). - fn build_test_scheduler(vshard_id: u32) -> (Scheduler, tempfile::TempDir) { - let registry = CalvinCompletionRegistry::new_detached(); - let dir = tempfile::tempdir().unwrap(); - let wal = Arc::new( - crate::wal::WalManager::open_for_testing(&dir.path().join("test.wal")).unwrap(), - ); - let (dispatcher, mut data_sides) = Dispatcher::new(1, 64); - let _data_side = data_sides - .pop() - .expect("one configured core has one data side"); - let shared = SharedState::new(dispatcher, wal).unwrap(); - - let rt = RoutingTable::uniform(1, &[1], 1); - let multi_raft = Arc::new(Mutex::new(MultiRaft::new(1, rt, dir.path().to_path_buf()))); + use std::collections::BTreeSet; - let sequencer_state_machine = Arc::new(Mutex::new(SequencerStateMachine::new( - HashMap::new(), - Arc::clone(®istry), - ))); - - let (_tx, receiver) = mpsc::channel(16); - let (_rr_tx, read_result_rx) = mpsc::channel(16); - let (_prom_tx, promotion_rx) = mpsc::unbounded_channel(); - let (verdict_tx, verdict_rx) = mpsc::channel(16); - registry.register_verdict_signal_sender(vshard_id, verdict_tx); - - let lock_manager = Arc::new(Mutex::new(LockManager::new())); - - let scheduler = Scheduler::new(SchedulerParams { - vshard_id, - receiver, - shared, - multi_raft, - sequencer_state_machine, - // A freshly-built scheduler has applied nothing, so its watermark is the - // not-yet-applied sentinel (matching `read_applied_recovery` for a clean - // node). Hardcoding `0` here would instead claim epoch 0 is fully applied, - // making the exactly-once gate (`AppliedGate::is_applied`) short-circuit - // every epoch-0 replay before it reaches the lock table — silently - // defeating the end-to-end drain tests below. - fully_applied_epoch: NOT_YET_APPLIED_EPOCH, - applied_tail: BTreeSet::new(), - rebuild_target_epoch: 0, - config: SchedulerConfig::default(), - metrics: SchedulerMetrics::new(), - read_result_rx, - lock_manager, - promotion_rx, - registry, - verdict_rx, - }); - (scheduler, dir) - } + use crate::control::cluster::calvin::scheduler::driver::core::test_support::build_test_scheduler; /// A freshly-recovered scheduler (`fully_applied_epoch` still the /// `NOT_YET_APPLIED_EPOCH` sentinel) with a REAL, non-zero rebuild target must diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/mod.rs new file mode 100644 index 000000000..392949c8e --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/mod.rs @@ -0,0 +1,10 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! How the Calvin scheduler hands its sequencer entries to the sequencer +//! Raft group. + +pub mod raft; +pub mod seam; + +pub use raft::{MAX_INFLIGHT_SEQUENCER_FORWARDS, RaftSequencerProposer}; +pub use seam::{ProposeDispatch, SequencerProposeError, SequencerProposer}; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/raft.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/raft.rs new file mode 100644 index 000000000..d5549ea9e --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/raft.rs @@ -0,0 +1,318 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! [`RaftSequencerProposer`]: the production [`SequencerProposer`]. +//! +//! On the sequencer leader it appends the entry to the local sequencer +//! group. On any other node it forwards the entry to the leader over the +//! cluster transport as a `DataProposeRequest` with the `Sequencer` target, +//! the same RPC that forwards data-group proposals. The leader's handler +//! proposes it to its sequencer group. + +use std::collections::BTreeSet; +use std::sync::{Arc, Mutex, Weak}; + +use tokio::sync::Semaphore; +use tracing::debug; + +use nodedb_cluster::MultiRaft; +use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use nodedb_cluster::rpc_codec::{DataProposeRequest, ProposeTarget, RaftRpc}; + +use super::seam::{ProposeDispatch, SequencerProposeError, SequencerProposer}; +use crate::control::cluster::warm_peers::register_peers_from_topology; +use crate::control::state::SharedState; + +/// Most forward RPCs one proposer keeps in flight. A proposal past this +/// limit fails with [`SequencerProposeError::ForwardBusy`], and the +/// scheduler proposes it again on a later stall tick. +pub const MAX_INFLIGHT_SEQUENCER_FORWARDS: usize = 64; + +/// How a proposal reaches the sequencer group from this node. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) enum SequencerRoute { + /// This node leads the sequencer group. + Local, + /// Another node leads the sequencer group. + Forward { leader: u64 }, +} + +/// Pick the route from this node's view of the sequencer group. +/// +/// `leader` is the leader this node observes, `0` while unknown. A node +/// that is not leader but observes itself as leader has a stale view, so it +/// has no leader to send to. +pub(super) fn sequencer_route( + is_leader: bool, + leader: u64, + local_node: u64, +) -> Result { + if is_leader { + return Ok(SequencerRoute::Local); + } + if leader == 0 || leader == local_node { + return Err(SequencerProposeError::NoLeader); + } + Ok(SequencerRoute::Forward { leader }) +} + +/// Proposes sequencer entries locally on the sequencer leader and forwards +/// them to the leader from every other node. +pub struct RaftSequencerProposer { + node_id: u64, + multi_raft: Arc>, + /// Source of the cluster transport and topology for forwards. Weak: the + /// node's `SharedState` holds this proposer, and a strong handle back + /// would keep the state, and every file it holds open, alive after + /// shutdown. + shared: Weak, + /// One permit per forward RPC in flight. + forwards: Arc, +} + +impl RaftSequencerProposer { + pub fn new(node_id: u64, multi_raft: Arc>, shared: &Arc) -> Self { + Self { + node_id, + multi_raft, + shared: Arc::downgrade(shared), + forwards: Arc::new(Semaphore::new(MAX_INFLIGHT_SEQUENCER_FORWARDS)), + } + } + + /// Send `bytes` to `leader` on a spawned task. + /// + /// The task logs a refused or failed forward at `debug`. The scheduler + /// does not need the RPC result: it proposes the entry again until the + /// completion registry shows it applied. + fn forward( + &self, + leader: u64, + bytes: Vec, + ) -> Result { + let Some(shared) = self.shared.upgrade() else { + return Err(SequencerProposeError::ShutDown); + }; + let Some(transport) = shared.cluster_transport.as_ref() else { + return Err(SequencerProposeError::NoTransport { leader }); + }; + let permit = Arc::clone(&self.forwards) + .try_acquire_owned() + .map_err(|_| SequencerProposeError::ForwardBusy { + leader, + limit: MAX_INFLIGHT_SEQUENCER_FORWARDS, + })?; + register_peers_from_topology(&shared, transport, &BTreeSet::from([leader])); + let transport = Arc::clone(transport); + tokio::spawn(async move { + let _permit = permit; + let rpc = RaftRpc::DataProposeRequest(DataProposeRequest { + target: ProposeTarget::Sequencer, + bytes, + }); + match transport.send_rpc(leader, rpc).await { + Ok(RaftRpc::DataProposeResponse(resp)) if resp.success => {} + Ok(RaftRpc::DataProposeResponse(resp)) => debug!( + leader, + leader_hint = ?resp.leader_hint, + error = %resp.error_message, + "calvin: sequencer leader refused a forwarded entry", + ), + Ok(other) => debug!( + leader, + response = ?other, + "calvin: unexpected reply to a forwarded sequencer entry", + ), + Err(e) => debug!( + leader, + error = %e, + "calvin: forward of a sequencer entry failed", + ), + } + }); + Ok(ProposeDispatch::Forwarded { leader }) + } +} + +impl SequencerProposer for RaftSequencerProposer { + fn propose(&self, bytes: Vec) -> Result { + let leader = { + let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + let route = sequencer_route( + mr.is_group_leader(SEQUENCER_GROUP_ID), + mr.group_leader(SEQUENCER_GROUP_ID), + self.node_id, + )?; + match route { + SequencerRoute::Local => { + mr.propose_to_group(SEQUENCER_GROUP_ID, bytes)?; + return Ok(ProposeDispatch::Local); + } + SequencerRoute::Forward { leader } => leader, + } + }; + self.forward(leader, bytes) + } +} + +#[cfg(test)] +mod tests { + use std::time::{Duration, Instant}; + + use nodedb_cluster::RoutingTable; + use nodedb_raft::message::AppendEntriesRequest; + + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::wal::WalManager; + + const LOCAL_NODE: u64 = 1; + const REMOTE_LEADER: u64 = 2; + + fn shared_state(dir: &std::path::Path) -> Arc { + let wal = Arc::new(WalManager::open_for_testing(&dir.join("test.wal")).expect("wal")); + let (dispatcher, _data_sides) = Dispatcher::new(1, 64); + SharedState::new(dispatcher, wal).expect("shared state") + } + + /// A `MultiRaft` on `LOCAL_NODE` with a sequencer group of `peers`. + fn multi_raft(dir: &std::path::Path, peers: Vec) -> Arc> { + let rt = RoutingTable::uniform(1, &[LOCAL_NODE], 1); + let mut mr = MultiRaft::new(LOCAL_NODE, rt, dir.to_path_buf()); + mr.add_group(SEQUENCER_GROUP_ID, peers) + .expect("add sequencer group"); + Arc::new(Mutex::new(mr)) + } + + #[test] + fn route_is_local_on_the_leader() { + assert_eq!( + sequencer_route(true, LOCAL_NODE, LOCAL_NODE).expect("route"), + SequencerRoute::Local + ); + } + + #[test] + fn route_forwards_to_a_remote_leader() { + assert_eq!( + sequencer_route(false, REMOTE_LEADER, LOCAL_NODE).expect("route"), + SequencerRoute::Forward { + leader: REMOTE_LEADER + } + ); + } + + #[test] + fn route_has_no_target_without_a_known_leader() { + assert!(matches!( + sequencer_route(false, 0, LOCAL_NODE), + Err(SequencerProposeError::NoLeader) + )); + assert!(matches!( + sequencer_route(false, LOCAL_NODE, LOCAL_NODE), + Err(SequencerProposeError::NoLeader) + )); + } + + #[tokio::test] + async fn sequencer_leader_appends_the_entry_locally() { + let dir = tempfile::tempdir().expect("tempdir"); + let mr = multi_raft(dir.path(), vec![]); + { + let mut guard = mr.lock().unwrap_or_else(|p| p.into_inner()); + if let Some(node) = guard.groups_mut().get_mut(&SEQUENCER_GROUP_ID) { + // no-determinism: test-only forced election deadline so the single voter campaigns immediately. + node.election_deadline_override(Instant::now() - Duration::from_millis(1)); + } + for _ in 0..20 { + guard.tick().expect("tick"); + if guard.is_group_leader(SEQUENCER_GROUP_ID) { + break; + } + } + assert!(guard.is_group_leader(SEQUENCER_GROUP_ID)); + } + let before = mr + .lock() + .unwrap_or_else(|p| p.into_inner()) + .last_log_index(SEQUENCER_GROUP_ID) + .unwrap_or(0); + let shared = shared_state(dir.path()); + let proposer = RaftSequencerProposer::new(LOCAL_NODE, Arc::clone(&mr), &shared); + + let dispatch = proposer.propose(vec![9, 9]).expect("local propose"); + + assert_eq!(dispatch, ProposeDispatch::Local); + let after = mr + .lock() + .unwrap_or_else(|p| p.into_inner()) + .last_log_index(SEQUENCER_GROUP_ID) + .unwrap_or(0); + assert_eq!(after, before + 1); + } + + /// A follower that knows the remote leader takes the forward path. The + /// fixture has no cluster transport, so the forward stops there with + /// the leader named, and nothing is appended to the local log. + #[tokio::test] + async fn follower_forwards_to_the_sequencer_leader() { + let dir = tempfile::tempdir().expect("tempdir"); + let mr = multi_raft(dir.path(), vec![REMOTE_LEADER]); + let before = { + let mut guard = mr.lock().unwrap_or_else(|p| p.into_inner()); + guard + .handle_append_entries(&AppendEntriesRequest { + term: 1, + leader_id: REMOTE_LEADER, + prev_log_index: 0, + prev_log_term: 0, + entries: Vec::new(), + leader_commit: 0, + group_id: SEQUENCER_GROUP_ID, + }) + .expect("heartbeat"); + assert_eq!(guard.group_leader(SEQUENCER_GROUP_ID), REMOTE_LEADER); + guard.last_log_index(SEQUENCER_GROUP_ID).unwrap_or(0) + }; + let shared = shared_state(dir.path()); + let proposer = RaftSequencerProposer::new(LOCAL_NODE, Arc::clone(&mr), &shared); + + let result = proposer.propose(vec![9, 9]); + + assert!( + matches!( + result, + Err(SequencerProposeError::NoTransport { + leader: REMOTE_LEADER + }) + ), + "{result:?}" + ); + let after = mr + .lock() + .unwrap_or_else(|p| p.into_inner()) + .last_log_index(SEQUENCER_GROUP_ID) + .unwrap_or(0); + assert_eq!(after, before, "a follower appends nothing locally"); + } + + /// The node's state holds its proposer. The proposer must not hold the + /// state back: that cycle keeps the catalog open after shutdown, and a + /// reopen on the same path fails to take the catalog lock. + #[test] + fn a_proposer_held_by_the_state_does_not_keep_the_state_alive() { + let dir = tempfile::tempdir().expect("tempdir"); + let mr = multi_raft(dir.path(), vec![LOCAL_NODE]); + let shared = shared_state(dir.path()); + let proposer: Arc = + Arc::new(RaftSequencerProposer::new(LOCAL_NODE, mr, &shared)); + assert!(shared.calvin.sequencer_proposer.set(proposer).is_ok()); + + assert_eq!(Arc::strong_count(&shared), 1); + let weak = Arc::downgrade(&shared); + drop(shared); + assert!( + weak.upgrade().is_none(), + "the state outlived its last owner" + ); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/seam.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/seam.rs new file mode 100644 index 000000000..876579c7f --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/seam.rs @@ -0,0 +1,55 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The [`SequencerProposer`] seam and its result types. + +use nodedb_cluster::error::ClusterError; + +/// Where a sequencer proposal went. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ProposeDispatch { + /// This node leads the sequencer group and appended the entry. + Local, + /// A forward task carries the entry to the sequencer leader. + Forwarded { leader: u64 }, +} + +/// Why a sequencer proposal did not leave this node. +#[derive(Debug, thiserror::Error)] +pub enum SequencerProposeError { + /// This node sees no sequencer leader, or sees itself as leader after it + /// stepped down. + #[error("no sequencer leader is known")] + NoLeader, + /// This node does not lead the sequencer group and has no cluster + /// transport to reach the leader. + #[error("sequencer leader is node {leader}, and this node has no cluster transport")] + NoTransport { leader: u64 }, + /// The forward limit is reached. The owed-entry sweep proposes the entry + /// again on a later tick. + #[error("{limit} sequencer forwards are in flight; the entry for node {leader} waits")] + ForwardBusy { leader: u64, limit: usize }, + /// The node shut down: its state is gone. + #[error("the node is shutting down")] + ShutDown, + /// The local sequencer group refused the proposal. + #[error("sequencer propose: {0}")] + Cluster(#[from] ClusterError), +} + +impl From for crate::Error { + fn from(e: SequencerProposeError) -> Self { + crate::Error::Dispatch { + detail: e.to_string(), + } + } +} + +/// Hands encoded sequencer entries to the sequencer Raft group. +/// +/// `Ok` means the entry left this node. It does not mean the entry is +/// applied: a leader change can drop it. The scheduler learns that an entry +/// is applied from the completion registry, and proposes it again until then. +pub trait SequencerProposer: Send + Sync { + /// Propose one msgpack-encoded `SequencerEntry`. + fn propose(&self, bytes: Vec) -> Result; +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_proposer.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_proposer.rs new file mode 100644 index 000000000..6602ec96e --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_proposer.rs @@ -0,0 +1,102 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Test fixtures for the scheduler's sequencer proposals: a capturing +//! [`SequencerProposer`] and a helper that makes the fixture node lead its +//! vShard's data group. + +use std::sync::{Arc, Mutex}; +use std::time::{Duration, Instant}; + +use nodedb_cluster::calvin::SequencerEntry; + +use super::scheduler::Scheduler; +use super::sequencer_proposer::{ProposeDispatch, SequencerProposeError, SequencerProposer}; + +/// Records every proposal it receives. The first `fail_first` proposals fail +/// the way a node that does not lead the sequencer group fails. Every later +/// one succeeds. +#[derive(Default)] +pub(super) struct CapturingProposer { + fail_first: usize, + /// Every proposal, decoded, with whether it succeeded. + attempts: Mutex>, +} + +impl CapturingProposer { + /// A proposer that accepts every proposal. + pub(super) fn accepting() -> Arc { + Arc::new(Self::default()) + } + + /// A proposer that refuses the first `n` proposals. + pub(super) fn failing_first(n: usize) -> Arc { + Arc::new(Self { + fail_first: n, + attempts: Mutex::new(Vec::new()), + }) + } + + /// Number of proposals received, failed ones included. + pub(super) fn attempt_count(&self) -> usize { + self.attempts + .lock() + .unwrap_or_else(|p| p.into_inner()) + .len() + } + + /// The proposals that succeeded, in order. + pub(super) fn accepted(&self) -> Vec { + self.attempts + .lock() + .unwrap_or_else(|p| p.into_inner()) + .iter() + .filter(|(_, ok)| *ok) + .map(|(entry, _)| entry.clone()) + .collect() + } +} + +impl SequencerProposer for CapturingProposer { + fn propose(&self, bytes: Vec) -> Result { + let entry: SequencerEntry = + zerompk::from_msgpack(&bytes).expect("scheduler proposes a valid SequencerEntry"); + let mut attempts = self.attempts.lock().unwrap_or_else(|p| p.into_inner()); + let ok = attempts.len() >= self.fail_first; + attempts.push((entry, ok)); + if ok { + Ok(ProposeDispatch::Local) + } else { + Err(SequencerProposeError::NoLeader) + } + } +} + +/// Make the fixture node the elected leader of the data group that owns +/// `scheduler`'s vShard, so leader-only proposals such as the commit vote +/// run. +pub(super) fn elect_data_group_leader(scheduler: &Scheduler) { + let mut mr = scheduler + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()); + let group_id = mr + .routing() + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard(scheduler.vshard_id) + .expect("the fixture routing maps every vShard"); + if !mr.contains_group(group_id) { + mr.add_group(group_id, vec![]).expect("add data group"); + } + if let Some(node) = mr.groups_mut().get_mut(&group_id) { + // no-determinism: test-only forced election deadline so the single voter campaigns immediately. + node.election_deadline_override(Instant::now() - Duration::from_millis(1)); + } + for _ in 0..20 { + mr.tick().expect("tick"); + if mr.vshard_role_is_leader(scheduler.vshard_id) { + return; + } + } + panic!("data group did not elect the single fixture node"); +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs new file mode 100644 index 000000000..0a91ddf3f --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs @@ -0,0 +1,482 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Shared test fixtures for the Calvin scheduler driver's `core` unit tests. + +use std::collections::{BTreeSet, HashMap}; +use std::sync::{Arc, Mutex}; +use std::time::{Duration, Instant}; + +use nodedb_cluster::MultiRaft; +use nodedb_cluster::RoutingTable; +use nodedb_cluster::calvin::types::{ + EngineKeySet, EngineTag, ReadKeyIdent, ReadWriteSet, SchedulerInput, SequencedTxn, SortedVec, + TxClass, VersionedReadEntry, VersionedReadSet, +}; +use nodedb_cluster::calvin::{CalvinCompletionRegistry, SequencerStateMachine}; +use nodedb_physical::physical_plan::wire as plan_wire; +use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; +use nodedb_types::{KeyRepr, QualifiedCollection, TenantId}; +use tokio::sync::mpsc; + +use crate::bridge::dispatch::{BridgeResponse, CoreChannelDataSide, Dispatcher}; +use crate::bridge::envelope::{ + Admission, ErrorCode, ExemptReason, Payload, Priority, Request, Response, Status, +}; +use crate::control::cluster::calvin::scheduler::driver::barrier::ReadResultEvent; +use crate::control::cluster::calvin::scheduler::driver::core::scheduler::{ + Scheduler, SchedulerParams, +}; +use crate::control::cluster::calvin::scheduler::driver::core::test_proposer::CapturingProposer; +use crate::control::cluster::calvin::scheduler::driver::types::{CommitState, PendingTxn}; +use crate::control::cluster::calvin::scheduler::lock_manager::{LockManager, TxnId}; +use crate::control::cluster::calvin::scheduler::metrics::SchedulerMetrics; +use crate::control::cluster::calvin::scheduler::{NOT_YET_APPLIED_EPOCH, SchedulerConfig}; +use crate::control::shutdown::ShutdownWatch; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, Lsn, ReadConsistency, RequestId, VShardId}; +use crate::wal::WalManager; + +/// Build a minimally-wired `Scheduler` for driver-level unit tests. The Data +/// Plane is NOT started — tests exercise Control-Plane routing, guards, and +/// request dispatch only, so no core loop is needed. The returned `TempDir` +/// must be kept alive for the scheduler's lifetime (backs the WAL and Raft +/// storage). +pub(super) fn build_test_scheduler(vshard_id: u32) -> (Scheduler, tempfile::TempDir) { + let registry = CalvinCompletionRegistry::new_detached(); + let dir = tempfile::tempdir().unwrap(); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("test.wal")).unwrap()); + let (dispatcher, mut data_sides) = Dispatcher::new(1, 64); + let _data_side = data_sides + .pop() + .expect("one configured core has one data side"); + let shared = SharedState::new(dispatcher, wal).unwrap(); + + let rt = RoutingTable::uniform(1, &[1], 1); + let multi_raft = Arc::new(Mutex::new(MultiRaft::new(1, rt, dir.path().to_path_buf()))); + + let sequencer_state_machine = Arc::new(Mutex::new(SequencerStateMachine::new( + HashMap::new(), + Arc::clone(®istry), + ))); + + let (_tx, receiver) = tokio::sync::mpsc::channel(16); + let (_rr_tx, read_result_rx) = tokio::sync::mpsc::channel(16); + let (_prom_tx, promotion_rx) = tokio::sync::mpsc::unbounded_channel(); + let (verdict_tx, verdict_rx) = tokio::sync::mpsc::channel(16); + registry.register_verdict_signal_sender(vshard_id, verdict_tx); + + let lock_manager = Arc::new(Mutex::new(LockManager::new())); + + let scheduler = Scheduler::new(SchedulerParams { + vshard_id, + receiver, + shared, + multi_raft, + sequencer_proposer: CapturingProposer::accepting(), + sequencer_state_machine, + // A freshly-built scheduler has applied nothing, so its watermark is the + // not-yet-applied sentinel (matching `read_applied_recovery` for a clean + // node). Hardcoding `0` here would instead claim epoch 0 is fully applied, + // making the exactly-once gate (`AppliedGate::is_applied`) short-circuit + // every epoch-0 replay before it reaches the lock table — silently + // defeating the end-to-end drain tests below. + fully_applied_epoch: NOT_YET_APPLIED_EPOCH, + applied_tail: BTreeSet::new(), + rebuild_target_epoch: 0, + config: SchedulerConfig::default(), + metrics: SchedulerMetrics::new(), + read_result_rx, + lock_manager, + promotion_rx, + registry, + verdict_rx, + }); + (scheduler, dir) +} + +/// Same minimal scheduler fixture as [`build_test_scheduler`], sharing a +/// caller-supplied completion `registry` (so several schedulers can register +/// against it) and retaining its Data-Plane request receiver for tests that +/// must observe scheduler dispatches. +pub(super) fn build_test_scheduler_with_data_side( + vshard_id: u32, + registry: Arc, +) -> (Scheduler, tempfile::TempDir, CoreChannelDataSide) { + let dir = tempfile::tempdir().unwrap(); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("test.wal")).unwrap()); + let (dispatcher, mut data_sides) = Dispatcher::new(1, 64); + let data_side = data_sides + .pop() + .expect("one configured core has one data side"); + let shared = SharedState::new(dispatcher, wal).unwrap(); + + let rt = RoutingTable::uniform(1, &[1], 1); + let multi_raft = Arc::new(Mutex::new(MultiRaft::new(1, rt, dir.path().to_path_buf()))); + + let sequencer_state_machine = Arc::new(Mutex::new(SequencerStateMachine::new( + HashMap::new(), + Arc::clone(®istry), + ))); + + let (_tx, receiver) = tokio::sync::mpsc::channel(16); + let (_rr_tx, read_result_rx) = tokio::sync::mpsc::channel(16); + let (_prom_tx, promotion_rx) = tokio::sync::mpsc::unbounded_channel(); + let (verdict_tx, verdict_rx) = tokio::sync::mpsc::channel(16); + registry.register_verdict_signal_sender(vshard_id, verdict_tx); + + let lock_manager = Arc::new(Mutex::new(LockManager::new())); + + let scheduler = Scheduler::new(SchedulerParams { + vshard_id, + receiver, + shared, + multi_raft, + sequencer_proposer: CapturingProposer::accepting(), + sequencer_state_machine, + fully_applied_epoch: NOT_YET_APPLIED_EPOCH, + applied_tail: BTreeSet::new(), + rebuild_target_epoch: 0, + config: SchedulerConfig::default(), + metrics: SchedulerMetrics::new(), + read_result_rx, + lock_manager, + promotion_rx, + registry, + verdict_rx, + }); + (scheduler, dir, data_side) +} + +/// Build a static-write `SequencedTxn` at `(epoch, position)`. +pub(super) fn make_sequenced_txn(epoch: u64, position: u32) -> SequencedTxn { + let write_set = ReadWriteSet::new(vec![EngineKeySet::Document { + collection: "test_coll".to_string(), + surrogates: SortedVec::new(vec![1]), + }]); + let tx_class = TxClass::new_single_vshard( + ReadWriteSet::new(vec![]), + write_set, + vec![], + TenantId::new(1), + None, + VersionedReadSet::default(), + ) + .expect("valid TxClass"); + SequencedTxn { + epoch, + position, + tx_class, + epoch_system_ms: 1_700_000_000_000, + epoch_vshard_txn_count: 1, + lock_owner: None, + } +} + +/// The vShard that `"test_coll"` homes to in the default database. A +/// scheduler built on this vShard owns the reads of [`make_validate_only_txn`]. +pub(super) fn test_coll_vshard() -> u32 { + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "test_coll") + .vshard() + .as_u32() +} + +/// Build a static `SequencedTxn` at `(epoch, position)` that reaches the +/// `CalvinExecuteStatic` stage dispatch on the [`test_coll_vshard`] scheduler. +/// +/// It carries an encoded empty plan batch and one versioned read on +/// `"test_coll"`, so that scheduler stages it as a validate-only read +/// participant. Its write set locks `"test_coll"` surrogate 1, the same key +/// as [`make_sequenced_txn`]. +pub(super) fn make_validate_only_txn(epoch: u64, position: u32) -> SequencedTxn { + let write_set = ReadWriteSet::new(vec![EngineKeySet::Document { + collection: "test_coll".to_string(), + surrogates: SortedVec::new(vec![1]), + }]); + let plans = plan_wire::encode_batch(&Vec::new()).expect("encode empty plan batch"); + let versioned_reads = VersionedReadSet::new(vec![VersionedReadEntry { + engine: EngineTag::Document, + collection: "test_coll".to_string(), + key: ReadKeyIdent::Point(KeyRepr::Surrogate(1)), + read_lsn: Lsn::ZERO, + }]); + let tx_class = TxClass::new_single_vshard( + ReadWriteSet::new(vec![]), + write_set, + plans, + TenantId::new(1), + None, + versioned_reads, + ) + .expect("valid TxClass"); + SequencedTxn { + epoch, + position, + tx_class, + epoch_system_ms: 1_700_000_000_000, + epoch_vshard_txn_count: 1, + lock_owner: None, + } +} + +/// Build a `SequencedTxn` at `(epoch, position)` whose one write plan, a +/// truncate of `"test_coll"`, homes to [`test_coll_vshard`]. +/// +/// The plan carries no identity to bind, so it reaches the stage dispatch of +/// either path unchanged. Its write set locks the same key as +/// [`make_sequenced_txn`]. +pub(super) fn make_local_write_txn(epoch: u64, position: u32) -> SequencedTxn { + let write_set = ReadWriteSet::new(vec![EngineKeySet::Document { + collection: "test_coll".to_string(), + surrogates: SortedVec::new(vec![1]), + }]); + let batch = vec![PhysicalPlan::Document(DocumentOp::Truncate { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "test_coll"), + restart_identity: false, + resolved_sum_targets: Vec::new(), + declared_primary_key: None, + })]; + let plans = plan_wire::encode_batch(&batch).expect("encode one truncate plan"); + let tx_class = TxClass::new_single_vshard( + ReadWriteSet::new(vec![]), + write_set, + plans, + TenantId::new(1), + None, + VersionedReadSet::default(), + ) + .expect("valid TxClass"); + SequencedTxn { + epoch, + position, + tx_class, + epoch_system_ms: 1_700_000_000_000, + epoch_vshard_txn_count: 1, + lock_owner: None, + } +} + +/// Upper bound on filler dispatches. The fixture dispatcher caps a tenant at +/// 64 in-flight requests, so the cap is hit long before this bound. +const MAX_FILLERS: usize = 4096; + +/// How long a test waits for a request to reach the Data Plane side. +const DATA_PLANE_WAIT: Duration = Duration::from_secs(5); + +/// A read request for `tenant_id` that holds one in-flight slot until the +/// Data Plane answers it. +fn filler_request(request_id: RequestId, tenant_id: TenantId) -> Request { + Request { + request_id, + tenant_id, + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), + plan: PhysicalPlan::Document(DocumentOp::PointGet { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "filler"), + document_id: "d".into(), + surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: Vec::new(), + rls_filters: Vec::new(), + system_time: nodedb_types::SystemTimeScope::Current, + valid_at_ms: None, + }), + // no-determinism: test-only filler deadline, never Calvin WAL data. + deadline: Instant::now() + Duration::from_secs(60), + priority: Priority::Normal, + trace_id: nodedb_types::TraceId([0u8; 16]), + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: crate::event::EventSource::User, + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + admission: Admission::Exempt(ExemptReason::Read), + } +} + +/// Dispatch filler reads for `tenant_id` until the dispatcher refuses the +/// tenant at its in-flight cap. Returns the filler request ids. +/// +/// Each accepted filler is popped off the request ring at once, so the ring +/// and the weighted-fair queue stay empty. The only refusal left is the +/// per-tenant in-flight cap, and this function panics on any other refusal. +pub(super) fn fill_tenant_inflight( + shared: &SharedState, + data_side: &mut CoreChannelDataSide, + tenant_id: TenantId, +) -> Vec { + let mut fillers = Vec::new(); + let mut dispatcher = shared.dispatcher.lock().unwrap_or_else(|p| p.into_inner()); + for _ in 0..MAX_FILLERS { + let request_id = shared.next_request_id(); + match dispatcher.dispatch(filler_request(request_id, tenant_id)) { + Ok(()) => { + fillers.push(request_id); + while data_side.request_rx.try_pop().is_ok() {} + } + Err(crate::Error::DispatchCapacity { + scope: crate::DispatchCapacityScope::TenantInflight { .. }, + }) => { + assert!( + !fillers.is_empty(), + "the cap must admit at least one filler" + ); + return fillers; + } + Err(other) => panic!("unexpected filler dispatch error: {other}"), + } + } + panic!("tenant in-flight cap not reached after {MAX_FILLERS} fillers"); +} + +/// Answer one filler request on the Data Plane side and poll it back, which +/// frees one in-flight slot for its tenant. +pub(super) fn release_filler( + shared: &SharedState, + data_side: &mut CoreChannelDataSide, + request_id: RequestId, +) { + let mut response = staged_response(Status::Ok, None); + response.request_id = request_id; + data_side + .response_tx + .try_push(BridgeResponse { inner: response }) + .expect("response ring has room for one filler response"); + let polled = shared.poll_and_route_responses(); + assert!(polled >= 1, "the filler response must be polled"); +} + +/// Wait until a request whose plan satisfies `wanted` reaches the Data Plane +/// side. Returns `false` if none arrives within [`DATA_PLANE_WAIT`]. +pub(super) async fn await_data_plane_request( + data_side: &mut CoreChannelDataSide, + wanted: impl Fn(&PhysicalPlan) -> bool, +) -> bool { + let wait = async { + loop { + while let Ok(request) = data_side.request_rx.try_pop() { + if wanted(&request.inner.plan) { + return; + } + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }; + tokio::time::timeout(DATA_PLANE_WAIT, wait).await.is_ok() +} + +/// A scheduler run loop spawned on the test runtime. +/// +/// Holds every input sender so no loop channel reports closed. +pub(super) struct RunningScheduler { + shutdown: ShutdownWatch, + handle: tokio::task::JoinHandle<()>, + input_tx: mpsc::Sender, + _read_result_tx: mpsc::Sender, + _promotion_tx: mpsc::UnboundedSender>, +} + +impl RunningScheduler { + /// The sender feeding the loop's sequenced-input receiver. + pub(super) fn input_tx(&self) -> &mpsc::Sender { + &self.input_tx + } + + /// Signal shutdown and wait for the loop to exit. + pub(super) async fn stop(self) { + self.shutdown.signal(); + tokio::time::timeout(DATA_PLANE_WAIT, self.handle) + .await + .expect("scheduler loop exits after shutdown") + .expect("scheduler loop does not panic"); + } +} + +/// Spawn `scheduler`'s run loop with open input channels and a short +/// liveness tick. +pub(super) fn spawn_scheduler_loop(mut scheduler: Scheduler) -> RunningScheduler { + let (input_tx, input_rx) = mpsc::channel(16); + let (read_result_tx, read_result_rx) = mpsc::channel(16); + let (promotion_tx, promotion_rx) = mpsc::unbounded_channel(); + scheduler.receiver = input_rx; + scheduler.read_result_rx = read_result_rx; + scheduler.promotion_rx = promotion_rx; + // The loop's liveness tick fires every quarter of this interval. + scheduler.config.verdict_stall_warn_ms = 200; + let shutdown = ShutdownWatch::new(); + let receiver = shutdown.subscribe(); + let handle = tokio::spawn(scheduler.run(receiver)); + RunningScheduler { + shutdown, + handle, + input_tx, + _read_result_tx: read_result_tx, + _promotion_tx: promotion_tx, + } +} + +/// A `PendingTxn` staged and parked awaiting the cross-shard commit verdict. +pub(super) fn staged_pending(txn: SequencedTxn, txn_id: TxnId) -> PendingTxn { + PendingTxn { + txn, + lock_owner: txn_id, + // no-determinism: test-only dispatch timestamp for a fabricated PendingTxn fixture. + dispatch_time: Instant::now(), + has_primary_write: true, + has_returning: false, + change_sets: Vec::new(), + commit_state: Some(CommitState::Staged), + verdict_deadline: None, + stage_error: None, + redo_records: None, + flush_scope: crate::control::cluster::calvin::scheduler::driver::types::FlushScope::default( + ), + } +} + +/// A staged executor `Response` carrying the given status and read-set vote. +pub(super) fn staged_response(status: Status, read_set_valid: Option) -> Response { + Response { + request_id: RequestId::new(1), + status, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: None, + read_set_valid, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } +} + +/// An executor `Response` with `Status::Error` carrying `code`. +pub(super) fn error_response(code: ErrorCode) -> Response { + let mut response = staged_response(Status::Error, None); + response.error_code = Some(Box::new(code)); + response +} + +/// Close the dispatcher's Data Plane enqueue gate, as a node shutdown does. +/// Every later dispatch is refused terminally. +pub(super) fn begin_data_plane_drain(shared: &SharedState) { + shared + .dispatcher + .lock() + .unwrap_or_else(|p| p.into_inner()) + .begin_data_plane_drain(); +} + +/// A scheduler on vShard 7 with `txn_id` pending in `state`. +pub(super) fn scheduler_with_pending( + txn_id: TxnId, + state: CommitState, +) -> (Scheduler, tempfile::TempDir) { + let (mut scheduler, dir) = build_test_scheduler(7); + let mut pending = staged_pending(make_sequenced_txn(txn_id.epoch, txn_id.position), txn_id); + pending.commit_state = Some(state); + scheduler.pending.insert(txn_id, pending); + (scheduler, dir) +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/write_version_record.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/write_version_record.rs index 497cc8111..0d27cb9e8 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/write_version_record.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/write_version_record.rs @@ -2,18 +2,16 @@ //! Post-apply write-version recording for committed Calvin transactions. //! -//! A Calvin apply's committed WAL LSN is allocated only AFTER the apply -//! succeeds — the `CalvinApplied` WAL record is appended on the Control Plane -//! once the executor response returns — so the apply itself carries no -//! committed LSN and the per-core write-version index cannot be advanced in -//! place (the dispatch stamps `wal_lsn: None`). Once the scheduler has that -//! LSN it dispatches a one-way, record-only op back to the same core, which -//! funnels the transaction's locally-applied write plans through the shared -//! write-version recorder at that LSN — the same shard-local WAL-LSN space the -//! single-shard fast path and read watermarks use. - -use std::sync::atomic::Ordering; +//! A Calvin apply's committed WAL LSN is the LSN of its `TransactionRedo` +//! record, or of the `CalvinApplied` marker a transaction that wrote nothing +//! here appends. The install records the collection floors and index-value +//! versions of the record at its LSN. The per-key versions of the local write +//! plans are recorded here: the scheduler dispatches a one-way, record-only op +//! back to the same core, which funnels the plans through the shared +//! write-version recorder at that LSN — the same shard-local WAL-LSN space +//! the single-shard fast path and read watermarks use. +use super::deferred::{DispatchOutcome, DispatchStep}; use super::scheduler::Scheduler; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; use crate::types::Lsn; @@ -22,7 +20,7 @@ use nodedb_physical::physical_plan::meta::MetaOp; impl Scheduler { /// Record the per-key write versions of a just-committed Calvin - /// transaction's locally-applied write plans at its CalvinApplied WAL + /// transaction's locally-applied write plans at its committed WAL /// `applied_lsn`. /// /// Dispatches a record-only [`MetaOp::RecordCalvinWriteVersions`] op back to @@ -35,11 +33,12 @@ impl Scheduler { /// Fire-and-forget: the recorded version is not needed to complete the /// transaction, so the response is drained and discarded. A brief index-lag /// window before the record op lands is harmless — nothing enforces read-set - /// validation against these versions yet. A dropped record (decode failure, - /// no local write plan, or dispatch backpressure) simply leaves the version + /// validation against these versions yet. A record refused at capacity is + /// parked and re-sent once capacity frees, never dropped. A decode or + /// routing failure, or a terminal dispatch refusal, leaves the version /// unrecorded and never blocks the commit. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn record_calvin_write_versions( - &self, + &mut self, txn_id: TxnId, applied_lsn: Lsn, ) { @@ -88,40 +87,67 @@ impl Scheduler { let plan = PhysicalPlan::Meta(MetaOp::RecordCalvinWriteVersions { tenant_id, plans: local, - epoch, - position, }); // The committed write-LSN for this Calvin apply — recorded against // every key the plans wrote, in the same WAL-LSN space as fast-path. let request = self.build_exempt_request(request_id, tenant_id, database_id, plan, Some(applied_lsn)); - // Register so the response routes to a real receiver (not the - // unknown-request warning path), then discard it — the recording is - // one-way. - let resp_rx = self.shared.tracker.register(request_id); - let dispatch_result = match self.shared.dispatcher.lock() { - Ok(mut d) => d.dispatch(request), - Err(poisoned) => poisoned.into_inner().dispatch(request), - }; - if let Err(e) = dispatch_result { - self.shared.tracker.cancel(&request_id); - tracing::warn!( - vshard_id = self.vshard_id, - epoch, - position, - error = %e, - "calvin: write-version record dispatch failed" - ); - return; + // The request carries everything a re-send needs, so a parked record + // outlives the txn's `pending` entry. + if let DispatchOutcome::Failed(error) = + self.dispatch_sequenced(txn_id, DispatchStep::WriteVersionRecord, request) + { + self.fail_dispatch_step(txn_id, DispatchStep::WriteVersionRecord, error); } - tokio::spawn(async move { - let mut rx = resp_rx; - let _ = rx.recv().await; - }); - self.shared - .calvin_counters - .write_versions_recorded - .fetch_add(1, Ordering::Relaxed); + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use nodedb_cluster::calvin::CalvinCompletionRegistry; + use nodedb_types::TenantId; + + use super::*; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + await_data_plane_request, build_test_scheduler_with_data_side, fill_tenant_inflight, + make_validate_only_txn, release_filler, spawn_scheduler_loop, staged_pending, + test_coll_vshard, + }; + + /// Once a Data Plane response frees tenant capacity, a write-version + /// record refused at capacity reaches the Data Plane. + #[tokio::test] + async fn refused_write_version_record_reaches_data_plane_after_capacity_frees() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let txn_id = TxnId::new(21, 0); + scheduler.pending.insert( + txn_id, + staged_pending(make_validate_only_txn(21, 0), txn_id), + ); + let shared = Arc::clone(&scheduler.shared); + let fillers = fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + + scheduler.record_calvin_write_versions(txn_id, Lsn::new(42)); + let running = spawn_scheduler_loop(scheduler); + release_filler(&shared, &mut data_side, fillers[0]); + + let arrived = await_data_plane_request(&mut data_side, |plan| { + matches!( + plan, + PhysicalPlan::Meta(MetaOp::RecordCalvinWriteVersions { .. }) + ) + }) + .await; + running.stop().await; + + assert!( + arrived, + "the refused write-version record must reach the Data Plane once capacity frees" + ); } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/mod.rs index 325fa7ea5..5baf6021c 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/mod.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/mod.rs @@ -8,4 +8,7 @@ pub mod types; pub use barrier::ReadResultEvent; pub use config::SchedulerConfig; -pub use core::{CalvinReadResultProposal, Scheduler, SchedulerParams, propose_calvin_read_result}; +pub use core::{ + CalvinReadResultProposal, RaftSequencerProposer, Scheduler, SchedulerParams, SequencerProposer, + propose_calvin_read_result, +}; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs index 9af0d5bef..f786560f9 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs @@ -23,7 +23,8 @@ pub(super) struct PendingTxn { /// reservation owns the lock). Used by `on_txn_complete` to `release` the /// correct lock-manager identity. pub lock_owner: TxnId, - /// Wall-clock time at dispatch (for lock-wait latency metrics). + /// Wall-clock time of the last stage dispatch attempt (for executor + /// latency metrics). A re-send after a capacity refusal resets it. /// /// `Instant::now()` is used here for observability only; never /// influences WAL bytes. @@ -31,7 +32,7 @@ pub(super) struct PendingTxn { /// Whether this vShard's slice carries a primary user data write (a non-edge /// Document/KV/Vector/Timeseries/Columnar/Array write). Only the primary-write /// participant deposits its applied `Response` (affected-count and any - /// RETURNING rows) into `SharedState::calvin_apply_results`. The implicit-edge + /// RETURNING rows) into `CalvinLocalState::apply_results`. The implicit-edge /// cleanup participants that dual-home alongside it carry no primary write and /// so never clobber the entry the coordinator drains. pub has_primary_write: bool, @@ -47,8 +48,8 @@ pub(super) struct PendingTxn { pub change_sets: Vec, /// Commit-resolution state for a static-set Calvin txn. /// - /// `Some(CommitState::Staged)` for a static txn dispatched via the - /// validate-and-stage path: its first executor response carries the local + /// `Some(CommitState::Staged)` for a txn dispatched, or parked for re-send, + /// via the validate-and-stage path: its first executor response carries the local /// commit vote and drives a flush-or-drop before the commit tail runs. /// `None` for dependent/active txns, which apply directly. pub commit_state: Option, @@ -63,6 +64,60 @@ pub(super) struct PendingTxn { /// `Instant::now()` is used for this deadline (observability / liveness /// only; never influences WAL bytes). pub verdict_deadline: Option, + /// Error text of a stage response that was not `Ok` on this replica. + /// + /// `Some` when this replica never staged the txn. An abort verdict drops it + /// as usual. A COMMIT verdict halts the scheduler, because the txn cannot + /// apply here while its peers apply it. + pub stage_error: Option, + /// The `TransactionRedo` record appended for a committed txn's flush, + /// under its outcome-floor window. The window settles when the txn + /// completes, and holds when the scheduler halts or stops with the txn + /// still pending. + pub redo_records: Option, + /// What a committed flush names beside its redo record, derived once + /// from this vShard's slice when it stages. + pub flush_scope: FlushScope, +} + +/// What a vShard's committed flush carries: the collections its slice +/// writes, their materialized-sum targets, and the redo record. The Data +/// Plane installs the record the way every committed transaction installs. +/// +/// The collections and sum targets are derived when the slice stages, from +/// the plans the stage sends. A flush that follows a COMMIT verdict +/// therefore never decodes or routes a plan, so it cannot fail after the +/// verdict is durable. +#[derive(Debug, Clone, Default)] +pub(super) struct FlushScope { + pub collections: Vec, + pub sum_targets: Vec, + /// The encoded redo record the resolve appended. Empty until the resolve + /// answers, and empty when the slice wrote nothing on this vShard. A + /// resent flush carries the same bytes. + pub redo: Vec, + /// Flushes sent for this txn. A refused install resends the flush until + /// the count reaches its bound. + pub sends: u32, +} + +impl FlushScope { + /// The flush scope of the local plans `plans`. + pub(super) fn of_plans(plans: &[nodedb_physical::physical_plan::PhysicalPlan]) -> Self { + Self { + collections: + crate::control::wal_replication::transaction_redo::collections::written_collections( + plans, + ), + sum_targets: + crate::control::wal_replication::transaction_redo::sum_targets::redo_sum_targets( + plans, + ), + // The resolve fills these once it answers. + redo: Vec::new(), + sends: 0, + } + } } /// Commit-resolution state of a staged static Calvin transaction. @@ -82,12 +137,14 @@ pub(in crate::control::cluster::calvin::scheduler::driver) enum CommitState { /// vote, so a torn commit (one shard flushes while a peer drops) is /// impossible. AwaitingVerdict, - /// The txn committed and a `MetaOp::CalvinResolve` has been dispatched to - /// resolve its staged post-images into a replayable `RedoRecord`; awaiting - /// that response before the redo is WAL-appended and the flush dispatched. + /// The txn committed and a `MetaOp::CalvinResolve` has been dispatched, or + /// parked for re-send at dispatcher capacity, to resolve its staged + /// post-images into a replayable `RedoRecord`; awaiting that response + /// before the redo is WAL-appended and the flush dispatched. AwaitingRedoResolve, /// A flush (`committed = true`) or drop (`committed = false`) has been - /// dispatched; awaiting its response before the commit tail runs. + /// dispatched, or parked for re-send at dispatcher capacity; awaiting its + /// response before the commit tail runs. /// /// `redo_lsn` is `Some(lsn)` when a `TransactionRedo` record was appended /// for this commit's non-empty write set — `commit_apply_tail` then only diff --git a/nodedb/src/control/cluster/calvin/scheduler/lock/manager.rs b/nodedb/src/control/cluster/calvin/scheduler/lock/manager.rs deleted file mode 100644 index 258d6c328..000000000 --- a/nodedb/src/control/cluster/calvin/scheduler/lock/manager.rs +++ /dev/null @@ -1,1136 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Deterministic lock manager for the Calvin scheduler. -//! -//! # Design -//! -//! The lock manager provides a deterministic, totally-ordered lock table over -//! per-key entries keyed by [`LockKey`]. Locks come in two modes: `Exclusive` -//! (one holder, excludes all others) and `Shared` (many compatible holders). -//! The Calvin batch acquire path takes every key in a transaction's -//! `read_set ∪ write_set` as an `Exclusive` lock; single-key `Shared` locks are -//! available via [`LockManager::acquire_shared`]. -//! -//! # Determinism -//! -//! `BTreeMap` is used throughout (not `HashMap`) so that iteration order is -//! deterministic and reproducible across replicas. This is a correctness -//! requirement, not a style preference. - -use std::collections::btree_map::Entry; -use std::collections::{BTreeMap, BTreeSet, VecDeque}; - -use smallvec::smallvec; - -use super::lock_entry::{AcquireOutcome, LockEntry, LockMode}; -use super::lock_key::{LockKey, TxnId}; - -// ── LockManager ─────────────────────────────────────────────────────────────── - -/// Deterministic Calvin lock manager for one vshard. -/// -/// Manages an in-memory lock table keyed by [`LockKey`]. The table is held in -/// a `BTreeMap` so iteration is always deterministic. -/// -/// # Key sets tracked per transaction -/// -/// - `held_locks`: key sets for transactions that are a current holder on ALL -/// their keys and are actively executing (i.e. dispatched to the Data Plane). -/// - `pending_keys`: key sets for transactions that are blocked waiting for at -/// least one key. When `release` promotes a blocked txn to holder on every -/// one of its keys, the entry moves from `pending_keys` to `held_locks`. -pub struct LockManager { - /// Per-key lock entries. Uses `BTreeMap` for deterministic iteration. - /// `pub(super)` so the sibling `reap` module can scan entries for - /// lease-expired reservations without a public accessor. - pub(super) table: BTreeMap, - /// Per-transaction set of currently held keys for **dispatched** txns. - /// Used by `release` to iterate the key set without a full table scan. - /// `pub(super)` — see `table`. - pub(super) held_locks: BTreeMap>, - /// Key sets for **blocked** (not-yet-dispatched) txns. Populated when - /// `acquire` returns `Blocked`; cleared (moved to `held_locks`) when all - /// keys have been acquired on the promotion path inside `release`. - pending_keys: BTreeMap>, -} - -/// Outcome of inspecting a single key during [`LockManager::acquire_shared`]. -enum SharedGrant { - /// The shared lock was granted (key was free or already held shared). - Granted, - /// The key is held exclusively by another txn; the request was enqueued. - Blocked, -} - -/// The wound-wait decision for an exclusive requester that meets a conflict. -enum ExclusiveWait { - /// Every conflicting holder is a shared reservation and the requester is - /// older than all of them: wound (revoke) those shared holders and proceed. - Wound, - /// The requester must block: a conflicting holder is exclusive, or the - /// requester is younger than some conflicting shared holder. - Block, -} - -/// The waiters promoted off one key when its holders drained, together with the -/// action to take on the now-empty entry. -enum Promotion { - /// No waiters remained; the entry should be removed entirely. - Freed, - /// These waiters were installed as the new holders. - Promoted(Vec), -} - -impl LockManager { - /// Create an empty lock manager. - pub fn new() -> Self { - Self { - table: BTreeMap::new(), - held_locks: BTreeMap::new(), - pending_keys: BTreeMap::new(), - } - } - - /// Attempt to acquire **exclusive** locks on all keys for `txn`. - /// - /// If every key is free, already held exclusively by `txn` (promoted from - /// waiter), or held **shared solely by `txn`** (a reservation this txn placed - /// earlier, now upgraded in place to exclusive), records `txn` as sole holder - /// of each key and returns [`AcquireOutcome::Ready`]. - /// - /// Otherwise the whole key set is classified under the WOUND-WAIT discipline - /// (see [`Self::wound_or_block`]). `TxnId` order (`(epoch, position)`) is - /// the replicated total order, so "older" means a smaller id and the - /// decision is a pure function of the lock table plus the replicated ids — - /// every replica computes it identically: - /// - Any conflicting holder is **exclusive** → `txn` waits (an exclusive - /// holder is executing/applied work and is never wounded). - /// - All conflicting holders are **shared** reservations and `txn` is older - /// than every one of them → **wound** them all (revoke to plain OCC, no - /// notification) and take every key. Wounding is silent. - /// - Otherwise (`txn` younger than some conflicting shared holder) → wait. - /// - /// On the wait path `txn` is enqueued as an exclusive waiter on every - /// conflicting key and holds none; its key set is stored in `pending_keys` - /// so `release` can promote it atomically when all keys become available. - /// The whole key set is evaluated before any mutation, so acquisition stays - /// all-keys-or-none — the manager never partially wounds and then blocks. - pub fn acquire(&mut self, txn: TxnId, keys: BTreeSet) -> AcquireOutcome { - // First pass: determine whether any key is held by a *different* txn. - // A key already held exclusively by `txn` (promoted via release) or held - // shared solely by `txn` (an earlier reservation) counts as available — - // the former is a no-op re-acquire, the latter a self-upgrade to exclusive. - let all_available = keys.iter().all(|k| { - self.table.get(k).is_none_or(|entry| { - entry.held_exclusively_by(txn) || entry.held_shared_solely_by(txn) - }) - }); - - if all_available { - // Acquire all keys. For keys not yet in the table (free), insert a - // new exclusive entry. For keys already held by this txn (promoted - // waiter), leave the entry unchanged — the waiter queue is intact. - for key in &keys { - match self.table.get_mut(key) { - None => { - self.table.insert( - key.clone(), - LockEntry { - mode: LockMode::Exclusive, - holders: smallvec![txn], - waiters: VecDeque::new(), - }, - ); - } - Some(entry) => { - // A key this txn already holds shared-solely is upgraded - // to exclusive in place (holders is exactly `[txn]`, so no - // holder change and any waiter queue stays intact). A key - // already held exclusively by `txn` is left unchanged. - if entry.held_shared_solely_by(txn) { - entry.mode = LockMode::Exclusive; - } - } - } - } - // Move out of pending (if the txn was previously blocked on this - // same key set) and into held_locks. - self.pending_keys.remove(&txn); - self.held_locks.insert(txn, keys); - return AcquireOutcome::Ready; - } - - // A conflict exists. Classify the WHOLE key set before mutating so the - // wound / block decision is atomic (never partially wound then block). - match self.wound_or_block(txn, &keys) { - ExclusiveWait::Wound => { - // Take every key exclusively. A conflicting shared entry has its - // holders revoked (the wounded readers, all younger than `txn`, - // degrade to plain OCC); `txn` becomes the sole holder while any - // existing waiters remain queued behind it. - for key in &keys { - match self.table.get_mut(key) { - Some(entry) => { - entry.mode = LockMode::Exclusive; - entry.holders.clear(); - entry.holders.push(txn); - } - None => { - self.table.insert( - key.clone(), - LockEntry { - mode: LockMode::Exclusive, - holders: smallvec![txn], - waiters: VecDeque::new(), - }, - ); - } - } - } - self.pending_keys.remove(&txn); - self.held_locks.insert(txn, keys); - AcquireOutcome::Ready - } - ExclusiveWait::Block => { - // Enqueue as an exclusive waiter on every key held by a - // different txn. - for key in &keys { - if let Some(entry) = self.table.get_mut(key) { - // No conflict on this key means it is held solely by `txn` - // (a shared reservation to be upgraded, or an exclusive - // re-acquire) — leave it untouched; `txn` keeps the key and - // upgrades it once its conflicting keys are free. Same - // predicate as the `all_available` check above. - if entry.held_exclusively_by(txn) || entry.held_shared_solely_by(txn) { - continue; - } - // Real conflict on this key. If `txn` also holds it shared - // (an upgrade that must wait behind an OLDER shared holder), - // drop its own shared hold — degrading that read to plain - // OCC, never worse than today — so the key can drain to - // empty and normal promotion can grant `txn` the exclusive - // lock later. Without this, `txn` would occupy the key - // forever and its own exclusive request could never fire. - entry.holders.retain(|h| *h != txn); - if !entry.has_waiter(txn) { - entry.waiters.push_back((txn, LockMode::Exclusive)); - } - } - // Free keys: no entry exists; the txn acquires them on the - // re-acquire path after all conflicting keys are released. - } - // Store the full key set so that release can promote this txn - // atomically once all its keys become available. - self.pending_keys.insert(txn, keys); - AcquireOutcome::Blocked - } - } - } - - /// Classify the wound-wait decision for an exclusive requester `txn` over - /// `keys`, given that at least one key already conflicts. - /// - /// Pure read over the lock table: any exclusive conflict forces - /// [`ExclusiveWait::Block`] (an exclusive holder is never wounded, so a mix - /// of exclusive and shared conflicts blocks too). Otherwise all conflicting - /// holders are shared reservations, and `txn` wounds them only when it is - /// older than every one (`txn < h` for each conflicting shared holder `h`); - /// if it is younger than any, it blocks. A key held only by `txn` itself is - /// not a conflict. - fn wound_or_block(&self, txn: TxnId, keys: &BTreeSet) -> ExclusiveWait { - let mut shared_conflicts: Vec = Vec::new(); - for key in keys { - if let Some(entry) = self.table.get(key) { - match entry.mode { - LockMode::Exclusive => { - // Exclusive entries have exactly one holder; a holder - // other than `txn` is an exclusive conflict. - if !entry.holders.contains(&txn) { - return ExclusiveWait::Block; - } - } - LockMode::Shared => { - for holder in &entry.holders { - if *holder != txn { - shared_conflicts.push(*holder); - } - } - } - } - } - } - // Wound only when there is a shared conflict AND `txn` is older than - // every conflicting shared holder; otherwise block. `shared_conflicts` - // only ever holds *other* txns' shared holders (a key held shared solely - // by `txn` never reaches here — it takes the self-upgrade path in - // `acquire`), so an empty set here means every conflict was exclusive. - if !shared_conflicts.is_empty() && shared_conflicts.iter().all(|holder| txn < *holder) { - ExclusiveWait::Wound - } else { - ExclusiveWait::Block - } - } - - /// Attempt to acquire a **shared** lock on a single `key` for `txn`. - /// - /// - Key free → create a shared entry holding `txn`, return - /// [`AcquireOutcome::Ready`]. - /// - Key held shared → add `txn` to the holders, return - /// [`AcquireOutcome::Ready`]. - /// - Key held exclusively by another txn → enqueue `txn` as a shared waiter - /// (FIFO) and return [`AcquireOutcome::Blocked`]. - /// - /// A shared request that meets an exclusive holder blocks FIFO for now; - /// wound-wait priority resolution lands in a following change. - pub fn acquire_shared(&mut self, txn: TxnId, key: LockKey) -> AcquireOutcome { - // Inspect / mutate the entry via the `Entry` API (which takes the key by - // value, sidestepping a get-then-insert borrow conflict) inside a scoped - // borrow so the map-level bookkeeping below can re-borrow `self`. - let grant = match self.table.entry(key.clone()) { - Entry::Vacant(slot) => { - slot.insert(LockEntry { - mode: LockMode::Shared, - holders: smallvec![txn], - waiters: VecDeque::new(), - }); - SharedGrant::Granted - } - Entry::Occupied(mut slot) => { - let entry = slot.get_mut(); - if entry.mode == LockMode::Shared { - if !entry.holders.contains(&txn) { - entry.holders.push(txn); - } - SharedGrant::Granted - } else { - // Held exclusively by another txn: block FIFO. - if !entry.has_waiter(txn) { - entry.waiters.push_back((txn, LockMode::Shared)); - } - SharedGrant::Blocked - } - } - }; - - match grant { - SharedGrant::Granted => { - self.pending_keys.remove(&txn); - self.held_locks.entry(txn).or_default().insert(key); - AcquireOutcome::Ready - } - SharedGrant::Blocked => { - let mut pending = BTreeSet::new(); - pending.insert(key); - self.pending_keys.insert(txn, pending); - AcquireOutcome::Blocked - } - } - } - - /// Non-blocking exclusive acquire: take all `keys` for `txn` iff every one is - /// free (or already held by `txn`), returning `true`; otherwise return - /// `false` WITHOUT enqueuing a waiter or recording any pending state. - /// - /// This is the fast path's probe. Unlike [`acquire`](Self::acquire), the - /// contended (`false`) path touches NOTHING — no holder, no `pending_keys`, - /// no waiter `VecDeque` — so a caller that does not intend to block (an - /// autocommit point write that will instead route to the scheduler) never - /// leaves an orphaned waiter that a later `release` would promote to an - /// unowned holder. It also never perturbs the FIFO ordering that Calvin - /// transactions depend on. - pub fn try_acquire(&mut self, txn: TxnId, keys: BTreeSet) -> bool { - if !self.is_ready(txn, &keys) { - // Contended: leave the table, waiter queues, and pending_keys - // completely untouched. - return false; - } - // Every key is free or already held by `txn`, so `acquire` takes its - // all-available path — it inserts the holder and never enqueues. - let outcome = self.acquire(txn, keys); - debug_assert_eq!( - outcome, - AcquireOutcome::Ready, - "try_acquire: is_ready was true but acquire returned Blocked" - ); - true - } - - /// Release all locks held by `txn`. - /// - /// `txn` is removed from every entry's holder set. When an entry's holders - /// drain to empty, its FIFO waiters are promoted mode-aware: a leading run - /// of shared waiters is promoted together, or a single leading exclusive - /// waiter is promoted alone. A waiter that becomes holder on ALL its - /// pending keys is moved from `pending_keys` to `held_locks` immediately. - /// - /// Returns the set of `TxnId`s that have been fully promoted (i.e. moved - /// into `held_locks`). The caller may use this list to dispatch those - /// transactions. - pub fn release(&mut self, txn: TxnId) -> Vec { - let held = match self.held_locks.remove(&txn) { - Some(h) => h, - None => return Vec::new(), - }; - - let mut newly_promoted: BTreeSet = BTreeSet::new(); - - for key in &held { - // Drop `txn` from this key's holders. If other (shared) holders - // remain, the key stays held and there is nothing to promote. - let now_empty = match self.table.get_mut(key) { - Some(entry) => { - entry.holders.retain(|h| *h != txn); - entry.holders.is_empty() - } - None => continue, - }; - if now_empty { - self.promote_waiters(key, &mut newly_promoted); - } - } - - newly_promoted.into_iter().collect() - } - - /// Promote the front of `key`'s waiter queue after its holders drained. - /// - /// A leading run of shared waiters is granted together; a single leading - /// exclusive waiter is granted alone; an empty queue frees the entry. Any - /// promoted txn that is now holder on all of its pending keys is moved into - /// `held_locks` and recorded in `newly_promoted`. - fn promote_waiters(&mut self, key: &LockKey, newly_promoted: &mut BTreeSet) { - // Decide the promotion inside a scoped borrow so the readiness sweep - // below can re-borrow the table. - let decision = match self.table.get_mut(key) { - Some(entry) => match entry.waiters.front().map(|(_, mode)| *mode) { - None => Promotion::Freed, - Some(LockMode::Exclusive) => match entry.waiters.pop_front() { - Some((next, _)) => { - entry.mode = LockMode::Exclusive; - entry.holders.clear(); - entry.holders.push(next); - Promotion::Promoted(vec![next]) - } - None => Promotion::Freed, - }, - Some(LockMode::Shared) => { - entry.mode = LockMode::Shared; - entry.holders.clear(); - let mut promoted = Vec::new(); - while matches!(entry.waiters.front(), Some((_, LockMode::Shared))) { - if let Some((next, _)) = entry.waiters.pop_front() { - entry.holders.push(next); - promoted.push(next); - } - } - Promotion::Promoted(promoted) - } - }, - None => return, - }; - - let promoted = match decision { - Promotion::Freed => { - self.table.remove(key); - return; - } - Promotion::Promoted(promoted) => promoted, - }; - - // For each promoted txn, check whether it is now holder on ALL of its - // pending keys. If so, it is fully ready — move to held_locks. Remove - // first (rather than `get` + a follow-up `remove`) so there is no - // unwrap/expect on a "just confirmed Some" invariant: the owned - // `pending` set is reinserted on the not-yet-ready path. - for next in promoted { - if let Some(pending) = self.pending_keys.remove(&next) { - let all_held = pending - .iter() - .all(|k| self.table.get(k).is_none_or(|e| e.holders.contains(&next))); - if all_held { - self.held_locks.insert(next, pending); - newly_promoted.insert(next); - } else { - self.pending_keys.insert(next, pending); - } - } - } - } - - /// Check whether a previously-blocked transaction is now ready. - /// - /// A transaction is ready when for every key in its key set, the key is - /// either: - /// - Not present in the lock table (free), or - /// - Present in the lock table with `txn` among the current holders - /// (shared or exclusive). - /// - /// This is called after `release` returns `txn_id` in the unblocked set. - /// If `is_ready` returns `true`, the caller calls `acquire` again which - /// will succeed on the all-available path (because the waiter was promoted). - pub fn is_ready(&self, txn: TxnId, keys: &BTreeSet) -> bool { - keys.iter().all(|key| { - match self.table.get(key) { - None => true, // key is free - Some(entry) => entry.holders.contains(&txn), // txn is a current holder - } - }) - } - - /// Number of holders of `key` that hold it as a Calvin read reservation - /// (a `TxnId` in the reservation position band), or 0 when the key is - /// unlocked or held only by non-reservation transactions. Used to observe - /// reservation install/release from outside the scheduler. - pub fn reservation_holder_count(&self, key: &LockKey) -> usize { - self.table - .get(key) - .map(|e| e.holders.iter().filter(|h| h.is_reservation()).count()) - .unwrap_or(0) - } - - /// Number of currently-held locks (entries in the lock table). - #[cfg(test)] - pub fn lock_count(&self) -> usize { - self.table.len() - } - - /// Number of transactions currently holding at least one lock. - #[cfg(test)] - pub fn holder_count(&self) -> usize { - self.held_locks.len() - } -} - -impl LockEntry { - /// Whether this entry is held exclusively by exactly `txn` (the self - /// re-acquire case on the exclusive path). - fn held_exclusively_by(&self, txn: TxnId) -> bool { - self.mode == LockMode::Exclusive && self.holders.len() == 1 && self.holders[0] == txn - } - - /// Whether this entry is held **shared** by exactly `txn` and no one else — - /// the self-upgrade case: `txn` may take the key exclusively because it is - /// the sole current holder. - fn held_shared_solely_by(&self, txn: TxnId) -> bool { - self.mode == LockMode::Shared && self.holders.len() == 1 && self.holders[0] == txn - } - - /// Whether `txn` is already enqueued as a waiter on this entry. - fn has_waiter(&self, txn: TxnId) -> bool { - self.waiters.iter().any(|(w, _)| *w == txn) - } -} - -impl Default for LockManager { - fn default() -> Self { - Self::new() - } -} - -// ── Tests ───────────────────────────────────────────────────────────────────── - -#[cfg(test)] -mod tests { - use super::*; - use std::sync::Arc; - - fn key(name: &str) -> LockKey { - LockKey::Surrogate { - collection: Arc::from(name), - surrogate: 1, - } - } - - fn keyset(names: &[&str]) -> BTreeSet { - names.iter().map(|n| key(n)).collect() - } - - fn txn(epoch: u64, pos: u32) -> TxnId { - TxnId::new(epoch, pos) - } - - #[test] - fn acquire_free_keys_returns_ready() { - let mut lm = LockManager::new(); - let t = txn(1, 0); - let outcome = lm.acquire(t, keyset(&["a", "b"])); - assert_eq!(outcome, AcquireOutcome::Ready); - assert_eq!(lm.lock_count(), 2); - } - - #[test] - fn acquire_held_key_returns_blocked_and_enqueues_waiter() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - let t2 = txn(1, 1); - lm.acquire(t1, keyset(&["x"])); - - let outcome = lm.acquire(t2, keyset(&["x"])); - assert_eq!(outcome, AcquireOutcome::Blocked); - - // t2 should be in the waiter queue for "x". - assert!(lm.table.get(&key("x")).unwrap().has_waiter(t2)); - } - - #[test] - fn release_returns_unblocked_waiter_ids() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - let t2 = txn(1, 1); - lm.acquire(t1, keyset(&["x"])); - lm.acquire(t2, keyset(&["x"])); - - let unblocked = lm.release(t1); - assert!(unblocked.contains(&t2)); - } - - #[test] - fn autocommit_holder_release_promotes_and_returns_scheduler_waiter() { - // Mirrors the write-admission fast path: an autocommit-band holder takes - // an uncontended key, a normal-band scheduler txn then blocks behind it, - // and the holder's release promotes that scheduler txn AND returns its id - // — the value the fast-path guard forwards to the scheduler on drop - // (previously discarded, stranding the promoted txn as a zombie holder). - let mut lm = LockManager::new(); - let autocommit = txn(TxnId::AUTOCOMMIT_EPOCH, 0); - let scheduler_txn = txn(9, 0); - - assert!( - lm.try_acquire(autocommit, keyset(&["k"])), - "the fast-path holder takes the uncontended key" - ); - assert_eq!( - lm.acquire(scheduler_txn, keyset(&["k"])), - AcquireOutcome::Blocked, - "the scheduler txn queues behind the fast-path holder" - ); - - let promoted = lm.release(autocommit); - assert_eq!( - promoted, - vec![scheduler_txn], - "release must return the promoted scheduler waiter" - ); - assert!( - lm.is_ready(scheduler_txn, &keyset(&["k"])), - "the promoted scheduler txn is now holder of the freed key" - ); - } - - #[test] - fn release_preserves_fifo_waiter_order() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - let t2 = txn(1, 1); - let t3 = txn(1, 2); - lm.acquire(t1, keyset(&["x"])); - lm.acquire(t2, keyset(&["x"])); - lm.acquire(t3, keyset(&["x"])); - - // Release t1 — t2 should become holder (FIFO). - lm.release(t1); - let holder = lm.table.get(&key("x")).unwrap().holders[0]; - assert_eq!(holder, t2); - - // Release t2 — t3 should become holder. - lm.release(t2); - let holder = lm.table.get(&key("x")).unwrap().holders[0]; - assert_eq!(holder, t3); - } - - #[test] - fn multi_key_txn_releases_all_atomically() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - lm.acquire(t1, keyset(&["a", "b", "c"])); - assert_eq!(lm.lock_count(), 3); - - lm.release(t1); - assert_eq!(lm.lock_count(), 0); - assert_eq!(lm.holder_count(), 0); - } - - #[test] - fn is_ready_returns_true_when_all_keys_free_or_self_at_front() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - let t2 = txn(1, 1); - lm.acquire(t1, keyset(&["x", "y"])); - lm.acquire(t2, keyset(&["x", "y"])); - - // t2 is not ready while t1 holds. - assert!(!lm.is_ready(t2, &keyset(&["x", "y"]))); - - // Release t1 — t2 becomes holder on both keys. - lm.release(t1); - // After release, t2 is promoted to holder on both keys. - assert!(lm.is_ready(t2, &keyset(&["x", "y"]))); - } - - #[test] - fn shared_shared_compatible() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - let t2 = txn(1, 1); - - assert_eq!(lm.acquire_shared(t1, key("s")), AcquireOutcome::Ready); - assert_eq!(lm.acquire_shared(t2, key("s")), AcquireOutcome::Ready); - - let entry = lm.table.get(&key("s")).unwrap(); - assert_eq!(entry.mode, LockMode::Shared); - assert!(entry.holders.contains(&t1)); - assert!(entry.holders.contains(&t2)); - } - - #[test] - fn shared_blocks_exclusive() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - let t2 = txn(1, 1); - - assert_eq!(lm.acquire_shared(t1, key("k")), AcquireOutcome::Ready); - assert_eq!( - lm.acquire(t2, keyset(&["k"])), - AcquireOutcome::Blocked, - "an exclusive request must block behind a shared holder" - ); - assert!(lm.table.get(&key("k")).unwrap().has_waiter(t2)); - } - - #[test] - fn exclusive_blocks_shared() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - let t2 = txn(1, 1); - - assert_eq!(lm.acquire(t1, keyset(&["k"])), AcquireOutcome::Ready); - assert_eq!( - lm.acquire_shared(t2, key("k")), - AcquireOutcome::Blocked, - "a shared request must block behind an exclusive holder" - ); - assert!(lm.table.get(&key("k")).unwrap().has_waiter(t2)); - } - - #[test] - fn release_promotes_shared_run_together() { - let mut lm = LockManager::new(); - let holder = txn(1, 0); - let s1 = txn(2, 0); - let s2 = txn(2, 1); - - // Exclusive holder, two shared waiters queued behind it. - assert_eq!(lm.acquire(holder, keyset(&["k"])), AcquireOutcome::Ready); - assert_eq!(lm.acquire_shared(s1, key("k")), AcquireOutcome::Blocked); - assert_eq!(lm.acquire_shared(s2, key("k")), AcquireOutcome::Blocked); - - // Releasing the exclusive holder promotes the whole run of shared - // waiters together. - let promoted = lm.release(holder); - assert!(promoted.contains(&s1)); - assert!(promoted.contains(&s2)); - - let entry = lm.table.get(&key("k")).unwrap(); - assert_eq!(entry.mode, LockMode::Shared); - assert!(entry.holders.contains(&s1)); - assert!(entry.holders.contains(&s2)); - } - - #[test] - fn release_promotes_single_exclusive_waiter() { - let mut lm = LockManager::new(); - let holder = txn(1, 0); - let x1 = txn(2, 0); - let x2 = txn(2, 1); - - assert_eq!(lm.acquire(holder, keyset(&["k"])), AcquireOutcome::Ready); - assert_eq!(lm.acquire(x1, keyset(&["k"])), AcquireOutcome::Blocked); - assert_eq!(lm.acquire(x2, keyset(&["k"])), AcquireOutcome::Blocked); - - // Only the single leading exclusive waiter is promoted. - let promoted = lm.release(holder); - assert_eq!(promoted, vec![x1]); - - let entry = lm.table.get(&key("k")).unwrap(); - assert_eq!(entry.mode, LockMode::Exclusive); - assert_eq!(entry.holders.len(), 1); - assert_eq!(entry.holders[0], x1); - // x2 is still waiting behind x1. - assert!(entry.has_waiter(x2)); - } - - #[test] - fn multi_holder_release() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - let t2 = txn(1, 1); - - assert_eq!(lm.acquire_shared(t1, key("k")), AcquireOutcome::Ready); - assert_eq!(lm.acquire_shared(t2, key("k")), AcquireOutcome::Ready); - assert_eq!(lm.lock_count(), 1); - - // Releasing one shared holder leaves the other holding the key. - lm.release(t1); - let entry = lm.table.get(&key("k")).unwrap(); - assert!(!entry.holders.contains(&t1)); - assert!(entry.holders.contains(&t2)); - assert_eq!(lm.lock_count(), 1); - - // Releasing the last shared holder frees the key. - lm.release(t2); - assert_eq!(lm.lock_count(), 0); - } - - #[test] - fn older_writer_wounds_shared() { - let mut lm = LockManager::new(); - let t2 = txn(1, 2); // shared holder - let t1 = txn(1, 1); // exclusive requester, older than t2 - - assert_eq!(lm.acquire_shared(t2, key("k")), AcquireOutcome::Ready); - // The older writer wounds the younger shared holder and proceeds. - assert_eq!(lm.acquire(t1, keyset(&["k"])), AcquireOutcome::Ready); - - let entry = lm.table.get(&key("k")).unwrap(); - assert_eq!(entry.mode, LockMode::Exclusive); - assert!(entry.holders.contains(&t1), "R is now the exclusive holder"); - assert!( - !entry.holders.contains(&t2), - "the wounded shared holder is gone" - ); - } - - #[test] - fn younger_writer_waits() { - let mut lm = LockManager::new(); - let t1 = txn(1, 1); // shared holder - let t2 = txn(1, 2); // exclusive requester, younger than t1 - - assert_eq!(lm.acquire_shared(t1, key("k")), AcquireOutcome::Ready); - // The younger writer must not wound; it waits behind the shared holder. - assert_eq!(lm.acquire(t2, keyset(&["k"])), AcquireOutcome::Blocked); - - let entry = lm.table.get(&key("k")).unwrap(); - assert!(entry.holders.contains(&t1), "the shared holder still holds"); - assert!(!entry.holders.contains(&t2), "R holds nothing"); - assert!(entry.has_waiter(t2), "R is enqueued as an exclusive waiter"); - } - - #[test] - fn exclusive_waits_on_exclusive() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - let t2 = txn(1, 1); - - assert_eq!(lm.acquire(t1, keyset(&["k"])), AcquireOutcome::Ready); - assert_eq!(lm.acquire(t2, keyset(&["k"])), AcquireOutcome::Blocked); - assert!(lm.table.get(&key("k")).unwrap().has_waiter(t2)); - } - - #[test] - fn exclusive_waits_on_exclusive_regardless_of_age() { - let mut lm = LockManager::new(); - let t2 = txn(1, 2); // exclusive holder (younger) - let t1 = txn(1, 1); // exclusive requester (older) - - assert_eq!(lm.acquire(t2, keyset(&["k"])), AcquireOutcome::Ready); - // An exclusive holder is NEVER wounded, even by an older writer. - assert_eq!(lm.acquire(t1, keyset(&["k"])), AcquireOutcome::Blocked); - - let entry = lm.table.get(&key("k")).unwrap(); - assert!( - entry.holders.contains(&t2), - "the exclusive holder is intact" - ); - assert!(!entry.holders.contains(&t1)); - assert!(entry.has_waiter(t1)); - } - - #[test] - fn multi_key_atomic_wound_takes_both() { - let mut lm = LockManager::new(); - let s1 = txn(1, 5); // shared holder on k1, younger than R - let s2 = txn(1, 6); // shared holder on k2, younger than R - let r = txn(1, 1); // exclusive requester, older than both - - assert_eq!(lm.acquire_shared(s1, key("k1")), AcquireOutcome::Ready); - assert_eq!(lm.acquire_shared(s2, key("k2")), AcquireOutcome::Ready); - - assert_eq!(lm.acquire(r, keyset(&["k1", "k2"])), AcquireOutcome::Ready); - - for k in ["k1", "k2"] { - let entry = lm.table.get(&key(k)).unwrap(); - assert_eq!(entry.mode, LockMode::Exclusive); - assert!(entry.holders.contains(&r), "R holds {k}"); - } - assert!(!lm.table.get(&key("k1")).unwrap().holders.contains(&s1)); - assert!(!lm.table.get(&key("k2")).unwrap().holders.contains(&s2)); - } - - #[test] - fn multi_key_atomic_wait_holds_none() { - let mut lm = LockManager::new(); - let s1 = txn(1, 5); // shared holder on k1, younger than R - let s2 = txn(1, 0); // shared holder on k2, OLDER than R - let r = txn(1, 1); // exclusive requester - - assert_eq!(lm.acquire_shared(s1, key("k1")), AcquireOutcome::Ready); - assert_eq!(lm.acquire_shared(s2, key("k2")), AcquireOutcome::Ready); - - // R is younger than the holder on k2, so it must wait on BOTH keys and - // hold neither (all-or-nothing). - assert_eq!( - lm.acquire(r, keyset(&["k1", "k2"])), - AcquireOutcome::Blocked - ); - - assert!( - !lm.table.get(&key("k1")).unwrap().holders.contains(&r), - "R holds no key" - ); - assert!(!lm.table.get(&key("k2")).unwrap().holders.contains(&r)); - // The older shared holder on k2 is untouched. - assert!(lm.table.get(&key("k2")).unwrap().holders.contains(&s2)); - } - - #[test] - fn crossed_reservations_are_acyclic() { - // T1 holds shared K1 and wants exclusive K2; T2 holds shared K2 and - // wants exclusive K1. The older writer's exclusive acquire wounds the - // younger's shared holding, breaking the cycle — no deadlock. - let mut lm = LockManager::new(); - let t1 = txn(1, 1); // older - let t2 = txn(1, 2); // younger - - assert_eq!(lm.acquire_shared(t1, key("k1")), AcquireOutcome::Ready); - assert_eq!(lm.acquire_shared(t2, key("k2")), AcquireOutcome::Ready); - - // T1 (older) acquires exclusive K2: wounds T2's shared holding and - // proceeds. - assert_eq!(lm.acquire(t1, keyset(&["k2"])), AcquireOutcome::Ready); - - let k2 = lm.table.get(&key("k2")).unwrap(); - assert_eq!(k2.mode, LockMode::Exclusive); - assert!(k2.holders.contains(&t1), "the older writer proceeds"); - assert!( - !k2.holders.contains(&t2), - "the younger's reservation is wounded away" - ); - } - - #[test] - fn shared_reservation_self_upgrades_to_exclusive() { - let mut lm = LockManager::new(); - let t = txn(1, 0); - - assert_eq!(lm.acquire_shared(t, key("k")), AcquireOutcome::Ready); - // The txn re-acquires its own shared reservation exclusively — this must - // NOT self-deadlock by blocking on its own held key. - assert_eq!( - lm.acquire(t, keyset(&["k"])), - AcquireOutcome::Ready, - "self-upgrade from shared to exclusive must not block" - ); - - let entry = lm.table.get(&key("k")).unwrap(); - assert_eq!(entry.mode, LockMode::Exclusive); - assert_eq!(entry.holders.len(), 1); - assert_eq!(entry.holders[0], t); - } - - #[test] - fn self_upgrade_with_other_shared_holder_blocks_or_wounds() { - // T_old is older than T_young: T_old's self-upgrade must wound T_young. - let mut lm = LockManager::new(); - let t_old = txn(1, 0); - let t_young = txn(1, 1); - - assert_eq!(lm.acquire_shared(t_old, key("k")), AcquireOutcome::Ready); - assert_eq!(lm.acquire_shared(t_young, key("k")), AcquireOutcome::Ready); - - assert_eq!( - lm.acquire(t_old, keyset(&["k"])), - AcquireOutcome::Ready, - "the older self-upgrader wounds the younger shared holder" - ); - let entry = lm.table.get(&key("k")).unwrap(); - assert_eq!(entry.mode, LockMode::Exclusive); - assert_eq!(entry.holders.len(), 1); - assert_eq!(entry.holders[0], t_old); - - // Symmetric case: the YOUNGER of the two self-upgrades and must block. - let mut lm = LockManager::new(); - let t_old = txn(1, 0); - let t_young = txn(1, 1); - - assert_eq!(lm.acquire_shared(t_old, key("k")), AcquireOutcome::Ready); - assert_eq!(lm.acquire_shared(t_young, key("k")), AcquireOutcome::Ready); - - assert_eq!( - lm.acquire(t_young, keyset(&["k"])), - AcquireOutcome::Blocked, - "the younger self-upgrader must wait behind the older shared holder" - ); - // t_young drops its own shared hold (degrading to plain OCC) so the key - // can drain to empty and its exclusive request can later be promoted; - // t_old remains the sole shared holder, and t_young is enqueued as an - // exclusive waiter rather than left stuck as a non-waiting holder. - let entry = lm.table.get(&key("k")).unwrap(); - assert_eq!(entry.mode, LockMode::Shared); - assert!(entry.holders.contains(&t_old)); - assert!(!entry.holders.contains(&t_young)); - assert!(entry.has_waiter(t_young)); - - // Once t_old releases, t_young is promoted to sole exclusive holder. - let unblocked = lm.release(t_old); - assert!(unblocked.contains(&t_young)); - let entry = lm.table.get(&key("k")).unwrap(); - assert_eq!(entry.mode, LockMode::Exclusive); - assert_eq!(entry.holders.len(), 1); - assert_eq!(entry.holders[0], t_young); - } - - #[test] - fn self_upgrade_mixed_with_conflict_on_other_key() { - let mut lm = LockManager::new(); - let t = txn(1, 0); - let u = txn(1, 1); - - // T reserves K1 shared; U holds K2 exclusively. - assert_eq!(lm.acquire_shared(t, key("k1")), AcquireOutcome::Ready); - assert_eq!(lm.acquire(u, keyset(&["k2"])), AcquireOutcome::Ready); - - // T tries to take both keys exclusively: K2 conflicts with U, so T must - // block on the whole set — and critically must NOT self-deadlock on K1. - assert_eq!( - lm.acquire(t, keyset(&["k1", "k2"])), - AcquireOutcome::Blocked, - "conflict on k2 blocks the whole set" - ); - - // After U releases K2, T's re-acquire succeeds and upgrades K1 in place. - lm.release(u); - assert_eq!( - lm.acquire(t, keyset(&["k1", "k2"])), - AcquireOutcome::Ready, - "once k2 frees up, t acquires both keys exclusively" - ); - let k1 = lm.table.get(&key("k1")).unwrap(); - assert_eq!(k1.mode, LockMode::Exclusive); - assert_eq!(k1.holders.len(), 1); - assert_eq!(k1.holders[0], t); - let k2 = lm.table.get(&key("k2")).unwrap(); - assert_eq!(k2.mode, LockMode::Exclusive); - assert_eq!(k2.holders.len(), 1); - assert_eq!(k2.holders[0], t); - } - - #[test] - fn two_non_conflicting_both_dispatch_immediately() { - let mut lm = LockManager::new(); - - let txn1 = TxnId::new(1, 0); - let txn2 = TxnId::new(1, 1); - - let keys1: BTreeSet = [LockKey::Surrogate { - collection: Arc::from("coll"), - surrogate: 1, - }] - .into(); - let keys2: BTreeSet = [LockKey::Surrogate { - collection: Arc::from("coll"), - surrogate: 2, - }] - .into(); - - let o1 = lm.acquire(txn1, keys1); - let o2 = lm.acquire(txn2, keys2); - - assert_eq!(o1, AcquireOutcome::Ready, "txn1 should be ready"); - assert_eq!( - o2, - AcquireOutcome::Ready, - "txn2 should be ready (disjoint keys)" - ); - } - - #[test] - fn two_conflicting_second_dispatches_after_first_completes() { - let mut lm = LockManager::new(); - - let txn1 = TxnId::new(1, 0); - let txn2 = TxnId::new(1, 1); - let shared_key: BTreeSet = [LockKey::Surrogate { - collection: Arc::from("coll"), - surrogate: 42, - }] - .into(); - - let o1 = lm.acquire(txn1, shared_key.clone()); - assert_eq!(o1, AcquireOutcome::Ready); - - let o2 = lm.acquire(txn2, shared_key.clone()); - assert_eq!(o2, AcquireOutcome::Blocked); - - let unblocked = lm.release(txn1); - assert!(unblocked.contains(&txn2)); - - assert!(lm.is_ready(txn2, &shared_key)); - } - - #[test] - fn many_mixed_deterministic_dispatch_order() { - let mut lm = LockManager::new(); - let mut dispatched: Vec = Vec::new(); - - let pairs = [(2, 0), (1, 1), (3, 0), (1, 0), (2, 1)]; - for (epoch, pos) in pairs { - let tid = TxnId::new(epoch, pos); - let keys: BTreeSet = [LockKey::Surrogate { - collection: Arc::from(format!("c_{epoch}_{pos}")), - surrogate: epoch as u32 * 10 + pos, - }] - .into(); - let outcome = lm.acquire(tid, keys); - if outcome == AcquireOutcome::Ready { - dispatched.push(tid); - } - } - - assert_eq!( - dispatched.len(), - 5, - "all non-conflicting txns should be ready" - ); - - let mut expected = pairs.map(|(e, p)| TxnId::new(e, p)).to_vec(); - expected.sort(); - let mut sorted_dispatched = dispatched.clone(); - sorted_dispatched.sort(); - assert_eq!(sorted_dispatched, expected); - } - - #[test] - fn cross_epoch_raw_blocks_correctly() { - let mut lm = LockManager::new(); - - let txn_n = TxnId::new(1, 0); - let txn_n1 = TxnId::new(2, 0); - - let key_k: BTreeSet = [LockKey::Surrogate { - collection: Arc::from("orders"), - surrogate: 100, - }] - .into(); - - let o1 = lm.acquire(txn_n, key_k.clone()); - assert_eq!(o1, AcquireOutcome::Ready); - - let o2 = lm.acquire(txn_n1, key_k.clone()); - assert_eq!(o2, AcquireOutcome::Blocked); - - let unblocked = lm.release(txn_n); - assert!(unblocked.contains(&txn_n1)); - assert!(lm.is_ready(txn_n1, &key_k)); - } -} diff --git a/nodedb/src/control/cluster/calvin/scheduler/lock/manager/acquire.rs b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/acquire.rs new file mode 100644 index 000000000..f896c15c9 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/acquire.rs @@ -0,0 +1,378 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Exclusive lock acquisition and waiter queueing. + +use std::collections::{BTreeSet, VecDeque}; + +use smallvec::smallvec; + +use crate::control::cluster::calvin::scheduler::lock::lock_entry::{ + AcquireOutcome, LockEntry, LockMode, +}; +use crate::control::cluster::calvin::scheduler::lock::lock_key::{LockKey, TxnId}; + +use super::types::{ExclusiveWait, LockManager}; + +impl LockManager { + /// Attempt to acquire **exclusive** locks on all keys for `txn`. + /// + /// If every key is free, already held exclusively by `txn` (promoted from + /// waiter), or held **shared solely by `txn`** (a reservation this txn placed + /// earlier, now upgraded in place to exclusive), records `txn` as sole holder + /// of each key and returns [`AcquireOutcome::Ready`]. + /// + /// Otherwise the whole key set is classified under the WOUND-WAIT discipline + /// (see [`Self::wound_or_block`]). `TxnId` order (`(epoch, position)`) is + /// the replicated total order, so "older" means a smaller id and the + /// decision is a pure function of the lock table plus the replicated ids — + /// every replica computes it identically: + /// - Any conflicting holder is **exclusive** → `txn` waits (an exclusive + /// holder is executing/applied work and is never wounded). + /// - All conflicting holders are **shared** reservations and `txn` is older + /// than every one of them → **wound** them all (revoke to plain OCC, no + /// notification) and take every key. Wounding is silent. + /// - Otherwise (`txn` younger than some conflicting shared holder) → wait. + /// + /// On the wait path `txn` is enqueued as an exclusive waiter on every + /// conflicting key and holds none; its key set is stored in `pending_keys` + /// so `release` can promote it atomically when all keys become available. + /// The whole key set is evaluated before any mutation, so acquisition stays + /// all-keys-or-none — the manager never partially wounds and then blocks. + pub fn acquire(&mut self, txn: TxnId, keys: BTreeSet) -> AcquireOutcome { + // First pass: determine whether any key is held by a *different* txn. + // A key already held exclusively by `txn` (promoted via release) or held + // shared solely by `txn` (an earlier reservation) counts as available — + // the former is a no-op re-acquire, the latter a self-upgrade to exclusive. + let all_available = keys.iter().all(|k| { + self.table.get(k).is_none_or(|entry| { + entry.held_exclusively_by(txn) || entry.held_shared_solely_by(txn) + }) + }); + + if all_available { + // Acquire all keys. For keys not yet in the table (free), insert a + // new exclusive entry. For keys already held by this txn (promoted + // waiter), leave the entry unchanged — the waiter queue is intact. + for key in &keys { + match self.table.get_mut(key) { + None => { + self.table.insert( + key.clone(), + LockEntry { + mode: LockMode::Exclusive, + holders: smallvec![txn], + waiters: VecDeque::new(), + }, + ); + } + Some(entry) => { + // A key this txn already holds shared-solely is upgraded + // to exclusive in place (holders is exactly `[txn]`, so no + // holder change and any waiter queue stays intact). A key + // already held exclusively by `txn` is left unchanged. + if entry.held_shared_solely_by(txn) { + entry.mode = LockMode::Exclusive; + } + } + } + } + // Move out of pending (if the txn was previously blocked on this + // same key set) and into held_locks. + self.pending_keys.remove(&txn); + self.held_locks.insert(txn, keys); + return AcquireOutcome::Ready; + } + + // A conflict exists. Classify the WHOLE key set before mutating so the + // wound / block decision is atomic (never partially wound then block). + match self.wound_or_block(txn, &keys) { + ExclusiveWait::Wound => { + // Take every key exclusively. A conflicting shared entry has its + // holders revoked (the wounded readers, all younger than `txn`, + // degrade to plain OCC); `txn` becomes the sole holder while any + // existing waiters remain queued behind it. + for key in &keys { + match self.table.get_mut(key) { + Some(entry) => { + entry.mode = LockMode::Exclusive; + entry.holders.clear(); + entry.holders.push(txn); + } + None => { + self.table.insert( + key.clone(), + LockEntry { + mode: LockMode::Exclusive, + holders: smallvec![txn], + waiters: VecDeque::new(), + }, + ); + } + } + } + self.pending_keys.remove(&txn); + self.held_locks.insert(txn, keys); + AcquireOutcome::Ready + } + ExclusiveWait::Block => { + // Enqueue as an exclusive waiter on every key held by a + // different txn. + for key in &keys { + if let Some(entry) = self.table.get_mut(key) { + // No conflict on this key means it is held solely by `txn` + // (a shared reservation to be upgraded, or an exclusive + // re-acquire) — leave it untouched; `txn` keeps the key and + // upgrades it once its conflicting keys are free. Same + // predicate as the `all_available` check above. + if entry.held_exclusively_by(txn) || entry.held_shared_solely_by(txn) { + continue; + } + // Real conflict on this key. If `txn` also holds it shared + // (an upgrade that must wait behind an OLDER shared holder), + // drop its own shared hold — degrading that read to plain + // OCC, never worse than today — so the key can drain to + // empty and normal promotion can grant `txn` the exclusive + // lock later. Without this, `txn` would occupy the key + // forever and its own exclusive request could never fire. + entry.holders.retain(|h| *h != txn); + if !entry.has_waiter(txn) { + entry.waiters.push_back((txn, LockMode::Exclusive)); + } + } + // Free keys: no entry exists; the txn acquires them on the + // re-acquire path after all conflicting keys are released. + } + // Store the full key set so that release can promote this txn + // atomically once all its keys become available. + self.pending_keys.insert(txn, keys); + AcquireOutcome::Blocked + } + } + } +} + +// ── Tests ───────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + + fn key(name: &str) -> LockKey { + LockKey::Surrogate { + collection: Arc::from(name), + surrogate: 1, + } + } + + fn keyset(names: &[&str]) -> BTreeSet { + names.iter().map(|n| key(n)).collect() + } + + fn txn(epoch: u64, pos: u32) -> TxnId { + TxnId::new(epoch, pos) + } + + #[test] + fn acquire_free_keys_returns_ready() { + let mut lm = LockManager::new(); + let t = txn(1, 0); + let outcome = lm.acquire(t, keyset(&["a", "b"])); + assert_eq!(outcome, AcquireOutcome::Ready); + assert_eq!(lm.lock_count(), 2); + } + + #[test] + fn acquire_held_key_returns_blocked_and_enqueues_waiter() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + let t2 = txn(1, 1); + lm.acquire(t1, keyset(&["x"])); + + let outcome = lm.acquire(t2, keyset(&["x"])); + assert_eq!(outcome, AcquireOutcome::Blocked); + + // t2 should be in the waiter queue for "x". + assert!(lm.table.get(&key("x")).unwrap().has_waiter(t2)); + } + + #[test] + fn exclusive_waits_on_exclusive() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + let t2 = txn(1, 1); + + assert_eq!(lm.acquire(t1, keyset(&["k"])), AcquireOutcome::Ready); + assert_eq!(lm.acquire(t2, keyset(&["k"])), AcquireOutcome::Blocked); + assert!(lm.table.get(&key("k")).unwrap().has_waiter(t2)); + } + + #[test] + fn shared_reservation_self_upgrades_to_exclusive() { + let mut lm = LockManager::new(); + let t = txn(1, 0); + + assert_eq!(lm.acquire_shared(t, key("k")), AcquireOutcome::Ready); + // The txn re-acquires its own shared reservation exclusively — this must + // NOT self-deadlock by blocking on its own held key. + assert_eq!( + lm.acquire(t, keyset(&["k"])), + AcquireOutcome::Ready, + "self-upgrade from shared to exclusive must not block" + ); + + let entry = lm.table.get(&key("k")).unwrap(); + assert_eq!(entry.mode, LockMode::Exclusive); + assert_eq!(entry.holders.len(), 1); + assert_eq!(entry.holders[0], t); + } + + #[test] + fn self_upgrade_with_other_shared_holder_blocks_or_wounds() { + // T_old is older than T_young: T_old's self-upgrade must wound T_young. + let mut lm = LockManager::new(); + let t_old = txn(1, 0); + let t_young = txn(1, 1); + + assert_eq!(lm.acquire_shared(t_old, key("k")), AcquireOutcome::Ready); + assert_eq!(lm.acquire_shared(t_young, key("k")), AcquireOutcome::Ready); + + assert_eq!( + lm.acquire(t_old, keyset(&["k"])), + AcquireOutcome::Ready, + "the older self-upgrader wounds the younger shared holder" + ); + let entry = lm.table.get(&key("k")).unwrap(); + assert_eq!(entry.mode, LockMode::Exclusive); + assert_eq!(entry.holders.len(), 1); + assert_eq!(entry.holders[0], t_old); + + // Symmetric case: the YOUNGER of the two self-upgrades and must block. + let mut lm = LockManager::new(); + let t_old = txn(1, 0); + let t_young = txn(1, 1); + + assert_eq!(lm.acquire_shared(t_old, key("k")), AcquireOutcome::Ready); + assert_eq!(lm.acquire_shared(t_young, key("k")), AcquireOutcome::Ready); + + assert_eq!( + lm.acquire(t_young, keyset(&["k"])), + AcquireOutcome::Blocked, + "the younger self-upgrader must wait behind the older shared holder" + ); + // t_young drops its own shared hold (degrading to plain OCC) so the key + // can drain to empty and its exclusive request can later be promoted; + // t_old remains the sole shared holder, and t_young is enqueued as an + // exclusive waiter rather than left stuck as a non-waiting holder. + let entry = lm.table.get(&key("k")).unwrap(); + assert_eq!(entry.mode, LockMode::Shared); + assert!(entry.holders.contains(&t_old)); + assert!(!entry.holders.contains(&t_young)); + assert!(entry.has_waiter(t_young)); + + // Once t_old releases, t_young is promoted to sole exclusive holder. + let unblocked = lm.release(t_old); + assert!(unblocked.contains(&t_young)); + let entry = lm.table.get(&key("k")).unwrap(); + assert_eq!(entry.mode, LockMode::Exclusive); + assert_eq!(entry.holders.len(), 1); + assert_eq!(entry.holders[0], t_young); + } + + #[test] + fn self_upgrade_mixed_with_conflict_on_other_key() { + let mut lm = LockManager::new(); + let t = txn(1, 0); + let u = txn(1, 1); + + // T reserves K1 shared; U holds K2 exclusively. + assert_eq!(lm.acquire_shared(t, key("k1")), AcquireOutcome::Ready); + assert_eq!(lm.acquire(u, keyset(&["k2"])), AcquireOutcome::Ready); + + // T tries to take both keys exclusively: K2 conflicts with U, so T must + // block on the whole set — and critically must NOT self-deadlock on K1. + assert_eq!( + lm.acquire(t, keyset(&["k1", "k2"])), + AcquireOutcome::Blocked, + "conflict on k2 blocks the whole set" + ); + + // After U releases K2, T's re-acquire succeeds and upgrades K1 in place. + lm.release(u); + assert_eq!( + lm.acquire(t, keyset(&["k1", "k2"])), + AcquireOutcome::Ready, + "once k2 frees up, t acquires both keys exclusively" + ); + let k1 = lm.table.get(&key("k1")).unwrap(); + assert_eq!(k1.mode, LockMode::Exclusive); + assert_eq!(k1.holders.len(), 1); + assert_eq!(k1.holders[0], t); + let k2 = lm.table.get(&key("k2")).unwrap(); + assert_eq!(k2.mode, LockMode::Exclusive); + assert_eq!(k2.holders.len(), 1); + assert_eq!(k2.holders[0], t); + } + + #[test] + fn two_non_conflicting_both_dispatch_immediately() { + let mut lm = LockManager::new(); + + let txn1 = TxnId::new(1, 0); + let txn2 = TxnId::new(1, 1); + + let keys1: BTreeSet = [LockKey::Surrogate { + collection: Arc::from("coll"), + surrogate: 1, + }] + .into(); + let keys2: BTreeSet = [LockKey::Surrogate { + collection: Arc::from("coll"), + surrogate: 2, + }] + .into(); + + let o1 = lm.acquire(txn1, keys1); + let o2 = lm.acquire(txn2, keys2); + + assert_eq!(o1, AcquireOutcome::Ready, "txn1 should be ready"); + assert_eq!( + o2, + AcquireOutcome::Ready, + "txn2 should be ready (disjoint keys)" + ); + } + + #[test] + fn many_mixed_deterministic_dispatch_order() { + let mut lm = LockManager::new(); + let mut dispatched: Vec = Vec::new(); + + let pairs = [(2, 0), (1, 1), (3, 0), (1, 0), (2, 1)]; + for (epoch, pos) in pairs { + let tid = TxnId::new(epoch, pos); + let keys: BTreeSet = [LockKey::Surrogate { + collection: Arc::from(format!("c_{epoch}_{pos}")), + surrogate: epoch as u32 * 10 + pos, + }] + .into(); + let outcome = lm.acquire(tid, keys); + if outcome == AcquireOutcome::Ready { + dispatched.push(tid); + } + } + + assert_eq!( + dispatched.len(), + 5, + "all non-conflicting txns should be ready" + ); + + let mut expected = pairs.map(|(e, p)| TxnId::new(e, p)).to_vec(); + expected.sort(); + let mut sorted_dispatched = dispatched.clone(); + sorted_dispatched.sort(); + assert_eq!(sorted_dispatched, expected); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/lock/manager/introspection.rs b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/introspection.rs new file mode 100644 index 000000000..ef9f4e13d --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/introspection.rs @@ -0,0 +1,96 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Read-only inspection of lock manager state — readiness checks and the +//! test-only counters used to assert on table/holder sizes. + +use std::collections::BTreeSet; + +use crate::control::cluster::calvin::scheduler::lock::lock_key::{LockKey, TxnId}; + +use super::types::LockManager; + +impl LockManager { + /// Check whether a previously-blocked transaction is now ready. + /// + /// A transaction is ready when for every key in its key set, the key is + /// either: + /// - Not present in the lock table (free), or + /// - Present in the lock table with `txn` among the current holders + /// (shared or exclusive). + /// + /// This is called after `release` returns `txn_id` in the unblocked set. + /// If `is_ready` returns `true`, the caller calls `acquire` again which + /// will succeed on the all-available path (because the waiter was promoted). + pub fn is_ready(&self, txn: TxnId, keys: &BTreeSet) -> bool { + keys.iter().all(|key| { + match self.table.get(key) { + None => true, // key is free + Some(entry) => entry.holders.contains(&txn), // txn is a current holder + } + }) + } + + /// Number of holders of `key` that hold it as a Calvin read reservation + /// (a `TxnId` in the reservation position band), or 0 when the key is + /// unlocked or held only by non-reservation transactions. Used to observe + /// reservation install/release from outside the scheduler. + pub fn reservation_holder_count(&self, key: &LockKey) -> usize { + self.table + .get(key) + .map(|e| e.holders.iter().filter(|h| h.is_reservation()).count()) + .unwrap_or(0) + } + + /// Number of currently-held locks (entries in the lock table). + #[cfg(test)] + pub fn lock_count(&self) -> usize { + self.table.len() + } + + /// Number of transactions currently holding at least one lock. + #[cfg(test)] + pub fn holder_count(&self) -> usize { + self.held_locks.len() + } +} + +// ── Tests ───────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + + fn key(name: &str) -> LockKey { + LockKey::Surrogate { + collection: Arc::from(name), + surrogate: 1, + } + } + + fn keyset(names: &[&str]) -> BTreeSet { + names.iter().map(|n| key(n)).collect() + } + + fn txn(epoch: u64, pos: u32) -> TxnId { + TxnId::new(epoch, pos) + } + + #[test] + fn is_ready_returns_true_when_all_keys_free_or_self_at_front() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + let t2 = txn(1, 1); + lm.acquire(t1, keyset(&["x", "y"])); + lm.acquire(t2, keyset(&["x", "y"])); + + // t2 is not ready while t1 holds. + assert!(!lm.is_ready(t2, &keyset(&["x", "y"]))); + + // Release t1 — t2 becomes holder on both keys. + lm.release(t1); + // After release, t2 is promoted to holder on both keys. + assert!(lm.is_ready(t2, &keyset(&["x", "y"]))); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/lock/manager/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/mod.rs new file mode 100644 index 000000000..e59e665b9 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/mod.rs @@ -0,0 +1,38 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Deterministic lock manager for the Calvin scheduler. +//! +//! # Design +//! +//! The lock manager provides a deterministic, totally-ordered lock table over +//! per-key entries keyed by +//! [`LockKey`](crate::control::cluster::calvin::scheduler::lock::LockKey). +//! Locks come in two modes: `Exclusive` (one holder, excludes all others) and +//! `Shared` (many compatible holders). The Calvin batch acquire path takes +//! every key in a transaction's `read_set ∪ write_set` as an `Exclusive` +//! lock; single-key `Shared` locks are available via +//! [`LockManager::acquire_shared`]. +//! +//! # Determinism +//! +//! `BTreeMap` is used throughout (not `HashMap`) so that iteration order is +//! deterministic and reproducible across replicas. This is a correctness +//! requirement, not a style preference. +//! +//! Split by concern: +//! - [`types`]: the lock table struct and its internal decision enums. +//! - [`acquire`]: exclusive lock acquisition and waiter queueing. +//! - [`wound_wait`]: shared-lock reservations and wound-wait conflict +//! resolution. +//! - [`release`]: lock release and FIFO/shared waiter promotion. +//! - [`try_acquire`]: the non-blocking exclusive fast path. +//! - [`introspection`]: readiness checks and test counters. + +mod acquire; +mod introspection; +mod release; +mod try_acquire; +mod types; +mod wound_wait; + +pub use types::LockManager; diff --git a/nodedb/src/control/cluster/calvin/scheduler/lock/manager/release.rs b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/release.rs new file mode 100644 index 000000000..1ba0550d5 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/release.rs @@ -0,0 +1,334 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Lock release and FIFO/shared waiter promotion. + +use std::collections::BTreeSet; + +use crate::control::cluster::calvin::scheduler::lock::lock_entry::LockMode; +use crate::control::cluster::calvin::scheduler::lock::lock_key::{LockKey, TxnId}; + +use super::types::{LockManager, Promotion}; + +impl LockManager { + /// Release all locks held by `txn`. + /// + /// `txn` is removed from every entry's holder set. When an entry's holders + /// drain to empty, its FIFO waiters are promoted mode-aware: a leading run + /// of shared waiters is promoted together, or a single leading exclusive + /// waiter is promoted alone. A waiter that becomes holder on ALL its + /// pending keys is moved from `pending_keys` to `held_locks` immediately. + /// + /// Returns the set of `TxnId`s that have been fully promoted (i.e. moved + /// into `held_locks`). The caller may use this list to dispatch those + /// transactions. + pub fn release(&mut self, txn: TxnId) -> Vec { + let held = match self.held_locks.remove(&txn) { + Some(h) => h, + None => return Vec::new(), + }; + + let mut newly_promoted: BTreeSet = BTreeSet::new(); + + for key in &held { + // Drop `txn` from this key's holders. If other (shared) holders + // remain, the key stays held and there is nothing to promote. + let now_empty = match self.table.get_mut(key) { + Some(entry) => { + entry.holders.retain(|h| *h != txn); + entry.holders.is_empty() + } + None => continue, + }; + if now_empty { + self.promote_waiters(key, &mut newly_promoted); + } + } + + newly_promoted.into_iter().collect() + } + + /// Promote the front of `key`'s waiter queue after its holders drained. + /// + /// A leading run of shared waiters is granted together; a single leading + /// exclusive waiter is granted alone; an empty queue frees the entry. Any + /// promoted txn that is now holder on all of its pending keys is moved into + /// `held_locks` and recorded in `newly_promoted`. + fn promote_waiters(&mut self, key: &LockKey, newly_promoted: &mut BTreeSet) { + // Decide the promotion inside a scoped borrow so the readiness sweep + // below can re-borrow the table. + let decision = match self.table.get_mut(key) { + Some(entry) => match entry.waiters.front().map(|(_, mode)| *mode) { + None => Promotion::Freed, + Some(LockMode::Exclusive) => match entry.waiters.pop_front() { + Some((next, _)) => { + entry.mode = LockMode::Exclusive; + entry.holders.clear(); + entry.holders.push(next); + Promotion::Promoted(vec![next]) + } + None => Promotion::Freed, + }, + Some(LockMode::Shared) => { + entry.mode = LockMode::Shared; + entry.holders.clear(); + let mut promoted = Vec::new(); + while matches!(entry.waiters.front(), Some((_, LockMode::Shared))) { + if let Some((next, _)) = entry.waiters.pop_front() { + entry.holders.push(next); + promoted.push(next); + } + } + Promotion::Promoted(promoted) + } + }, + None => return, + }; + + let promoted = match decision { + Promotion::Freed => { + self.table.remove(key); + return; + } + Promotion::Promoted(promoted) => promoted, + }; + + // For each promoted txn, check whether it is now holder on ALL of its + // pending keys. If so, it is fully ready — move to held_locks. Remove + // first (rather than `get` + a follow-up `remove`) so there is no + // unwrap/expect on a "just confirmed Some" invariant: the owned + // `pending` set is reinserted on the not-yet-ready path. + for next in promoted { + if let Some(pending) = self.pending_keys.remove(&next) { + let all_held = pending + .iter() + .all(|k| self.table.get(k).is_none_or(|e| e.holders.contains(&next))); + if all_held { + self.held_locks.insert(next, pending); + newly_promoted.insert(next); + } else { + self.pending_keys.insert(next, pending); + } + } + } + } +} + +// ── Tests ───────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + use crate::control::cluster::calvin::scheduler::lock::lock_entry::AcquireOutcome; + + fn key(name: &str) -> LockKey { + LockKey::Surrogate { + collection: Arc::from(name), + surrogate: 1, + } + } + + fn keyset(names: &[&str]) -> BTreeSet { + names.iter().map(|n| key(n)).collect() + } + + fn txn(epoch: u64, pos: u32) -> TxnId { + TxnId::new(epoch, pos) + } + + #[test] + fn release_returns_unblocked_waiter_ids() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + let t2 = txn(1, 1); + lm.acquire(t1, keyset(&["x"])); + lm.acquire(t2, keyset(&["x"])); + + let unblocked = lm.release(t1); + assert!(unblocked.contains(&t2)); + } + + #[test] + fn autocommit_holder_release_promotes_and_returns_scheduler_waiter() { + // Mirrors the write-admission fast path: an autocommit-band holder takes + // an uncontended key, a normal-band scheduler txn then blocks behind it, + // and the holder's release promotes that scheduler txn AND returns its id + // — the value the fast-path guard forwards to the scheduler on drop + // (previously discarded, stranding the promoted txn as a zombie holder). + let mut lm = LockManager::new(); + let autocommit = txn(TxnId::AUTOCOMMIT_EPOCH, 0); + let scheduler_txn = txn(9, 0); + + assert!( + lm.try_acquire(autocommit, keyset(&["k"])), + "the fast-path holder takes the uncontended key" + ); + assert_eq!( + lm.acquire(scheduler_txn, keyset(&["k"])), + AcquireOutcome::Blocked, + "the scheduler txn queues behind the fast-path holder" + ); + + let promoted = lm.release(autocommit); + assert_eq!( + promoted, + vec![scheduler_txn], + "release must return the promoted scheduler waiter" + ); + assert!( + lm.is_ready(scheduler_txn, &keyset(&["k"])), + "the promoted scheduler txn is now holder of the freed key" + ); + } + + #[test] + fn release_preserves_fifo_waiter_order() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + let t2 = txn(1, 1); + let t3 = txn(1, 2); + lm.acquire(t1, keyset(&["x"])); + lm.acquire(t2, keyset(&["x"])); + lm.acquire(t3, keyset(&["x"])); + + // Release t1 — t2 should become holder (FIFO). + lm.release(t1); + let holder = lm.table.get(&key("x")).unwrap().holders[0]; + assert_eq!(holder, t2); + + // Release t2 — t3 should become holder. + lm.release(t2); + let holder = lm.table.get(&key("x")).unwrap().holders[0]; + assert_eq!(holder, t3); + } + + #[test] + fn multi_key_txn_releases_all_atomically() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + lm.acquire(t1, keyset(&["a", "b", "c"])); + assert_eq!(lm.lock_count(), 3); + + lm.release(t1); + assert_eq!(lm.lock_count(), 0); + assert_eq!(lm.holder_count(), 0); + } + + #[test] + fn release_promotes_shared_run_together() { + let mut lm = LockManager::new(); + let holder = txn(1, 0); + let s1 = txn(2, 0); + let s2 = txn(2, 1); + + // Exclusive holder, two shared waiters queued behind it. + assert_eq!(lm.acquire(holder, keyset(&["k"])), AcquireOutcome::Ready); + assert_eq!(lm.acquire_shared(s1, key("k")), AcquireOutcome::Blocked); + assert_eq!(lm.acquire_shared(s2, key("k")), AcquireOutcome::Blocked); + + // Releasing the exclusive holder promotes the whole run of shared + // waiters together. + let promoted = lm.release(holder); + assert!(promoted.contains(&s1)); + assert!(promoted.contains(&s2)); + + let entry = lm.table.get(&key("k")).unwrap(); + assert_eq!(entry.mode, LockMode::Shared); + assert!(entry.holders.contains(&s1)); + assert!(entry.holders.contains(&s2)); + } + + #[test] + fn release_promotes_single_exclusive_waiter() { + let mut lm = LockManager::new(); + let holder = txn(1, 0); + let x1 = txn(2, 0); + let x2 = txn(2, 1); + + assert_eq!(lm.acquire(holder, keyset(&["k"])), AcquireOutcome::Ready); + assert_eq!(lm.acquire(x1, keyset(&["k"])), AcquireOutcome::Blocked); + assert_eq!(lm.acquire(x2, keyset(&["k"])), AcquireOutcome::Blocked); + + // Only the single leading exclusive waiter is promoted. + let promoted = lm.release(holder); + assert_eq!(promoted, vec![x1]); + + let entry = lm.table.get(&key("k")).unwrap(); + assert_eq!(entry.mode, LockMode::Exclusive); + assert_eq!(entry.holders.len(), 1); + assert_eq!(entry.holders[0], x1); + // x2 is still waiting behind x1. + assert!(entry.has_waiter(x2)); + } + + #[test] + fn multi_holder_release() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + let t2 = txn(1, 1); + + assert_eq!(lm.acquire_shared(t1, key("k")), AcquireOutcome::Ready); + assert_eq!(lm.acquire_shared(t2, key("k")), AcquireOutcome::Ready); + assert_eq!(lm.lock_count(), 1); + + // Releasing one shared holder leaves the other holding the key. + lm.release(t1); + let entry = lm.table.get(&key("k")).unwrap(); + assert!(!entry.holders.contains(&t1)); + assert!(entry.holders.contains(&t2)); + assert_eq!(lm.lock_count(), 1); + + // Releasing the last shared holder frees the key. + lm.release(t2); + assert_eq!(lm.lock_count(), 0); + } + + #[test] + fn two_conflicting_second_dispatches_after_first_completes() { + let mut lm = LockManager::new(); + + let txn1 = TxnId::new(1, 0); + let txn2 = TxnId::new(1, 1); + let shared_key: BTreeSet = [LockKey::Surrogate { + collection: Arc::from("coll"), + surrogate: 42, + }] + .into(); + + let o1 = lm.acquire(txn1, shared_key.clone()); + assert_eq!(o1, AcquireOutcome::Ready); + + let o2 = lm.acquire(txn2, shared_key.clone()); + assert_eq!(o2, AcquireOutcome::Blocked); + + let unblocked = lm.release(txn1); + assert!(unblocked.contains(&txn2)); + + assert!(lm.is_ready(txn2, &shared_key)); + } + + #[test] + fn cross_epoch_raw_blocks_correctly() { + let mut lm = LockManager::new(); + + let txn_n = TxnId::new(1, 0); + let txn_n1 = TxnId::new(2, 0); + + let key_k: BTreeSet = [LockKey::Surrogate { + collection: Arc::from("orders"), + surrogate: 100, + }] + .into(); + + let o1 = lm.acquire(txn_n, key_k.clone()); + assert_eq!(o1, AcquireOutcome::Ready); + + let o2 = lm.acquire(txn_n1, key_k.clone()); + assert_eq!(o2, AcquireOutcome::Blocked); + + let unblocked = lm.release(txn_n); + assert!(unblocked.contains(&txn_n1)); + assert!(lm.is_ready(txn_n1, &key_k)); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/lock/manager/try_acquire.rs b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/try_acquire.rs new file mode 100644 index 000000000..0dee86f99 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/try_acquire.rs @@ -0,0 +1,40 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Non-blocking exclusive acquire fast path. + +use std::collections::BTreeSet; + +use crate::control::cluster::calvin::scheduler::lock::lock_entry::AcquireOutcome; +use crate::control::cluster::calvin::scheduler::lock::lock_key::{LockKey, TxnId}; + +use super::types::LockManager; + +impl LockManager { + /// Non-blocking exclusive acquire: take all `keys` for `txn` iff every one is + /// free (or already held by `txn`), returning `true`; otherwise return + /// `false` WITHOUT enqueuing a waiter or recording any pending state. + /// + /// This is the fast path's probe. Unlike [`acquire`](Self::acquire), the + /// contended (`false`) path touches NOTHING — no holder, no `pending_keys`, + /// no waiter `VecDeque` — so a caller that does not intend to block (an + /// autocommit point write that will instead route to the scheduler) never + /// leaves an orphaned waiter that a later `release` would promote to an + /// unowned holder. It also never perturbs the FIFO ordering that Calvin + /// transactions depend on. + pub fn try_acquire(&mut self, txn: TxnId, keys: BTreeSet) -> bool { + if !self.is_ready(txn, &keys) { + // Contended: leave the table, waiter queues, and pending_keys + // completely untouched. + return false; + } + // Every key is free or already held by `txn`, so `acquire` takes its + // all-available path — it inserts the holder and never enqueues. + let outcome = self.acquire(txn, keys); + debug_assert_eq!( + outcome, + AcquireOutcome::Ready, + "try_acquire: is_ready was true but acquire returned Blocked" + ); + true + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/lock/manager/types.rs b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/types.rs new file mode 100644 index 000000000..93e6862bd --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/types.rs @@ -0,0 +1,101 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The lock table struct, its internal decision enums, and the small +//! per-entry predicates the acquire/release paths share. + +use std::collections::{BTreeMap, BTreeSet}; + +use crate::control::cluster::calvin::scheduler::lock::lock_entry::{LockEntry, LockMode}; +use crate::control::cluster::calvin::scheduler::lock::lock_key::{LockKey, TxnId}; + +/// Deterministic Calvin lock manager for one vshard. +/// +/// Manages an in-memory lock table keyed by [`LockKey`]. The table is held in +/// a `BTreeMap` so iteration is always deterministic. +/// +/// # Key sets tracked per transaction +/// +/// - `held_locks`: key sets for transactions that are a current holder on ALL +/// their keys and are actively executing (i.e. dispatched to the Data Plane). +/// - `pending_keys`: key sets for transactions that are blocked waiting for at +/// least one key. When `release` promotes a blocked txn to holder on every +/// one of its keys, the entry moves from `pending_keys` to `held_locks`. +pub struct LockManager { + /// Per-key lock entries. Uses `BTreeMap` for deterministic iteration. + /// Visible to the whole `lock` module so the sibling `reap` module can + /// scan entries for lease-expired reservations without a public accessor. + pub(in crate::control::cluster::calvin::scheduler::lock) table: BTreeMap, + /// Per-transaction set of currently held keys for **dispatched** txns. + /// Used by `release` to iterate the key set without a full table scan. + /// Visible to the whole `lock` module — see `table`. + pub(in crate::control::cluster::calvin::scheduler::lock) held_locks: + BTreeMap>, + /// Key sets for **blocked** (not-yet-dispatched) txns. Populated when + /// `acquire` returns `Blocked`; cleared (moved to `held_locks`) when all + /// keys have been acquired on the promotion path inside `release`. + pub(super) pending_keys: BTreeMap>, +} + +/// Outcome of inspecting a single key during [`LockManager::acquire_shared`]. +pub(super) enum SharedGrant { + /// The shared lock was granted (key was free or already held shared). + Granted, + /// The key is held exclusively by another txn; the request was enqueued. + Blocked, +} + +/// The wound-wait decision for an exclusive requester that meets a conflict. +pub(super) enum ExclusiveWait { + /// Every conflicting holder is a shared reservation and the requester is + /// older than all of them: wound (revoke) those shared holders and proceed. + Wound, + /// The requester must block: a conflicting holder is exclusive, or the + /// requester is younger than some conflicting shared holder. + Block, +} + +/// The waiters promoted off one key when its holders drained, together with the +/// action to take on the now-empty entry. +pub(super) enum Promotion { + /// No waiters remained; the entry should be removed entirely. + Freed, + /// These waiters were installed as the new holders. + Promoted(Vec), +} + +impl LockManager { + /// Create an empty lock manager. + pub fn new() -> Self { + Self { + table: BTreeMap::new(), + held_locks: BTreeMap::new(), + pending_keys: BTreeMap::new(), + } + } +} + +impl Default for LockManager { + fn default() -> Self { + Self::new() + } +} + +impl LockEntry { + /// Whether this entry is held exclusively by exactly `txn` (the self + /// re-acquire case on the exclusive path). + pub(super) fn held_exclusively_by(&self, txn: TxnId) -> bool { + self.mode == LockMode::Exclusive && self.holders.len() == 1 && self.holders[0] == txn + } + + /// Whether this entry is held **shared** by exactly `txn` and no one else — + /// the self-upgrade case: `txn` may take the key exclusively because it is + /// the sole current holder. + pub(super) fn held_shared_solely_by(&self, txn: TxnId) -> bool { + self.mode == LockMode::Shared && self.holders.len() == 1 && self.holders[0] == txn + } + + /// Whether `txn` is already enqueued as a waiter on this entry. + pub(super) fn has_waiter(&self, txn: TxnId) -> bool { + self.waiters.iter().any(|(w, _)| *w == txn) + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/lock/manager/wound_wait.rs b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/wound_wait.rs new file mode 100644 index 000000000..2a77b1a45 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/wound_wait.rs @@ -0,0 +1,313 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Shared-lock reservations and the wound-wait conflict resolution used by +//! exclusive acquisition. + +use std::collections::btree_map::Entry; +use std::collections::{BTreeSet, VecDeque}; + +use smallvec::smallvec; + +use crate::control::cluster::calvin::scheduler::lock::lock_entry::{ + AcquireOutcome, LockEntry, LockMode, +}; +use crate::control::cluster::calvin::scheduler::lock::lock_key::{LockKey, TxnId}; + +use super::types::{ExclusiveWait, LockManager, SharedGrant}; + +impl LockManager { + /// Classify the wound-wait decision for an exclusive requester `txn` over + /// `keys`, given that at least one key already conflicts. + /// + /// Pure read over the lock table: any exclusive conflict forces + /// [`ExclusiveWait::Block`] (an exclusive holder is never wounded, so a mix + /// of exclusive and shared conflicts blocks too). Otherwise all conflicting + /// holders are shared reservations, and `txn` wounds them only when it is + /// older than every one (`txn < h` for each conflicting shared holder `h`); + /// if it is younger than any, it blocks. A key held only by `txn` itself is + /// not a conflict. + pub(super) fn wound_or_block(&self, txn: TxnId, keys: &BTreeSet) -> ExclusiveWait { + let mut shared_conflicts: Vec = Vec::new(); + for key in keys { + if let Some(entry) = self.table.get(key) { + match entry.mode { + LockMode::Exclusive => { + // Exclusive entries have exactly one holder; a holder + // other than `txn` is an exclusive conflict. + if !entry.holders.contains(&txn) { + return ExclusiveWait::Block; + } + } + LockMode::Shared => { + for holder in &entry.holders { + if *holder != txn { + shared_conflicts.push(*holder); + } + } + } + } + } + } + // Wound only when there is a shared conflict AND `txn` is older than + // every conflicting shared holder; otherwise block. `shared_conflicts` + // only ever holds *other* txns' shared holders (a key held shared solely + // by `txn` never reaches here — it takes the self-upgrade path in + // `acquire`), so an empty set here means every conflict was exclusive. + if !shared_conflicts.is_empty() && shared_conflicts.iter().all(|holder| txn < *holder) { + ExclusiveWait::Wound + } else { + ExclusiveWait::Block + } + } + + /// Attempt to acquire a **shared** lock on a single `key` for `txn`. + /// + /// - Key free → create a shared entry holding `txn`, return + /// [`AcquireOutcome::Ready`]. + /// - Key held shared → add `txn` to the holders, return + /// [`AcquireOutcome::Ready`]. + /// - Key held exclusively by another txn → enqueue `txn` as a shared waiter + /// (FIFO) and return [`AcquireOutcome::Blocked`]. + /// + /// A shared request that meets an exclusive holder blocks FIFO for now; + /// wound-wait priority resolution lands in a following change. + pub fn acquire_shared(&mut self, txn: TxnId, key: LockKey) -> AcquireOutcome { + // Inspect / mutate the entry via the `Entry` API (which takes the key by + // value, sidestepping a get-then-insert borrow conflict) inside a scoped + // borrow so the map-level bookkeeping below can re-borrow `self`. + let grant = match self.table.entry(key.clone()) { + Entry::Vacant(slot) => { + slot.insert(LockEntry { + mode: LockMode::Shared, + holders: smallvec![txn], + waiters: VecDeque::new(), + }); + SharedGrant::Granted + } + Entry::Occupied(mut slot) => { + let entry = slot.get_mut(); + if entry.mode == LockMode::Shared { + if !entry.holders.contains(&txn) { + entry.holders.push(txn); + } + SharedGrant::Granted + } else { + // Held exclusively by another txn: block FIFO. + if !entry.has_waiter(txn) { + entry.waiters.push_back((txn, LockMode::Shared)); + } + SharedGrant::Blocked + } + } + }; + + match grant { + SharedGrant::Granted => { + self.pending_keys.remove(&txn); + self.held_locks.entry(txn).or_default().insert(key); + AcquireOutcome::Ready + } + SharedGrant::Blocked => { + let mut pending = BTreeSet::new(); + pending.insert(key); + self.pending_keys.insert(txn, pending); + AcquireOutcome::Blocked + } + } + } +} + +// ── Tests ───────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + + fn key(name: &str) -> LockKey { + LockKey::Surrogate { + collection: Arc::from(name), + surrogate: 1, + } + } + + fn keyset(names: &[&str]) -> BTreeSet { + names.iter().map(|n| key(n)).collect() + } + + fn txn(epoch: u64, pos: u32) -> TxnId { + TxnId::new(epoch, pos) + } + + #[test] + fn shared_shared_compatible() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + let t2 = txn(1, 1); + + assert_eq!(lm.acquire_shared(t1, key("s")), AcquireOutcome::Ready); + assert_eq!(lm.acquire_shared(t2, key("s")), AcquireOutcome::Ready); + + let entry = lm.table.get(&key("s")).unwrap(); + assert_eq!(entry.mode, LockMode::Shared); + assert!(entry.holders.contains(&t1)); + assert!(entry.holders.contains(&t2)); + } + + #[test] + fn shared_blocks_exclusive() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + let t2 = txn(1, 1); + + assert_eq!(lm.acquire_shared(t1, key("k")), AcquireOutcome::Ready); + assert_eq!( + lm.acquire(t2, keyset(&["k"])), + AcquireOutcome::Blocked, + "an exclusive request must block behind a shared holder" + ); + assert!(lm.table.get(&key("k")).unwrap().has_waiter(t2)); + } + + #[test] + fn exclusive_blocks_shared() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + let t2 = txn(1, 1); + + assert_eq!(lm.acquire(t1, keyset(&["k"])), AcquireOutcome::Ready); + assert_eq!( + lm.acquire_shared(t2, key("k")), + AcquireOutcome::Blocked, + "a shared request must block behind an exclusive holder" + ); + assert!(lm.table.get(&key("k")).unwrap().has_waiter(t2)); + } + + #[test] + fn older_writer_wounds_shared() { + let mut lm = LockManager::new(); + let t2 = txn(1, 2); // shared holder + let t1 = txn(1, 1); // exclusive requester, older than t2 + + assert_eq!(lm.acquire_shared(t2, key("k")), AcquireOutcome::Ready); + // The older writer wounds the younger shared holder and proceeds. + assert_eq!(lm.acquire(t1, keyset(&["k"])), AcquireOutcome::Ready); + + let entry = lm.table.get(&key("k")).unwrap(); + assert_eq!(entry.mode, LockMode::Exclusive); + assert!(entry.holders.contains(&t1), "R is now the exclusive holder"); + assert!( + !entry.holders.contains(&t2), + "the wounded shared holder is gone" + ); + } + + #[test] + fn younger_writer_waits() { + let mut lm = LockManager::new(); + let t1 = txn(1, 1); // shared holder + let t2 = txn(1, 2); // exclusive requester, younger than t1 + + assert_eq!(lm.acquire_shared(t1, key("k")), AcquireOutcome::Ready); + // The younger writer must not wound; it waits behind the shared holder. + assert_eq!(lm.acquire(t2, keyset(&["k"])), AcquireOutcome::Blocked); + + let entry = lm.table.get(&key("k")).unwrap(); + assert!(entry.holders.contains(&t1), "the shared holder still holds"); + assert!(!entry.holders.contains(&t2), "R holds nothing"); + assert!(entry.has_waiter(t2), "R is enqueued as an exclusive waiter"); + } + + #[test] + fn exclusive_waits_on_exclusive_regardless_of_age() { + let mut lm = LockManager::new(); + let t2 = txn(1, 2); // exclusive holder (younger) + let t1 = txn(1, 1); // exclusive requester (older) + + assert_eq!(lm.acquire(t2, keyset(&["k"])), AcquireOutcome::Ready); + // An exclusive holder is NEVER wounded, even by an older writer. + assert_eq!(lm.acquire(t1, keyset(&["k"])), AcquireOutcome::Blocked); + + let entry = lm.table.get(&key("k")).unwrap(); + assert!( + entry.holders.contains(&t2), + "the exclusive holder is intact" + ); + assert!(!entry.holders.contains(&t1)); + assert!(entry.has_waiter(t1)); + } + + #[test] + fn multi_key_atomic_wound_takes_both() { + let mut lm = LockManager::new(); + let s1 = txn(1, 5); // shared holder on k1, younger than R + let s2 = txn(1, 6); // shared holder on k2, younger than R + let r = txn(1, 1); // exclusive requester, older than both + + assert_eq!(lm.acquire_shared(s1, key("k1")), AcquireOutcome::Ready); + assert_eq!(lm.acquire_shared(s2, key("k2")), AcquireOutcome::Ready); + + assert_eq!(lm.acquire(r, keyset(&["k1", "k2"])), AcquireOutcome::Ready); + + for k in ["k1", "k2"] { + let entry = lm.table.get(&key(k)).unwrap(); + assert_eq!(entry.mode, LockMode::Exclusive); + assert!(entry.holders.contains(&r), "R holds {k}"); + } + assert!(!lm.table.get(&key("k1")).unwrap().holders.contains(&s1)); + assert!(!lm.table.get(&key("k2")).unwrap().holders.contains(&s2)); + } + + #[test] + fn multi_key_atomic_wait_holds_none() { + let mut lm = LockManager::new(); + let s1 = txn(1, 5); // shared holder on k1, younger than R + let s2 = txn(1, 0); // shared holder on k2, OLDER than R + let r = txn(1, 1); // exclusive requester + + assert_eq!(lm.acquire_shared(s1, key("k1")), AcquireOutcome::Ready); + assert_eq!(lm.acquire_shared(s2, key("k2")), AcquireOutcome::Ready); + + // R is younger than the holder on k2, so it must wait on BOTH keys and + // hold neither (all-or-nothing). + assert_eq!( + lm.acquire(r, keyset(&["k1", "k2"])), + AcquireOutcome::Blocked + ); + + assert!( + !lm.table.get(&key("k1")).unwrap().holders.contains(&r), + "R holds no key" + ); + assert!(!lm.table.get(&key("k2")).unwrap().holders.contains(&r)); + // The older shared holder on k2 is untouched. + assert!(lm.table.get(&key("k2")).unwrap().holders.contains(&s2)); + } + + #[test] + fn crossed_reservations_are_acyclic() { + // T1 holds shared K1 and wants exclusive K2; T2 holds shared K2 and + // wants exclusive K1. The older writer's exclusive acquire wounds the + // younger's shared holding, breaking the cycle — no deadlock. + let mut lm = LockManager::new(); + let t1 = txn(1, 1); // older + let t2 = txn(1, 2); // younger + + assert_eq!(lm.acquire_shared(t1, key("k1")), AcquireOutcome::Ready); + assert_eq!(lm.acquire_shared(t2, key("k2")), AcquireOutcome::Ready); + + // T1 (older) acquires exclusive K2: wounds T2's shared holding and + // proceeds. + assert_eq!(lm.acquire(t1, keyset(&["k2"])), AcquireOutcome::Ready); + + let k2 = lm.table.get(&key("k2")).unwrap(); + assert_eq!(k2.mode, LockMode::Exclusive); + assert!(k2.holders.contains(&t1), "the older writer proceeds"); + assert!( + !k2.holders.contains(&t2), + "the younger's reservation is wounded away" + ); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/metrics.rs b/nodedb/src/control/cluster/calvin/scheduler/metrics.rs index d8ac0d837..9019eb688 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/metrics.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/metrics.rs @@ -58,6 +58,28 @@ pub struct SchedulerMetrics { /// flags that catch-up is relying on snapshot coverage rather than log replay /// and warrants operator attention. pub catch_up_log_compacted: AtomicU64, + /// Scheduler dispatches the bridge dispatcher refused at capacity. Each + /// refusal parks the request for re-send; none is an abort. + pub dispatch_deferred_count: AtomicU64, + /// Requests parked for re-send right now, waiting for dispatcher capacity. + pub dispatch_deferred_depth: AtomicU64, + /// Intake gate state: 1 while the scheduler takes no new sequenced input, + /// 0 while it does. + pub intake_gate_closed: AtomicU64, + /// In-flight backlog: pending, blocked, and dependent-barrier txns. + pub intake_backlog: AtomicU64, + /// Intake gate closures by reason. Indexes are the constants in + /// [`intake_closure_reason`]. + pub intake_gate_closed_counts: [AtomicU64; 3], + /// Apply halt state: 1 once the scheduler halted, 0 while it applies. + pub apply_halted: AtomicU64, + /// Reason of the halt. An index into [`apply_halt_reason`], read only + /// while `apply_halted` is 1. + pub apply_halt_reason: AtomicU64, + /// Owed sequencer entries re-proposed because their effect was not yet + /// applied, by kind. Indexes are the constants in + /// [`sequencer_propose_kind`]. + pub sequencer_propose_retry_counts: [AtomicU64; 4], } /// Reason codes for `nodedb_calvin_infra_abort_total`. @@ -79,6 +101,48 @@ pub mod infra_abort_reason { ]; } +/// Reason codes for `nodedb_calvin_intake_gate_closed_total`. +pub mod intake_closure_reason { + pub const DEFERRED_DISPATCH: usize = 0; + pub const BACKLOG_FULL: usize = 1; + pub const APPLY_HALTED: usize = 2; + + pub const LABELS: &[&str] = &["deferred_dispatch", "backlog_full", "apply_halted"]; +} + +/// Reason codes for `nodedb_calvin_apply_halted`. +pub mod apply_halt_reason { + pub const DRAINING: usize = 0; + pub const DISPATCH_REFUSED: usize = 1; + pub const RESPONSE_DISCONNECTED: usize = 2; + pub const RESOLVE_FAILED: usize = 3; + pub const FLUSH_FAILED: usize = 4; + pub const LOCAL_STAGE_FAILED: usize = 5; + pub const IDENTITY_BIND_FAILED: usize = 6; + pub const WAL_APPEND_FAILED: usize = 7; + + pub const LABELS: &[&str] = &[ + "draining", + "dispatch_refused", + "response_disconnected", + "resolve_failed", + "flush_failed", + "local_stage_failed", + "identity_bind_failed", + "wal_append_failed", + ]; +} + +/// Kind codes for `nodedb_calvin_sequencer_propose_retry_total`. +pub mod sequencer_propose_kind { + pub const VOTE: usize = 0; + pub const COMPLETION_ACK: usize = 1; + pub const OLLP_MISMATCH: usize = 2; + pub const ROUTING_FAILED: usize = 3; + + pub const LABELS: &[&str] = &["vote", "completion_ack", "ollp_mismatch", "routing_failed"]; +} + impl SchedulerMetrics { pub fn new() -> Arc { Arc::new(Self::default()) @@ -135,6 +199,55 @@ impl SchedulerMetrics { self.catch_up_log_compacted.fetch_add(1, Ordering::Relaxed); } + /// Record that the dispatcher refused a scheduler dispatch at capacity. + pub fn record_dispatch_deferred(&self) { + self.dispatch_deferred_count.fetch_add(1, Ordering::Relaxed); + } + + /// Set the number of requests parked for re-send. + pub fn set_dispatch_deferred_depth(&self, depth: usize) { + self.dispatch_deferred_depth + .store(depth as u64, Ordering::Relaxed); + } + + /// Record that the intake gate closed for `reason`. + /// + /// `reason` must be one of the constants in [`intake_closure_reason`]. + pub fn record_intake_gate_closed(&self, reason: usize) { + if let Some(counter) = self.intake_gate_closed_counts.get(reason) { + counter.fetch_add(1, Ordering::Relaxed); + } + } + + /// Set the intake gate state gauge. + pub fn set_intake_gate_closed(&self, closed: bool) { + self.intake_gate_closed + .store(u64::from(closed), Ordering::Relaxed); + } + + /// Set the apply halt gauge for `reason`. + /// + /// `reason` must be one of the constants in [`apply_halt_reason`]. + pub fn set_apply_halted(&self, reason: usize) { + self.apply_halt_reason + .store(reason as u64, Ordering::Relaxed); + self.apply_halted.store(1, Ordering::Relaxed); + } + + /// Record that an owed sequencer entry of `kind` was re-proposed. + /// + /// `kind` must be one of the constants in [`sequencer_propose_kind`]. + pub fn record_sequencer_propose_retry(&self, kind: usize) { + if let Some(counter) = self.sequencer_propose_retry_counts.get(kind) { + counter.fetch_add(1, Ordering::Relaxed); + } + } + + /// Set the in-flight backlog gauge. + pub fn set_intake_backlog(&self, backlog: usize) { + self.intake_backlog.store(backlog as u64, Ordering::Relaxed); + } + /// Record the end-to-end executor txn duration (dispatch → response). /// /// Increments the appropriate histogram bucket and the running sum. @@ -290,6 +403,8 @@ impl SchedulerMetrics { self.catch_up_log_compacted.load(Ordering::Relaxed) ); + self.render_flow_prometheus(&mut out, &label); + out } } @@ -309,6 +424,14 @@ impl Default for SchedulerMetrics { verdict_stall_count: AtomicU64::new(0), catch_up_replayed: AtomicU64::new(0), catch_up_log_compacted: AtomicU64::new(0), + dispatch_deferred_count: AtomicU64::new(0), + dispatch_deferred_depth: AtomicU64::new(0), + intake_gate_closed: AtomicU64::new(0), + intake_backlog: AtomicU64::new(0), + intake_gate_closed_counts: std::array::from_fn(|_| AtomicU64::new(0)), + apply_halted: AtomicU64::new(0), + apply_halt_reason: AtomicU64::new(0), + sequencer_propose_retry_counts: std::array::from_fn(|_| AtomicU64::new(0)), } } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/metrics_flow.rs b/nodedb/src/control/cluster/calvin/scheduler/metrics_flow.rs new file mode 100644 index 000000000..019437c27 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/metrics_flow.rs @@ -0,0 +1,135 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Prometheus rendering of the scheduler's flow metrics: deferred dispatch, +//! the intake gate, the apply halt, and sequencer propose retries. + +use std::fmt::Write as _; +use std::sync::atomic::Ordering; + +use super::metrics::{ + SchedulerMetrics, apply_halt_reason, intake_closure_reason, sequencer_propose_kind, +}; + +impl SchedulerMetrics { + /// Append the flow metrics for the vShard in `label` to `out`. + pub(super) fn render_flow_prometheus(&self, out: &mut String, label: &str) { + let _ = writeln!( + out, + "# HELP nodedb_calvin_dispatch_deferred_total \ + Scheduler dispatches refused at dispatcher capacity and parked for re-send." + ); + let _ = writeln!(out, "# TYPE nodedb_calvin_dispatch_deferred_total counter"); + let _ = writeln!( + out, + "nodedb_calvin_dispatch_deferred_total{{{label}}} {}", + self.dispatch_deferred_count.load(Ordering::Relaxed) + ); + + let _ = writeln!( + out, + "# HELP nodedb_calvin_dispatch_deferred_depth \ + Scheduler requests parked for re-send, waiting for dispatcher capacity." + ); + let _ = writeln!(out, "# TYPE nodedb_calvin_dispatch_deferred_depth gauge"); + let _ = writeln!( + out, + "nodedb_calvin_dispatch_deferred_depth{{{label}}} {}", + self.dispatch_deferred_depth.load(Ordering::Relaxed) + ); + + let _ = writeln!( + out, + "# HELP nodedb_calvin_intake_gate_closed \ + 1 while the scheduler takes no new sequenced input." + ); + let _ = writeln!(out, "# TYPE nodedb_calvin_intake_gate_closed gauge"); + let _ = writeln!( + out, + "nodedb_calvin_intake_gate_closed{{{label}}} {}", + self.intake_gate_closed.load(Ordering::Relaxed) + ); + + let _ = writeln!( + out, + "# HELP nodedb_calvin_intake_backlog \ + Pending, blocked, and dependent-barrier txns in the scheduler." + ); + let _ = writeln!(out, "# TYPE nodedb_calvin_intake_backlog gauge"); + let _ = writeln!( + out, + "nodedb_calvin_intake_backlog{{{label}}} {}", + self.intake_backlog.load(Ordering::Relaxed) + ); + + let _ = writeln!( + out, + "# HELP nodedb_calvin_intake_gate_closed_total \ + Times the scheduler stopped taking new sequenced input, by reason." + ); + let _ = writeln!(out, "# TYPE nodedb_calvin_intake_gate_closed_total counter"); + for (i, &reason_label) in intake_closure_reason::LABELS.iter().enumerate() { + let _ = writeln!( + out, + "nodedb_calvin_intake_gate_closed_total{{{label},reason=\"{reason_label}\"}} {}", + self.intake_gate_closed_counts[i].load(Ordering::Relaxed) + ); + } + + let _ = writeln!( + out, + "# HELP nodedb_calvin_apply_halted \ + 1 once the scheduler halted on an apply error it cannot mark applied, by reason." + ); + let _ = writeln!(out, "# TYPE nodedb_calvin_apply_halted gauge"); + let halted = self.apply_halted.load(Ordering::Relaxed) == 1; + let halt_reason = self.apply_halt_reason.load(Ordering::Relaxed); + for (i, &reason_label) in apply_halt_reason::LABELS.iter().enumerate() { + let value = u64::from(halted && halt_reason == i as u64); + let _ = writeln!( + out, + "nodedb_calvin_apply_halted{{{label},reason=\"{reason_label}\"}} {value}" + ); + } + let _ = writeln!( + out, + "# HELP nodedb_calvin_sequencer_propose_retry_total \ + Owed sequencer entries re-proposed because their effect was not yet applied, by kind." + ); + let _ = writeln!( + out, + "# TYPE nodedb_calvin_sequencer_propose_retry_total counter" + ); + for (i, &kind_label) in sequencer_propose_kind::LABELS.iter().enumerate() { + let _ = writeln!( + out, + "nodedb_calvin_sequencer_propose_retry_total{{{label},kind=\"{kind_label}\"}} {}", + self.sequencer_propose_retry_counts[i].load(Ordering::Relaxed) + ); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn propose_retry_counter_renders_per_kind() { + let m = SchedulerMetrics::new(); + m.record_sequencer_propose_retry(sequencer_propose_kind::VOTE); + m.record_sequencer_propose_retry(sequencer_propose_kind::VOTE); + m.record_sequencer_propose_retry(sequencer_propose_kind::COMPLETION_ACK); + let out = m.render_prometheus(3); + assert!( + out.contains( + "nodedb_calvin_sequencer_propose_retry_total{vshard=\"3\",kind=\"vote\"} 2" + ) + ); + assert!(out.contains( + "nodedb_calvin_sequencer_propose_retry_total{vshard=\"3\",kind=\"completion_ack\"} 1" + )); + assert!(out.contains( + "nodedb_calvin_sequencer_propose_retry_total{vshard=\"3\",kind=\"routing_failed\"} 0" + )); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/mod.rs index 4e7249c27..26126e309 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/mod.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/mod.rs @@ -1,19 +1,25 @@ // SPDX-License-Identifier: BUSL-1.1 pub mod applied_gate; +pub mod applied_mirror; +pub mod cut_floor; pub mod driver; pub mod lock; pub mod metrics; +mod metrics_flow; pub mod recovery; pub use applied_gate::AppliedGate; +pub use applied_mirror::{AppliedMirror, AppliedMirrors}; pub use driver::{ - CalvinReadResultProposal, ReadResultEvent, Scheduler, SchedulerConfig, SchedulerParams, - propose_calvin_read_result, + CalvinReadResultProposal, RaftSequencerProposer, ReadResultEvent, Scheduler, SchedulerConfig, + SchedulerParams, SequencerProposer, propose_calvin_read_result, }; pub use lock::{AcquireOutcome, HotKeyTable, LockKey, LockManager, LockMode, TxnId}; // Existing call sites reference this module as `scheduler::lock_manager::…`; // keep that path stable via an alias while the module lives under `lock/`. pub use lock as lock_manager; pub use metrics::SchedulerMetrics; -pub use recovery::{AppliedRecovery, NOT_YET_APPLIED_EPOCH, read_applied_recovery}; +pub use recovery::{ + AppliedRecovery, NOT_YET_APPLIED_EPOCH, read_applied_recovery, recover_applied, +}; diff --git a/nodedb/src/control/cluster/calvin/scheduler/recovery.rs b/nodedb/src/control/cluster/calvin/scheduler/recovery.rs index ebecaf198..4d61ae75e 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/recovery.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/recovery.rs @@ -123,6 +123,58 @@ pub fn read_applied_recovery(wal: &WalManager, vshard_id: u32) -> crate::Result< }) } +/// This vShard's applied state after a restart: the state the last +/// checkpoint saved in `catalog`, together with the markers the WAL still +/// holds. +/// +/// A checkpoint deletes WAL segments that hold applied markers, while the +/// sequencer log keeps delivering their entries after a restart. Without the +/// saved state the scheduler would take an applied transaction for a new +/// one: its local stage refuses the rows it already wrote, the scheduler +/// halts, and the transaction's completion ack never settles. +pub fn recover_applied( + wal: &WalManager, + catalog: &crate::control::security::catalog::SystemCatalog, + vshard_id: u32, +) -> crate::Result { + let from_wal = read_applied_recovery(wal, vshard_id)?; + let Some(saved) = catalog.load_calvin_applied(vshard_id)? else { + return Ok(from_wal); + }; + Ok(merge_saved(from_wal, saved.fully_applied_epoch, saved.tail)) +} + +/// Fold a saved `(fully_applied_epoch, tail)` into a WAL scan's result. +fn merge_saved( + from_wal: AppliedRecovery, + fully_applied_epoch: u64, + saved_tail: BTreeSet<(u64, u32)>, +) -> AppliedRecovery { + let above_watermark = + |epoch: u64| fully_applied_epoch == NOT_YET_APPLIED_EPOCH || epoch > fully_applied_epoch; + let applied_tail: BTreeSet<(u64, u32)> = from_wal + .applied_tail + .into_iter() + .chain(saved_tail) + .filter(|(epoch, _)| above_watermark(*epoch)) + .collect(); + let mut max_applied_epoch = from_wal.max_applied_epoch; + let candidates = applied_tail + .iter() + .map(|(epoch, _)| *epoch) + .chain((fully_applied_epoch != NOT_YET_APPLIED_EPOCH).then_some(fully_applied_epoch)); + for epoch in candidates { + if max_applied_epoch == NOT_YET_APPLIED_EPOCH || epoch > max_applied_epoch { + max_applied_epoch = epoch; + } + } + AppliedRecovery { + fully_applied_epoch, + applied_tail, + max_applied_epoch, + } +} + /// Decode a WAL record's logical [`RecordType`], stripping the encryption /// flag (bit 31) before comparing. fn record_type_of(record: &WalRecord) -> Option { @@ -173,10 +225,16 @@ mod tests { use crate::types::VShardId; // Epoch 5: position 0 applied, position 1 NOT applied. Epoch 2: pos 0. - wal.append_calvin_applied(VShardId::new(1), 2, 0).unwrap(); - wal.append_calvin_applied(VShardId::new(1), 5, 0).unwrap(); + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(1), 2, 0) + .unwrap(); + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(1), 5, 0) + .unwrap(); // A different vshard (must be ignored). - wal.append_calvin_applied(VShardId::new(2), 99, 0).unwrap(); + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(2), 99, 0) + .unwrap(); wal.sync().unwrap(); let rec = read_applied_recovery(&wal, 1).unwrap(); @@ -205,7 +263,8 @@ mod tests { // Epoch 7 carries two independent positions on this vShard; only // position 0 committed before the crash. - wal.append_calvin_applied(VShardId::new(vshard), 7, 0) + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(vshard), 7, 0) .unwrap(); wal.sync().unwrap(); @@ -229,7 +288,8 @@ mod tests { // A pure-read/empty-ops txn still writes a standalone CalvinApplied // marker at (epoch 1, position 0). - wal.append_calvin_applied(VShardId::new(vshard), 1, 0) + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(vshard), 1, 0) .unwrap(); // A write-bearing Calvin txn journals its applied-marker as a @@ -244,15 +304,19 @@ mod tests { epoch: 1, position: 1, vshard_id: vshard, + collections: Vec::new(), + sum_targets: Vec::new(), }), }; - wal.append_transaction_redo( - TenantId::new(0), - VShardId::new(vshard), - DatabaseId::DEFAULT, - &write_bearing, - ) - .unwrap(); + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) + .append_transaction_redo( + TenantId::new(0), + VShardId::new(vshard), + DatabaseId::DEFAULT, + &write_bearing, + ) + .unwrap(); // A single-shard TransactionRedo (calvin_stamp: None) must be ignored // by Calvin recovery. @@ -264,13 +328,15 @@ mod tests { }], calvin_stamp: None, }; - wal.append_transaction_redo( - TenantId::new(0), - VShardId::new(vshard), - DatabaseId::DEFAULT, - &single_shard, - ) - .unwrap(); + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) + .append_transaction_redo( + TenantId::new(0), + VShardId::new(vshard), + DatabaseId::DEFAULT, + &single_shard, + ) + .unwrap(); wal.sync().unwrap(); @@ -286,4 +352,40 @@ mod tests { assert_eq!(rec.applied_tail.len(), 2); assert_eq!(rec.max_applied_epoch, 1); } + + #[test] + fn saved_state_restores_what_a_truncated_wal_lost() { + let dir = TempDir::new().unwrap(); + let wal = open_wal(&dir); + use crate::types::VShardId; + // The WAL still holds only the marker written after the checkpoint. + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(1), 6, 0) + .unwrap(); + wal.sync().unwrap(); + let catalog_dir = TempDir::new().unwrap(); + let catalog = crate::control::security::catalog::SystemCatalog::open( + &catalog_dir.path().join("system.redb"), + ) + .unwrap(); + catalog + .save_calvin_applied(vec![ + crate::control::security::catalog::calvin_applied::StoredCalvinApplied { + vshard_id: 1, + fully_applied_epoch: 2, + tail: [(4, 1)].into_iter().collect(), + }, + ]) + .unwrap(); + + let rec = recover_applied(&wal, &catalog, 1).unwrap(); + assert_eq!(rec.fully_applied_epoch, 2); + assert!(rec.applied_tail.contains(&(4, 1))); + assert!(rec.applied_tail.contains(&(6, 0))); + assert_eq!(rec.max_applied_epoch, 6); + + // A vShard with nothing saved keeps the plain WAL scan. + let other = recover_applied(&wal, &catalog, 9).unwrap(); + assert_eq!(other, read_applied_recovery(&wal, 9).unwrap()); + } } diff --git a/nodedb/src/control/cluster/data_plane_error_wire.rs b/nodedb/src/control/cluster/data_plane_error_wire.rs index 2ad97269b..958957427 100644 --- a/nodedb/src/control/cluster/data_plane_error_wire.rs +++ b/nodedb/src/control/cluster/data_plane_error_wire.rs @@ -8,9 +8,11 @@ //! fails to compile here until it is mirrored on the wire instead of silently //! degrading to `Internal` and losing its SQLSTATE at the coordinator. -use nodedb_cluster::rpc_codec::{DataPlaneErrorCode, TypedClusterError}; +use nodedb_cluster::rpc_codec::{ + DataPlaneCounterFault, DataPlaneErrorCode, DataPlaneSyncHold, TypedClusterError, +}; -use crate::bridge::envelope::ErrorCode; +use crate::bridge::envelope::{CounterFault, ErrorCode, SyncHold}; /// Map a local-execution [`crate::Error`] to the wire error a remote caller /// receives. @@ -42,14 +44,133 @@ pub(crate) fn execution_error_to_typed(err: crate::Error) -> TypedClusterError { constraint, detail, }, - other => { - let message = other.to_string(); - let code = u32::from(nodedb_types::error::NodeDbError::from(other).code().0); - TypedClusterError::Internal { code, message } - } + // A capacity refusal crosses as its own verdict, so the coordinator + // answers the retryable overload class. + capacity @ crate::Error::DispatchCapacity { .. } => TypedClusterError::DataPlane { + code: DataPlaneErrorCode::DispatchCapacity { + reason: capacity.to_string(), + }, + }, + // Every other error crosses as its public numeric code and message. + // The coordinator renders the SQLSTATE that code maps to. + other @ (crate::Error::TxnOverlayMemoryExceeded { .. } + | crate::Error::RejectedAuthz { .. } + | crate::Error::OffsetRegression { .. } + | crate::Error::ConflictRetry { .. } + | crate::Error::CalvinSerializationConflict + | crate::Error::CalvinParticipantError + | crate::Error::RejectedPrevalidation { .. } + | crate::Error::RetryableRefusal { .. } + | crate::Error::AppendOnlyViolation { .. } + | crate::Error::BalanceViolation { .. } + | crate::Error::MaterializedSumTargetNotFound { .. } + | crate::Error::MaterializedSumResolutionMissing { .. } + | crate::Error::PeriodLocked { .. } + | crate::Error::PeriodLockMisconfigured { .. } + | crate::Error::RetentionViolation { .. } + | crate::Error::LegalHoldActive { .. } + | crate::Error::StateTransitionViolation { .. } + | crate::Error::TransitionCheckViolation { .. } + | crate::Error::TypeGuardViolation { .. } + | crate::Error::TypeMismatch { .. } + | crate::Error::InsufficientBalance { .. } + | crate::Error::RateExceeded { .. } + | crate::Error::CollectionNotFound { .. } + | crate::Error::DocumentNotFound { .. } + | crate::Error::CollectionDeactivated { .. } + | crate::Error::VShardAdmissionCapacityExceeded { .. } + | crate::Error::CrdtAdmissionRetriesExhausted { .. } + | crate::Error::CrdtAdmissionInvalidPlan { .. } + | crate::Error::CrdtAdmissionCallerFence + | crate::Error::CrdtApplyRequiresAdmission + | crate::Error::CrdtApplyForbiddenInTransaction + | crate::Error::NotInTransactionBlock { .. } + | crate::Error::CrdtAdmissionTimeout { .. } + | crate::Error::NoLeader { .. } + | crate::Error::NotLeader { .. } + | crate::Error::FanOutExceeded { .. } + | crate::Error::CrossCollectionNotColocated { .. } + | crate::Error::SourceFrozen { .. } + | crate::Error::CloneWriteRequiresMaterialize { .. } + | crate::Error::BadRequest { .. } + | crate::Error::BackupTenantMismatch { .. } + | crate::Error::BackupKeyMismatch + | crate::Error::QuotaOvercommit { .. } + | crate::Error::PlanError { .. } + | crate::Error::FeatureNotSupported { .. } + | crate::Error::UndefinedFunction { .. } + | crate::Error::UndefinedObject { .. } + | crate::Error::ObjectNotInPrerequisiteState { .. } + | crate::Error::UndefinedColumn { .. } + | crate::Error::AmbiguousColumn { .. } + | crate::Error::UnknownStrictField { .. } + | crate::Error::DivisionByZero + | crate::Error::DataException { .. } + | crate::Error::InvalidLimitValue { .. } + | crate::Error::RetryableSchemaChanged { .. } + | crate::Error::RetryableLeaderChange { .. } + | crate::Error::GroupQuorumUnavailable { .. } + | crate::Error::GroupMarksUnavailable { .. } + | crate::Error::MetadataLeaderUnavailable + | crate::Error::AuthorizationStateBehind { .. } + | crate::Error::ExecutionLimitExceeded { .. } + | crate::Error::LimitExceeded { .. } + | crate::Error::Wal(_) + | crate::Error::Dispatch { .. } + | crate::Error::Storage { .. } + | crate::Error::ColdStorage { .. } + | crate::Error::Serialization { .. } + | crate::Error::Codec { .. } + | crate::Error::SegmentCorrupted { .. } + | crate::Error::MemoryExhausted { .. } + | crate::Error::Backpressure { .. } + | crate::Error::Crdt(_) + | crate::Error::Io(_) + | crate::Error::Config { .. } + | crate::Error::Encryption { .. } + | crate::Error::Bridge { .. } + | crate::Error::VersionCompat { .. } + | crate::Error::Internal { .. } + | crate::Error::Shaping(_) + | crate::Error::RemoteTyped { .. } + | crate::Error::Ddl(_) + | crate::Error::DescriptorVersionAnomaly { .. } + | crate::Error::CollectionPurgeRowMissing { .. } + | crate::Error::CatalogIntegrityViolation { .. } + | crate::Error::Promql(_) + | crate::Error::DependentObjectsExist { .. } + | crate::Error::RoleInUse { .. } + | crate::Error::CascadeCycle { .. } + | crate::Error::CrossShardInExplicitTransaction + | crate::Error::SequencerUnavailable + | crate::Error::SessionCapExceeded { .. } + | crate::Error::SessionIdleTimeout + | crate::Error::SessionTokenExpired + | crate::Error::SessionKilledByAdmin + | crate::Error::SessionUserDropped + | crate::Error::OidcProviderTenantUnbound + | crate::Error::OidcProviderTenantUnavailable { .. } + | crate::Error::ExternalRoleUndefined { .. } + | crate::Error::OidcNoDefaultDatabase { .. } + | crate::Error::TenantVectorDimExceeded { .. } + | crate::Error::TenantGraphDepthExceeded { .. } + | crate::Error::RoleInheritanceCycle { .. } + | crate::Error::RoleInheritanceDepthExceeded { .. } + | crate::Error::OllpExhausted { .. } + | crate::Error::MirrorReadOnly { .. } + | crate::Error::StaleReadNotLeader { .. }) => numeric_typed(other), } } +/// The wire error for a local error with no typed wire carrier: its public +/// numeric code from `NodeDbError::from(err).code()`, and its message. The +/// coordinator rebuilds it as `Error::RemoteTyped`. +pub(crate) fn numeric_typed(err: crate::Error) -> TypedClusterError { + let message = err.to_string(); + let code = u32::from(nodedb_types::error::NodeDbError::from(err).code().0); + TypedClusterError::Internal { code, message } +} + /// Widen a pointer-width count to the wire's fixed `u64`. fn to_wire_count(value: usize) -> u64 { value as u64 @@ -70,6 +191,22 @@ impl From for DataPlaneErrorCode { } ErrorCode::RejectedPrevalidation { reason } => Self::RejectedPrevalidation { reason }, ErrorCode::RetryableRefusal { reason } => Self::RetryableRefusal { reason }, + ErrorCode::SyncRejected { + violation, + applied_seq, + provenance, + } => Self::SyncRejected { + violation, + applied_seq, + producer_id: provenance.producer_id, + epoch: provenance.epoch, + stream_id: provenance.stream_id, + seq: provenance.seq, + }, + ErrorCode::SyncNotApplied { hold, applied_seq } => Self::SyncNotApplied { + hold: sync_hold_to_wire(hold), + applied_seq, + }, ErrorCode::NotFound => Self::NotFound, ErrorCode::RejectedAuthz { resource } => Self::RejectedAuthz { resource }, ErrorCode::ConflictRetry => Self::ConflictRetry, @@ -114,7 +251,10 @@ impl From for DataPlaneErrorCode { ErrorCode::TypeMismatch { collection, detail } => { Self::TypeMismatch { collection, detail } } - ErrorCode::OverflowError { collection } => Self::OverflowError { collection }, + ErrorCode::CounterFault { collection, fault } => Self::CounterFault { + collection, + fault: counter_fault_to_wire(fault), + }, ErrorCode::InsufficientBalance { collection, detail } => { Self::InsufficientBalance { collection, detail } } @@ -148,6 +288,16 @@ impl From for DataPlaneErrorCode { limit: to_wire_count(limit), }, ErrorCode::DivisionByZero => Self::DivisionByZero, + ErrorCode::UndefinedFunction { name } => Self::UndefinedFunction { name }, + ErrorCode::DataException { detail } => Self::DataException { detail }, + ErrorCode::DispatchCapacity { reason } => Self::DispatchCapacity { reason }, + ErrorCode::ExpiredBeforeExecution => Self::ExpiredBeforeExecution, + ErrorCode::BadRequest { detail } => Self::BadRequest { detail }, + ErrorCode::TransactionRollback { detail } => Self::TransactionRollback { detail }, + ErrorCode::ActiveSqlTransaction { detail } => Self::ActiveSqlTransaction { detail }, + ErrorCode::DependentObjectsExist { object, detail } => { + Self::DependentObjectsExist { object, detail } + } } } } @@ -163,6 +313,27 @@ impl From for ErrorCode { Self::RejectedPrevalidation { reason } } DataPlaneErrorCode::RetryableRefusal { reason } => Self::RetryableRefusal { reason }, + DataPlaneErrorCode::SyncRejected { + violation, + applied_seq, + producer_id, + epoch, + stream_id, + seq, + } => Self::SyncRejected { + violation, + applied_seq, + provenance: nodedb_types::sync::wire::SyncProvenance { + producer_id, + epoch, + stream_id, + seq, + }, + }, + DataPlaneErrorCode::SyncNotApplied { hold, applied_seq } => Self::SyncNotApplied { + hold: sync_hold_from_wire(hold), + applied_seq, + }, DataPlaneErrorCode::NotFound => Self::NotFound, DataPlaneErrorCode::RejectedAuthz { resource } => Self::RejectedAuthz { resource }, DataPlaneErrorCode::ConflictRetry => Self::ConflictRetry, @@ -211,7 +382,10 @@ impl From for ErrorCode { DataPlaneErrorCode::TypeMismatch { collection, detail } => { Self::TypeMismatch { collection, detail } } - DataPlaneErrorCode::OverflowError { collection } => Self::OverflowError { collection }, + DataPlaneErrorCode::CounterFault { collection, fault } => Self::CounterFault { + collection, + fault: counter_fault_from_wire(fault), + }, DataPlaneErrorCode::InsufficientBalance { collection, detail } => { Self::InsufficientBalance { collection, detail } } @@ -249,10 +423,63 @@ impl From for ErrorCode { } } DataPlaneErrorCode::DivisionByZero => Self::DivisionByZero, + DataPlaneErrorCode::UndefinedFunction { name } => Self::UndefinedFunction { name }, + DataPlaneErrorCode::DataException { detail } => Self::DataException { detail }, + DataPlaneErrorCode::DispatchCapacity { reason } => Self::DispatchCapacity { reason }, + DataPlaneErrorCode::ExpiredBeforeExecution => Self::ExpiredBeforeExecution, + DataPlaneErrorCode::BadRequest { detail } => Self::BadRequest { detail }, + DataPlaneErrorCode::TransactionRollback { detail } => { + Self::TransactionRollback { detail } + } + DataPlaneErrorCode::ActiveSqlTransaction { detail } => { + Self::ActiveSqlTransaction { detail } + } + DataPlaneErrorCode::DependentObjectsExist { object, detail } => { + Self::DependentObjectsExist { object, detail } + } } } } +/// The wire form of a sync hold. +fn sync_hold_to_wire(hold: SyncHold) -> DataPlaneSyncHold { + match hold { + SyncHold::Duplicate => DataPlaneSyncHold::Duplicate, + SyncHold::Fenced => DataPlaneSyncHold::Fenced, + SyncHold::Gap { expected } => DataPlaneSyncHold::Gap { expected }, + } +} + +/// The sync hold a wire form names. +fn sync_hold_from_wire(hold: DataPlaneSyncHold) -> SyncHold { + match hold { + DataPlaneSyncHold::Duplicate => SyncHold::Duplicate, + DataPlaneSyncHold::Fenced => SyncHold::Fenced, + DataPlaneSyncHold::Gap { expected } => SyncHold::Gap { expected }, + } +} + +/// The wire form of a counter fault. Both types live in other crates, so the +/// mapping is a function, not a `From` impl. +fn counter_fault_to_wire(fault: CounterFault) -> DataPlaneCounterFault { + match fault { + CounterFault::NotAnInteger => DataPlaneCounterFault::NotAnInteger, + CounterFault::NotAFloat => DataPlaneCounterFault::NotAFloat, + CounterFault::IntegerOverflow => DataPlaneCounterFault::IntegerOverflow, + CounterFault::NonFinite => DataPlaneCounterFault::NonFinite, + } +} + +/// The counter fault a wire form carries. +fn counter_fault_from_wire(fault: DataPlaneCounterFault) -> CounterFault { + match fault { + DataPlaneCounterFault::NotAnInteger => CounterFault::NotAnInteger, + DataPlaneCounterFault::NotAFloat => CounterFault::NotAFloat, + DataPlaneCounterFault::IntegerOverflow => CounterFault::IntegerOverflow, + DataPlaneCounterFault::NonFinite => CounterFault::NonFinite, + } +} + #[cfg(test)] mod tests { use super::*; @@ -300,6 +527,61 @@ mod tests { } } + #[test] + fn dispatch_capacity_code_roundtrips_verbatim() { + let original = ErrorCode::DispatchCapacity { + reason: "tenant 1 holds 64/64 in-flight requests".into(), + }; + let wire = DataPlaneErrorCode::from(original.clone()); + assert_eq!( + wire, + DataPlaneErrorCode::DispatchCapacity { + reason: "tenant 1 holds 64/64 in-flight requests".into(), + } + ); + assert_eq!(ErrorCode::from(wire), original); + } + + #[test] + fn sync_not_applied_roundtrips_verbatim() { + for hold in [ + SyncHold::Duplicate, + SyncHold::Fenced, + SyncHold::Gap { expected: 7 }, + ] { + let original = ErrorCode::SyncNotApplied { + hold, + applied_seq: 6, + }; + let wire = DataPlaneErrorCode::from(original.clone()); + assert_eq!(ErrorCode::from(wire), original); + } + } + + #[test] + fn expired_before_execution_roundtrips_verbatim() { + let wire = DataPlaneErrorCode::from(ErrorCode::ExpiredBeforeExecution); + assert_eq!(wire, DataPlaneErrorCode::ExpiredBeforeExecution); + assert_eq!(ErrorCode::from(wire), ErrorCode::ExpiredBeforeExecution); + } + + #[test] + fn counter_fault_roundtrips_verbatim() { + for fault in [ + CounterFault::NotAnInteger, + CounterFault::NotAFloat, + CounterFault::IntegerOverflow, + CounterFault::NonFinite, + ] { + let original = ErrorCode::CounterFault { + collection: "counters".into(), + fault, + }; + let wire = DataPlaneErrorCode::from(original.clone()); + assert_eq!(ErrorCode::from(wire), original); + } + } + #[test] fn counted_code_roundtrips_across_the_u64_wire_field() { let original = ErrorCode::RecursionDepthExceeded { diff --git a/nodedb/src/control/cluster/metadata_applier/catalog_ddl.rs b/nodedb/src/control/cluster/metadata_applier/catalog_ddl.rs index a1e95c75b..56d491cdd 100644 --- a/nodedb/src/control/cluster/metadata_applier/catalog_ddl.rs +++ b/nodedb/src/control/cluster/metadata_applier/catalog_ddl.rs @@ -107,9 +107,12 @@ impl MetadataCommitApplier { } debug!(kind = stamped.kind(), "catalog_entry: applying to redb"); - if !catalog_entry::apply::apply_to(&stamped, catalog)? { + let outcome = catalog_entry::apply::apply_to(&stamped, catalog)?; + if !outcome.wrote() { // A `Put*` that wrote nothing (e.g. an if-absent create for a - // descriptor that already exists) still concludes its DDL. + // descriptor that already exists) still concludes its DDL. So + // does a refused entry: every node refuses it at this position, + // and the proposer reports the refusal to its client. self.clear_implicit_drain(&stamped); return Ok(()); } @@ -143,9 +146,7 @@ impl MetadataCommitApplier { // the commit. emit_ddl_audit(&shared, raft_index, &stamped, audit.as_ref()); - catalog_entry::post_apply::spawn_post_apply_async_side_effects( - stamped, shared, raft_index, - ); + catalog_entry::post_apply::spawn_post_apply_async_side_effects(stamped, shared); } Ok(()) } diff --git a/nodedb/src/control/cluster/metadata_applier/wedge.rs b/nodedb/src/control/cluster/metadata_applier/wedge.rs index f27ca69a5..f4e8c1c57 100644 --- a/nodedb/src/control/cluster/metadata_applier/wedge.rs +++ b/nodedb/src/control/cluster/metadata_applier/wedge.rs @@ -62,7 +62,111 @@ pub fn classify(error: &crate::Error) -> ApplyFailureClass { crate::Error::BadRequest { .. } | crate::Error::TypeMismatch { .. } => { ApplyFailureClass::Permanent } - _ => ApplyFailureClass::Transient, + // Not provably a pure function of the entry and persisted state. + crate::Error::RejectedConstraint { .. } + | crate::Error::TxnOverlayMemoryExceeded { .. } + | crate::Error::RejectedAuthz { .. } + | crate::Error::OffsetRegression { .. } + | crate::Error::DeadlineExceeded { .. } + | crate::Error::ConflictRetry { .. } + | crate::Error::CalvinSerializationConflict + | crate::Error::CalvinParticipantError + | crate::Error::RejectedPrevalidation { .. } + | crate::Error::RetryableRefusal { .. } + | crate::Error::AppendOnlyViolation { .. } + | crate::Error::BalanceViolation { .. } + | crate::Error::MaterializedSumTargetNotFound { .. } + | crate::Error::MaterializedSumResolutionMissing { .. } + | crate::Error::PeriodLocked { .. } + | crate::Error::PeriodLockMisconfigured { .. } + | crate::Error::RetentionViolation { .. } + | crate::Error::LegalHoldActive { .. } + | crate::Error::StateTransitionViolation { .. } + | crate::Error::TransitionCheckViolation { .. } + | crate::Error::TypeGuardViolation { .. } + | crate::Error::InsufficientBalance { .. } + | crate::Error::RateExceeded { .. } + | crate::Error::CollectionNotFound { .. } + | crate::Error::DocumentNotFound { .. } + | crate::Error::CollectionDeactivated { .. } + | crate::Error::VShardAdmissionCapacityExceeded { .. } + | crate::Error::CrdtAdmissionRetriesExhausted { .. } + | crate::Error::CrdtAdmissionInvalidPlan { .. } + | crate::Error::CrdtAdmissionCallerFence + | crate::Error::CrdtApplyRequiresAdmission + | crate::Error::CrdtApplyForbiddenInTransaction + | crate::Error::NotInTransactionBlock { .. } + | crate::Error::CrdtAdmissionTimeout { .. } + | crate::Error::NoLeader { .. } + | crate::Error::NotLeader { .. } + | crate::Error::FanOutExceeded { .. } + | crate::Error::CrossCollectionNotColocated { .. } + | crate::Error::SourceFrozen { .. } + | crate::Error::CloneWriteRequiresMaterialize { .. } + | crate::Error::BackupTenantMismatch { .. } + | crate::Error::BackupKeyMismatch + | crate::Error::QuotaOvercommit { .. } + | crate::Error::PlanError { .. } + | crate::Error::FeatureNotSupported { .. } + | crate::Error::UndefinedFunction { .. } + | crate::Error::UndefinedObject { .. } + | crate::Error::ObjectNotInPrerequisiteState { .. } + | crate::Error::UndefinedColumn { .. } + | crate::Error::AmbiguousColumn { .. } + | crate::Error::UnknownStrictField { .. } + | crate::Error::DivisionByZero + | crate::Error::DataException { .. } + | crate::Error::InvalidLimitValue { .. } + | crate::Error::RetryableSchemaChanged { .. } + | crate::Error::RetryableLeaderChange { .. } + | crate::Error::GroupQuorumUnavailable { .. } + | crate::Error::GroupMarksUnavailable { .. } + | crate::Error::MetadataLeaderUnavailable + | crate::Error::AuthorizationStateBehind { .. } + | crate::Error::ExecutionLimitExceeded { .. } + | crate::Error::LimitExceeded { .. } + | crate::Error::Wal(_) + | crate::Error::Dispatch { .. } + | crate::Error::DispatchCapacity { .. } + | crate::Error::Storage { .. } + | crate::Error::ColdStorage { .. } + | crate::Error::SegmentCorrupted { .. } + | crate::Error::MemoryExhausted { .. } + | crate::Error::Backpressure { .. } + | crate::Error::Crdt(_) + | crate::Error::Io(_) + | crate::Error::Config { .. } + | crate::Error::Encryption { .. } + | crate::Error::Bridge { .. } + | crate::Error::VersionCompat { .. } + | crate::Error::Internal { .. } + | crate::Error::Shaping(_) + | crate::Error::Ddl(_) + | crate::Error::RemoteTyped { .. } + | crate::Error::CollectionPurgeRowMissing { .. } + | crate::Error::DataPlane(_) + | crate::Error::Promql(_) + | crate::Error::DependentObjectsExist { .. } + | crate::Error::RoleInUse { .. } + | crate::Error::CascadeCycle { .. } + | crate::Error::CrossShardInExplicitTransaction + | crate::Error::SequencerUnavailable + | crate::Error::SessionCapExceeded { .. } + | crate::Error::SessionIdleTimeout + | crate::Error::SessionTokenExpired + | crate::Error::SessionKilledByAdmin + | crate::Error::SessionUserDropped + | crate::Error::OidcProviderTenantUnbound + | crate::Error::OidcProviderTenantUnavailable { .. } + | crate::Error::ExternalRoleUndefined { .. } + | crate::Error::OidcNoDefaultDatabase { .. } + | crate::Error::TenantVectorDimExceeded { .. } + | crate::Error::TenantGraphDepthExceeded { .. } + | crate::Error::RoleInheritanceCycle { .. } + | crate::Error::RoleInheritanceDepthExceeded { .. } + | crate::Error::OllpExhausted { .. } + | crate::Error::MirrorReadOnly { .. } + | crate::Error::StaleReadNotLeader { .. } => ApplyFailureClass::Transient, } } diff --git a/nodedb/src/control/cluster/mod.rs b/nodedb/src/control/cluster/mod.rs index c704618ff..547f495a5 100644 --- a/nodedb/src/control/cluster/mod.rs +++ b/nodedb/src/control/cluster/mod.rs @@ -2,23 +2,8 @@ //! Cluster mode startup and integration. //! -//! Bridges `nodedb-cluster` (Raft, transport, routing, metadata group) -//! into the main server. Split into one concern per file: -//! -//! - [`init`] — cluster startup (transport, catalog, bootstrap/join/restart). -//! - [`start_raft`] — Raft event loop + RPC server + applier wiring. -//! - [`handle`] — the `ClusterHandle` passed between init and start_raft. -//! - [`core_stall`] — samples every Data Plane core's event-loop -//! liveness counter and marks the cores that stopped advancing. -//! - [`decommission_bridge`] — drives `nodedb-cluster`'s per-node -//! decommission signal into this process's `ShutdownWatch`. -//! - [`spsc_applier`] — committed data-group entries → SPSC bridge. -//! - [`metadata_applier`] — committed metadata-group entries → -//! `MetadataCache` + optional redb writeback. The per-Raft-group -//! apply watermark watchers themselves now live in -//! [`nodedb_cluster::GroupAppliedWatchers`] and are bumped from -//! the Raft tick loop so every group (metadata + data) shares one -//! primitive. +//! Bridges `nodedb-cluster` (Raft, transport, routing, metadata group) into +//! the main server. Each sub-module documents its own concern. pub mod array_cluster_exec; pub mod array_cluster_helpers; @@ -54,7 +39,7 @@ pub use init::{init_cluster, init_cluster_with_transport, init_single_node_calvi pub use metadata_applier::MetadataCommitApplier; pub use read_index::{MultiRaftReadGate, RaftReadGate, ReadIndexRefusal}; pub use recovery_check::{VerifyReport, verify_and_repair}; -pub use sequencer_halt::SequencerHaltMarker; +pub use sequencer_halt::{CalvinApplyHalt, CalvinApplyHaltMarker, SequencerHaltMarker}; pub use spsc_applier::SpscCommitApplier; pub use start_raft::start_raft; pub use tls::resolve_credentials; diff --git a/nodedb/src/control/cluster/read_index.rs b/nodedb/src/control/cluster/read_index.rs index 20a97a19d..28a93ed20 100644 --- a/nodedb/src/control/cluster/read_index.rs +++ b/nodedb/src/control/cluster/read_index.rs @@ -12,7 +12,7 @@ //! table to decide the read belonged here, so it builds the redirect from //! what it knows rather than having a second, staler answer passed back. -use std::sync::{Arc, Mutex}; +use std::sync::{Arc, Mutex, Weak}; use std::time::Duration; use async_trait::async_trait; @@ -43,16 +43,87 @@ pub trait RaftReadGate: Send + Sync { /// Whether this node's replica of `group_id` is within `max_staleness` of /// the leader. Local state only — no quorum round, so this does not block. fn within_staleness_bound(&self, group_id: u64, max_staleness: Duration) -> bool; + + /// A read index for `group_id` on any node: confirmed here when this node + /// leads the group, asked of the leader otherwise. + /// + /// Once this node has applied the group through the returned index, its + /// state holds every entry committed before the call. A gate whose node + /// always leads its groups answers with [`Self::confirm_leader`]. + async fn read_index(&self, group_id: u64, timeout: Duration) -> Result { + self.confirm_leader(group_id, timeout).await + } } +/// The Raft loop type the production gate forwards read-index requests +/// through. +type GateRaftLoop = nodedb_cluster::RaftLoop< + crate::control::cluster::spsc_applier::SpscCommitApplier, + crate::control::LocalPlanExecutor, +>; + /// Production implementation, backed by the Raft loop's coordinator. +/// +/// Holds the loop weakly: the loop keeps `SharedState` alive, and the gate +/// lives on `SharedState`, so a strong reference would pin both. pub struct MultiRaftReadGate { multi_raft: Arc>, + raft_loop: Weak, } impl MultiRaftReadGate { - pub fn new(multi_raft: Arc>) -> Self { - Self { multi_raft } + pub fn new(multi_raft: Arc>, raft_loop: Weak) -> Self { + Self { + multi_raft, + raft_loop, + } + } +} + +/// The refusal a read-index error means to the caller. +fn refusal_of(error: ClusterError) -> ReadIndexRefusal { + match error { + ClusterError::ReadIndexTimeout { waited_ms, .. } => ReadIndexRefusal::Timeout { waited_ms }, + // Not hosted here, not leading, leadership lost mid-probe, or the + // leader unreachable: the caller asks again later. Every other error + // also leaves this node unable to prove its leadership now. + ClusterError::Raft(_) + | ClusterError::VShardNotMapped { .. } + | ClusterError::GroupNotFound { .. } + | ClusterError::LearnerNotCaughtUp { .. } + | ClusterError::MigrationInProgress { .. } + | ClusterError::MigrationPauseBudgetExceeded { .. } + | ClusterError::NodeUnreachable { .. } + | ClusterError::GhostNotFound { .. } + | ClusterError::Transport { .. } + | ClusterError::ShardTimeout { .. } + | ClusterError::StreamTerminal { .. } + | ClusterError::Storage { .. } + | ClusterError::DataPlane { .. } + | ClusterError::Codec { .. } + | ClusterError::UnsupportedWireVersion { .. } + | ClusterError::CircuitOpen { .. } + | ClusterError::JoinGroupDisappeared { .. } + | ClusterError::JoinCommitTimeout { .. } + | ClusterError::ReadIndexNotLeader { .. } + | ClusterError::Config { .. } + | ClusterError::MigrationCheckpoint(_) + | ClusterError::MigrationRecovery(_) + | ClusterError::WrongOwner { .. } + | ClusterError::Calvin(_) + | ClusterError::SnapshotCrcMismatch { .. } + | ClusterError::SnapshotOffsetRegression { .. } + | ClusterError::PartialSnapshotCorrupt { .. } + | ClusterError::PartialSnapshotCleanupFailed { .. } + | ClusterError::SnapshotApplyFailed { .. } + | ClusterError::Mirror(_) + | ClusterError::BspBarrier(_) + | ClusterError::VectorGather(_) + | ClusterError::SpatialGather(_) + | ClusterError::Bm25Gather(_) + | ClusterError::TsGather(_) + | ClusterError::RemoteUntyped { .. } + | ClusterError::ShardExecution { .. } => ReadIndexRefusal::NotLeader, } } @@ -63,16 +134,20 @@ impl RaftReadGate for MultiRaftReadGate { group_id: u64, timeout: Duration, ) -> Result { - match nodedb_cluster::confirm_read_index(&self.multi_raft, group_id, timeout).await { - Ok(index) => Ok(index), - Err(ClusterError::ReadIndexTimeout { waited_ms, .. }) => { - Err(ReadIndexRefusal::Timeout { waited_ms }) - } - // Every other path out of the confirmation — not hosted here, not - // leading, leadership lost mid-probe — means the same thing to the - // caller: ask the leader instead. - Err(_) => Err(ReadIndexRefusal::NotLeader), - } + nodedb_cluster::confirm_read_index(&self.multi_raft, group_id, timeout) + .await + .map_err(refusal_of) + } + + async fn read_index(&self, group_id: u64, timeout: Duration) -> Result { + let Some(raft_loop) = self.raft_loop.upgrade() else { + // The loop is gone: the node is shutting down. + return Err(ReadIndexRefusal::NotLeader); + }; + raft_loop + .read_index_via_leader(group_id, timeout) + .await + .map_err(refusal_of) } fn within_staleness_bound(&self, group_id: u64, max_staleness: Duration) -> bool { diff --git a/nodedb/src/control/cluster/sequencer_halt.rs b/nodedb/src/control/cluster/sequencer_halt.rs index 279c5d2a2..ff4d9e466 100644 --- a/nodedb/src/control/cluster/sequencer_halt.rs +++ b/nodedb/src/control/cluster/sequencer_halt.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Node-wide marker for a halted Calvin sequencer state machine. +//! Node-wide markers for a halted Calvin sequencer state machine and a halted +//! Calvin scheduler. //! //! The sequencer stops applying epoch batches when a NEW committed entry //! re-mints an epoch this replica already consumed — the one divergence that @@ -16,6 +17,11 @@ //! fast instead of hanging, and this marker makes the degradation visible on the //! same surfaces a wedged metadata applier uses, so it can never be mistaken for //! a healthy node. +//! +//! A Calvin scheduler halts one vShard when a replica-local error leaves a +//! sequenced txn neither applied nor identically aborted on this replica. The +//! same scoping holds: the node keeps serving, and [`CalvinApplyHaltMarker`] +//! makes the lost vShard visible on the same surfaces. use std::sync::OnceLock; @@ -29,6 +35,7 @@ use nodedb_cluster::calvin::SequencerHalt; #[derive(Debug, Default)] pub struct SequencerHaltMarker { halt: OnceLock, + apply: CalvinApplyHaltMarker, } impl SequencerHaltMarker { @@ -45,6 +52,49 @@ impl SequencerHaltMarker { pub fn is_halted(&self) -> bool { self.halt.get().is_some() } + + /// The node-wide marker for halted Calvin schedulers. + pub fn apply_halt(&self) -> &CalvinApplyHaltMarker { + &self.apply + } +} + +/// Why one vShard's Calvin scheduler stopped applying sequenced txns. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct CalvinApplyHalt { + pub vshard_id: u32, + /// Epoch of the txn held unapplied. + pub epoch: u64, + /// Position of the txn held unapplied. + pub position: u32, + /// Halt reason label, as on `nodedb_calvin_apply_halted`. + pub reason: &'static str, + /// Sub-operation of the txn that failed. + pub step: &'static str, + pub error: String, +} + +/// First-writer-wins record of the first Calvin scheduler that halted on this +/// node. A halted scheduler stays halted until restart, so nothing clears it. +#[derive(Debug, Default)] +pub struct CalvinApplyHaltMarker { + halt: OnceLock, +} + +impl CalvinApplyHaltMarker { + /// Record the first halt. Later calls are ignored. + pub fn record(&self, halt: CalvinApplyHalt) { + let _ = self.halt.set(halt); + } + + /// The recorded halt, if a scheduler on this node halted. + pub fn report(&self) -> Option<&CalvinApplyHalt> { + self.halt.get() + } + + pub fn is_halted(&self) -> bool { + self.halt.get().is_some() + } } #[cfg(test)] @@ -67,6 +117,31 @@ mod tests { assert!(marker.report().is_none()); } + fn apply_halt(vshard_id: u32) -> CalvinApplyHalt { + CalvinApplyHalt { + vshard_id, + epoch: 9, + position: 1, + reason: "flush_failed", + step: "flush", + error: "flush returned Error".to_string(), + } + } + + #[test] + fn apply_marker_keeps_the_first_recorded_halt() { + let marker = SequencerHaltMarker::default(); + assert!(!marker.apply_halt().is_halted()); + marker.apply_halt().record(apply_halt(3)); + marker.apply_halt().record(apply_halt(4)); + assert!(marker.apply_halt().is_halted()); + assert_eq!(marker.apply_halt().report().map(|h| h.vshard_id), Some(3)); + assert!( + !marker.is_halted(), + "a scheduler halt leaves the sequencer halt clear" + ); + } + #[test] fn marker_keeps_the_first_recorded_halt() { let marker = SequencerHaltMarker::default(); diff --git a/nodedb/src/control/cluster/snapshot_applier.rs b/nodedb/src/control/cluster/snapshot_applier.rs index 098836a69..15784306d 100644 --- a/nodedb/src/control/cluster/snapshot_applier.rs +++ b/nodedb/src/control/cluster/snapshot_applier.rs @@ -29,7 +29,7 @@ use std::time::Duration; use nodedb_cluster::routing::vshard_for_collection; use nodedb_types::Surrogate; -use nodedb_types::id::DatabaseId; +use nodedb_types::id::{CollectionKey, DatabaseId, QualifiedCollection}; use crate::bridge::envelope::PhysicalPlan; use crate::control::state::SharedState; @@ -100,26 +100,39 @@ impl nodedb_cluster::SnapshotApplier for DataPlaneSnapshotApplier { // catalog access, so this resolution is Control-Plane (applier) only. let mut clear_vshards: Vec = group_vshards.iter().copied().collect(); clear_vshards.sort_unstable(); - let mut collections_to_clear: Vec<(u64, String)> = Vec::new(); + // + // Every database's collections are listed. Each entry names its + // database and the collection as the Data Plane stores it there. + let mut collections_to_clear: Vec<(u64, u64, String)> = Vec::new(); if !group_vshards.is_empty() { let catalog = self.shared.credentials.catalog(); let collections = catalog - .load_all_collections(DatabaseId::DEFAULT) + .load_all_collections_across_databases() .map_err(|e| Box::new(e) as Box)?; for coll in collections.iter().filter(|c| { c.is_active - && group_vshards.contains(&vshard_for_collection(DatabaseId::DEFAULT, &c.name)) + && group_vshards.contains(&vshard_for_collection(CollectionKey::from_bare( + c.database_id, + &c.name, + ))) }) { - collections_to_clear.push((coll.tenant_id, coll.name.clone())); + collections_to_clear.push(( + coll.database_id.as_u64(), + coll.tenant_id, + QualifiedCollection::new(coll.database_id, &coll.name) + .as_str() + .to_string(), + )); } } // Reuse the existing local restore handler with replace_mode = true so a // Raft install OVERWRITES present keys. The handler installs by the - // snapshot's own per-key tenant/db prefixes, so the `tenant_id` plan - // field is only the dispatch routing key — mirror the local RESTORE - // dispatch (DEFAULT db, "__system" collection). Tenant 0 is used as the - // representative routing tenant for the multi-tenant payload. + // snapshot's own per-entry database and tenant, so the `tenant_id` plan + // field and the dispatch database are only routing keys — mirror the + // local RESTORE dispatch (DEFAULT db, "__system" collection). Tenant 0 + // is used as the representative routing tenant for the multi-tenant, + // multi-database payload. let plan = PhysicalPlan::Meta(MetaOp::RestoreTenantSnapshot { tenant_id: 0, snapshot: snapshot_bytes.to_vec(), @@ -133,8 +146,7 @@ impl nodedb_cluster::SnapshotApplier for DataPlaneSnapshotApplier { crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( crate::control::server::shared::ddl::sync_dispatch::SystemReason::ClusterSnapshot, TenantId::new(0), - DatabaseId::DEFAULT, - "__system", + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "__system"), plan, ), SNAPSHOT_APPLY_TIMEOUT, @@ -158,9 +170,8 @@ impl nodedb_cluster::SnapshotApplier for DataPlaneSnapshotApplier { for e in &snap.surrogate_pk { catalog .put_surrogate( - DatabaseId::DEFAULT, + CollectionKey::from_bare(DatabaseId::new(e.database_id), &e.collection), TenantId::new(e.tenant_id), - &e.collection, &e.pk, Surrogate::new(e.surrogate), ) @@ -168,6 +179,22 @@ impl nodedb_cluster::SnapshotApplier for DataPlaneSnapshotApplier { } } + // Take the group's tenant write marks, durably, before this node + // reports the group applied through the snapshot. + if !snap.group_write_marks.is_empty() { + self.shared + .tenant_marks + .raise_group_entries(group_id, &snap.group_write_marks); + self.shared + .tenant_marks + .persist(self.shared.credentials.catalog()) + .map_err(|err| Box::new(err) as Box)?; + } + + // The install emitted no per-row events, so the permission cache + // reloads before this node reports coverage of the group again. + self.shared.authorization_fence.note_snapshot_installed(); + Ok(()) } } diff --git a/nodedb/src/control/cluster/snapshot_builder.rs b/nodedb/src/control/cluster/snapshot_builder.rs index 2647633f2..754182730 100644 --- a/nodedb/src/control/cluster/snapshot_builder.rs +++ b/nodedb/src/control/cluster/snapshot_builder.rs @@ -9,9 +9,11 @@ //! chunked `InstallSnapshot` RPC for a lagging/new follower. //! //! The build reuses the existing Data-Plane snapshot builder -//! (`MetaOp::CreateTenantSnapshot`) per tenant, then FILTERS every section down -//! to the collections whose vshard belongs to the target Raft group, and merges -//! the per-tenant slices into one `TenantDataSnapshot` for the wire. +//! (`MetaOp::CreateTenantSnapshot`) per tenant and per database the tenant has +//! collections in, then FILTERS every section down to the collections whose +//! vshard belongs to the target Raft group, and merges the slices into one +//! `TenantDataSnapshot` for the wire. Every section entry names its database, +//! so the follower installs each row in the database it came from. //! //! The vshard-partitioned engines are filtered and shipped, including graph //! `edges` (the edge key already embeds the collection, so it is routed through @@ -20,22 +22,20 @@ //! and is shipped to the group that owns that collection's vshard — the same //! per-collection vshard filter as every other section. -use std::collections::HashSet; +use std::collections::{BTreeMap, BTreeSet, HashSet}; use std::sync::Arc; use std::time::Duration; use nodedb_types::id::DatabaseId; use crate::Error; -use crate::bridge::envelope::PhysicalPlan; use crate::control::backup::snapshot_keys::{ - extract_db_scoped_collection, extract_db_tenant_scoped_collection, + extract_db_scoped_collection, extract_db_tenant_scoped_collection, vshard_of_stored, }; use crate::control::security::catalog::SystemCatalog; use crate::control::state::SharedState; use crate::engine::graph::edge_store::parse_versioned_edge_key; use crate::types::{SurrogateBindEntry, TenantDataSnapshot, TenantId}; -use nodedb_physical::physical_plan::MetaOp; /// Per-tenant snapshot dispatch timeout (mirrors the backup orchestrator). const TENANT_SNAPSHOT_TIMEOUT: Duration = Duration::from_secs(120); @@ -52,41 +52,50 @@ impl DataPlaneSnapshotBuilder { Self { shared } } - /// Compute the vshard for a `(DEFAULT db, collection)` pair. - /// - /// One helper, used uniformly by every section's filter so the - /// vshard-of-key logic is never duplicated. Matches the canonical routing - /// function (`vshard_for_collection`) used by the RESTORE topology splitter. - fn vshard_of(collection: &str) -> u32 { - nodedb_cluster::routing::vshard_for_collection(DatabaseId::DEFAULT, collection) + /// Every tenant with an active collection, and the databases it has + /// active collections in. + fn tenant_databases(catalog: &SystemCatalog) -> Result>, Error> { + let mut tenants: BTreeMap> = BTreeMap::new(); + for coll in catalog + .load_all_collections_across_databases()? + .iter() + .filter(|c| c.is_active) + { + tenants + .entry(coll.tenant_id) + .or_default() + .insert(coll.database_id.as_u64()); + } + Ok(tenants) } /// Capture PK→surrogate bindings for every active collection whose vshard - /// belongs to the target group, for each enumerated tenant. + /// belongs to the target group, for each enumerated tenant, in every + /// database. /// - /// Uses the SAME `vshard_of` membership filter every section uses (one - /// source of truth), so only in-group collections' identities ship — never - /// more, never less than the data sections carry. + /// Routes each collection by its `(database, bare name)` key, the same + /// key every section's filter routes by, so only in-group collections' + /// identities ship — never more, never less than the data sections carry. fn capture_surrogates( catalog: &SystemCatalog, - tenants: &[u64], + tenants: &BTreeMap>, group_vshards: &HashSet, merged: &mut TenantDataSnapshot, ) -> Result<(), Error> { - let collections = catalog.load_all_collections(DatabaseId::DEFAULT)?; - let tenant_set: HashSet = tenants.iter().copied().collect(); + let collections = catalog.load_all_collections_across_databases()?; for coll in collections .iter() - .filter(|c| c.is_active && tenant_set.contains(&c.tenant_id)) - .filter(|c| group_vshards.contains(&Self::vshard_of(&c.name))) + .filter(|c| c.is_active && tenants.contains_key(&c.tenant_id)) { - let bindings = catalog.scan_surrogates_for_collection( - DatabaseId::DEFAULT, - TenantId::new(coll.tenant_id), - &coll.name, - )?; + let key = nodedb_types::CollectionKey::from_bare(coll.database_id, &coll.name); + if !group_vshards.contains(&nodedb_cluster::routing::vshard_for_collection(key)) { + continue; + } + let bindings = + catalog.scan_surrogates_for_collection(key, TenantId::new(coll.tenant_id))?; for (pk, surrogate) in bindings { merged.surrogate_pk.push(SurrogateBindEntry { + database_id: coll.database_id.as_u64(), tenant_id: coll.tenant_id, collection: coll.name.clone(), pk, @@ -97,44 +106,41 @@ impl DataPlaneSnapshotBuilder { Ok(()) } - /// Build the merged, group-filtered snapshot for `tenant_id`. + /// Build the group-filtered snapshot of `tenant_id` in `database_id` and + /// merge it into `merged`. async fn build_tenant_filtered( &self, tenant_id: u64, + database_id: DatabaseId, group_vshards: &HashSet, merged: &mut TenantDataSnapshot, ) -> Result<(), Error> { - let plan = PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { tenant_id }); - let bytes = crate::control::server::shared::ddl::sync_dispatch::dispatch_system( + let bytes = crate::control::server::exchange::snapshot_tenant_on_local_cores( &self.shared, - crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( - crate::control::server::shared::ddl::sync_dispatch::SystemReason::ClusterSnapshot, - TenantId::new(tenant_id), - DatabaseId::DEFAULT, - "__system", - plan, - ), + TenantId::new(tenant_id), + database_id, TENANT_SNAPSHOT_TIMEOUT, ) .await?; let snap: TenantDataSnapshot = zerompk::from_msgpack(&bytes).map_err(|e| Error::Internal { - detail: format!("snapshot build: decode tenant {tenant_id} snapshot: {e}"), + detail: format!( + "snapshot build: decode tenant {tenant_id} snapshot of database {}: {e}", + database_id.as_u64() + ), })?; + // Every section names its collection as the Data Plane stores it in + // `database_id`. + let in_group = + |stored: &str| group_vshards.contains(&vshard_of_stored(database_id, stored)); // db-tenant-scoped sections: key shape "{db}:{tid}:{collection}[:suffix]" - let in_group_db_tenant_scoped = |key: &str| { - extract_db_tenant_scoped_collection(key, tenant_id) - .map(|c| group_vshards.contains(&Self::vshard_of(c))) - .unwrap_or(false) - }; + let in_group_db_tenant_scoped = + |key: &str| extract_db_tenant_scoped_collection(key, tenant_id).is_some_and(in_group); // db-scoped sections: key shape "{db}:{tid}:{collection}" (coll may contain ':') - let in_group_db_scoped = |key: &str| { - extract_db_scoped_collection(key, tenant_id) - .map(|c| group_vshards.contains(&Self::vshard_of(c))) - .unwrap_or(false) - }; + let in_group_db_scoped = + |key: &str| extract_db_scoped_collection(key, tenant_id).is_some_and(in_group); for (k, v) in snap.documents { if in_group_db_tenant_scoped(&k) { @@ -146,6 +152,16 @@ impl DataPlaneSnapshotBuilder { merged.indexes.push((k, v)); } } + for (k, v) in snap.documents_versioned { + if in_group_db_tenant_scoped(&k) { + merged.documents_versioned.push((k, v)); + } + } + for (k, v) in snap.indexes_versioned { + if in_group_db_tenant_scoped(&k) { + merged.indexes_versioned.push((k, v)); + } + } for (k, v) in snap.vectors { if in_group_db_tenant_scoped(&k) { merged.vectors.push((k, v)); @@ -156,13 +172,12 @@ impl DataPlaneSnapshotBuilder { merged.timeseries.push((k, v)); } } - // kv_tables: the key IS the collection name → route directly. + // kv_tables / flushed_ts_segments / columnar_engines: db-scoped keys. for (k, v) in snap.kv_tables { - if group_vshards.contains(&Self::vshard_of(&k)) { + if in_group_db_scoped(&k) { merged.kv_tables.push((k, v)); } } - // flushed_ts_segments / columnar_engines: db-scoped keys. for blob in snap.flushed_ts_segments { if in_group_db_scoped(&blob.collection_key) { merged.flushed_ts_segments.push(blob); @@ -186,21 +201,22 @@ impl DataPlaneSnapshotBuilder { // Graph edges: the versioned edge key embeds the collection as its // FIRST `\x00`-delimited component, and edge writes are homed at - // `vshard_for_collection(DEFAULT, collection)` — the SAME routing - // function `Self::vshard_of` uses. So edges route through the identical - // vshard filter every other section uses. The restore path parses the - // key and rebuilds CSR, so no key transformation is needed here. + // the vshard of the collection's `(database, bare name)` key — the SAME + // routing every other section's filter uses. The restore path parses + // the key and rebuilds CSR, so no key transformation is needed here. // - // Unlike every other section, the edge key does NOT carry the tenant, - // so the merged multi-tenant snapshot (applied ONCE with no per-tenant - // dispatch) carries edges tenant-aware via `tenant_edges` — pushing to - // the no-tenant `edges` field here would install them under the wrong - // tenant on apply. + // Unlike every other section, the edge key carries neither the + // database nor the tenant, so the merged snapshot (applied ONCE with no + // per-database or per-tenant dispatch) carries edges via + // `tenant_edges` — pushing to the plain `edges` field here would + // install them under the wrong database and tenant on apply. for (key, value) in snap.edges { match parse_versioned_edge_key(&key) { Some((collection, ..)) => { - if group_vshards.contains(&Self::vshard_of(collection)) { - merged.tenant_edges.push((tenant_id, key, value)); + if in_group(collection) { + merged + .tenant_edges + .push((database_id.as_u64(), tenant_id, key, value)); } } None => { @@ -218,18 +234,16 @@ impl DataPlaneSnapshotBuilder { // single collection; include it iff that collection's vshard belongs to // this group — the same per-collection vshard filter every other engine // uses. - for (database_id, tid, collection, bytes) in snap.crdt_state { - if group_vshards.contains(&Self::vshard_of(&collection)) { - merged - .crdt_state - .push((database_id, tid, collection, bytes)); + for (crdt_db, tid, collection, bytes) in snap.crdt_state { + if in_group(collection.as_str()) { + merged.crdt_state.push((crdt_db, tid, collection, bytes)); } } // CRDT constraints: same per-collection vshard filter as `crdt_state` // — each entry is routed by its single collection's vshard. for entry in snap.crdt_constraints { - if group_vshards.contains(&Self::vshard_of(&entry.collection)) { + if in_group(entry.collection.as_str()) { merged.crdt_constraints.push(entry); } } @@ -264,31 +278,26 @@ impl nodedb_cluster::SnapshotBuilder for DataPlaneSnapshotBuilder { return Ok(Vec::new()); } - // Enumerate tenants from the system catalog — the same source the backup - // orchestrator's catalog sections use. Every active collection carries - // its `tenant_id`; the distinct set is the tenants to snapshot. When no - // catalog is configured there is nothing durable to enumerate, so ship - // an empty (well-formed) snapshot. - let tenants: Vec = { - let catalog = self.shared.credentials.catalog(); - - let collections = catalog - .load_all_collections(DatabaseId::DEFAULT) - .map_err(|e| Box::new(e) as Box)?; - let mut set: HashSet = HashSet::new(); - for coll in collections.iter().filter(|c| c.is_active) { - set.insert(coll.tenant_id); - } - let mut v: Vec = set.into_iter().collect(); - v.sort_unstable(); - v - }; + // Enumerate tenants and their databases from the system catalog — the + // same source the backup orchestrator's catalog sections use. Every + // active collection carries its `tenant_id` and `database_id`; each + // distinct pair is one snapshot to take. With no collection there is + // nothing durable to enumerate, so ship an empty (well-formed) snapshot. + let tenants = Self::tenant_databases(self.shared.credentials.catalog()) + .map_err(|e| Box::new(e) as Box)?; let mut merged = TenantDataSnapshot::default(); - for tenant_id in &tenants { - self.build_tenant_filtered(*tenant_id, &group_vshards, &mut merged) + for (tenant_id, databases) in &tenants { + for database_id in databases { + self.build_tenant_filtered( + *tenant_id, + DatabaseId::new(*database_id), + &group_vshards, + &mut merged, + ) .await .map_err(|e| Box::new(e) as Box)?; + } } // Capture the PK→surrogate identity map for every in-group collection. @@ -303,6 +312,10 @@ impl nodedb_cluster::SnapshotBuilder for DataPlaneSnapshotBuilder { .map_err(|e| Box::new(e) as Box)?; } + // The group's tenant write marks travel with its data: the follower + // that installs the snapshot never applies the entries it covers. + merged.group_write_marks = self.shared.tenant_marks.group_entries(group_id); + // Always return a well-formed serialized struct (even when empty) so the // follower-apply unit receives a decodable payload rather than a stub. let out = zerompk::to_msgpack_vec(&merged).map_err(|e| { diff --git a/nodedb/src/control/cluster/start_raft/group_setup.rs b/nodedb/src/control/cluster/start_raft/group_setup.rs index cdb670232..2fa4da5cc 100644 --- a/nodedb/src/control/cluster/start_raft/group_setup.rs +++ b/nodedb/src/control/cluster/start_raft/group_setup.rs @@ -92,11 +92,10 @@ pub(super) fn build_group_setup( // Build the propose tracker and distributed applier. // - // The tracker is wired with the per-group apply watermark - // registry so every `tracker.complete(group_id, idx, _)` call - // also bumps the watcher — coupling the "data applied on this - // node" signal to the single source of truth that proposers - // and cross-node visibility waits both consume. + // The tracker is wired with the per-group apply watermark registry. The + // apply loop bumps it through the tracker once every entry of a group up + // to an index finished, so proposers and cross-node visibility waits read + // one in-order "data applied on this node" signal. let tracker = Arc::new(ProposeTracker::new().with_group_watchers(handle.group_watchers.clone())); let (dist_applier, apply_rx) = create_distributed_applier(tracker.clone()); diff --git a/nodedb/src/control/cluster/start_raft/loop_build.rs b/nodedb/src/control/cluster/start_raft/loop_build.rs index a1b591409..9607675f3 100644 --- a/nodedb/src/control/cluster/start_raft/loop_build.rs +++ b/nodedb/src/control/cluster/start_raft/loop_build.rs @@ -84,6 +84,34 @@ pub(super) fn build_raft_loop( replication_factor, } = setup; + // The authorization lease runs on the Raft timing of this node. + let lease_timing = crate::control::security::auth_lease::LeaseTiming::from_raft( + multi_raft.election_timeout_min(), + multi_raft.heartbeat_interval(), + )?; + if !shared.authorization_fence.install_timing(lease_timing) { + tracing::warn!( + "authorization lease timing already set — start_raft appears to have run twice" + ); + } + let lease_service = Arc::new( + crate::control::security::auth_lease::LeaderLeaseService::new( + Arc::downgrade(shared), + lease_timing, + ), + ); + if !shared + .authorization_fence + .install_leader(Arc::clone(&lease_service)) + { + tracing::warn!( + "authorization lease service already set — start_raft appears to have run twice" + ); + } + // Authorization coverage of the sequencer group settles each completion + // ack against this node's schedulers. + calvin_completion_registry.applied_acks.enable(); + let raft_loop = Arc::new( nodedb_cluster::RaftLoop::new( multi_raft, @@ -109,6 +137,7 @@ pub(super) fn build_raft_loop( .with_calvin_submit_inbox(hooks.calvin_submit_inbox) .with_reserve_read(hooks.reserve_read) .with_release_reservation(hooks.release_reservation) + .with_auth_lease(lease_service) .with_data_dir(data_dir.to_path_buf()) .with_snapshot_chunk_bytes(snapshot_chunk_bytes) .with_orphan_partial_max_age_secs(orphan_partial_max_age_secs) diff --git a/nodedb/src/control/cluster/start_raft/mod.rs b/nodedb/src/control/cluster/start_raft/mod.rs index da76d588b..b2fc86371 100644 --- a/nodedb/src/control/cluster/start_raft/mod.rs +++ b/nodedb/src/control/cluster/start_raft/mod.rs @@ -22,6 +22,7 @@ mod group_setup; mod hooks; mod loop_build; mod observability; +mod propose_error; mod proposer_wiring; pub use core::start_raft; diff --git a/nodedb/src/control/cluster/start_raft/observability.rs b/nodedb/src/control/cluster/start_raft/observability.rs index 067fdcaee..8ca55225a 100644 --- a/nodedb/src/control/cluster/start_raft/observability.rs +++ b/nodedb/src/control/cluster/start_raft/observability.rs @@ -100,14 +100,36 @@ pub(super) fn finish_observability( // Publish the leadership confirmer so a linearizable read served on this // node proves against a quorum that it is still the leader. Holds the - // same coordinator mutex the loop ticks, not the loop itself. - let gate: Arc = Arc::new( - crate::control::cluster::read_index::MultiRaftReadGate::new(raft_loop.multi_raft_handle()), - ); + // same coordinator mutex the loop ticks, and the loop only weakly, to ask + // a group leader for a read index from a follower. + let gate: Arc = + Arc::new(crate::control::cluster::read_index::MultiRaftReadGate::new( + raft_loop.multi_raft_handle(), + Arc::downgrade(&raft_loop), + )); if shared.raft_read_gate.set(gate).is_err() { tracing::warn!("raft_read_gate already set — start_raft appears to have run twice"); } + // Renew this node's authorization lease with the metadata leader. Its + // confirmed coverage needs the read gate above. + if let Some(timing) = shared.authorization_fence.timing() { + let renew_state = Arc::clone(shared); + crate::control::shutdown::spawn_loop( + &shared.loop_registry, + &shared.shutdown, + "auth_lease_renew", + crate::control::shutdown::ShutdownPhase::DrainingControlPlane, + move |shutdown| { + crate::control::security::auth_lease::renew_loop::run_renew_loop( + renew_state, + timing, + shutdown, + ) + }, + ); + } + // Publish this node's cluster-epoch state so the routing gate can tell // whether this node has missed a topology transition before it coordinates // work on a view of the cluster that may already be superseded. diff --git a/nodedb/src/control/cluster/start_raft/propose_error.rs b/nodedb/src/control/cluster/start_raft/propose_error.rs new file mode 100644 index 000000000..ce7aa5673 --- /dev/null +++ b/nodedb/src/control/cluster/start_raft/propose_error.rs @@ -0,0 +1,123 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The error an async Raft propose returns to its statement. + +use nodedb_cluster::ClusterError; +use nodedb_raft::RaftError; + +use crate::types::VShardId; + +/// The error an async propose returns for a cluster error. +/// +/// A group with no leader to take the proposal right now accepts the same +/// proposal once it has one, so the proposal is retried: +/// [`crate::Error::NoLeader`]. That covers an election, a leadership transfer +/// in flight, a leader that stepped down after this node or a forwarding node +/// chose it, and a vShard whose owner is moving. A forwarded refusal arrives +/// here with its typed Raft error (`DataProposeResponse::refusal_error`). A +/// typed verdict keeps its class. Every other failure is final here. +pub(super) fn async_propose_error(vshard_id: u32, error: ClusterError) -> crate::Error { + match error { + ClusterError::Raft( + RaftError::LeadershipTransferInProgress | RaftError::NotLeader { .. }, + ) + | ClusterError::ReadIndexNotLeader { .. } + | ClusterError::MigrationInProgress { .. } + | ClusterError::WrongOwner { .. } => crate::Error::NoLeader { + vshard_id: VShardId::new(vshard_id), + }, + // The forward did not answer before its timeout. The statement's + // deadline class, the same class the array fan-out gives it. + ClusterError::ShardTimeout { .. } => crate::Error::DeadlineExceeded { + request_id: crate::types::RequestId::new(0), + }, + ClusterError::DataPlane { code } => crate::Error::DataPlane(code.into()), + ClusterError::ShardExecution { error, .. } | ClusterError::StreamTerminal { error, .. } => { + crate::Error::from(*error) + } + other @ (ClusterError::Raft( + RaftError::LogCompacted { .. } + | RaftError::CompactionAheadOfApplied { .. } + | RaftError::ProposalRejected { .. } + | RaftError::InvalidTransferTarget { .. } + | RaftError::GroupNotFound { .. } + | RaftError::Transport { .. } + | RaftError::Storage { .. } + | RaftError::Serialization { .. } + | RaftError::SnapshotFormat { .. } + | RaftError::Shutdown, + ) + | ClusterError::VShardNotMapped { .. } + | ClusterError::GroupNotFound { .. } + | ClusterError::LearnerNotCaughtUp { .. } + | ClusterError::MigrationPauseBudgetExceeded { .. } + | ClusterError::NodeUnreachable { .. } + | ClusterError::GhostNotFound { .. } + | ClusterError::Transport { .. } + | ClusterError::Storage { .. } + | ClusterError::Codec { .. } + | ClusterError::UnsupportedWireVersion { .. } + | ClusterError::CircuitOpen { .. } + | ClusterError::JoinGroupDisappeared { .. } + | ClusterError::JoinCommitTimeout { .. } + | ClusterError::ReadIndexTimeout { .. } + | ClusterError::Config { .. } + | ClusterError::MigrationCheckpoint(_) + | ClusterError::MigrationRecovery(_) + | ClusterError::Calvin(_) + | ClusterError::SnapshotCrcMismatch { .. } + | ClusterError::SnapshotOffsetRegression { .. } + | ClusterError::PartialSnapshotCorrupt { .. } + | ClusterError::PartialSnapshotCleanupFailed { .. } + | ClusterError::SnapshotApplyFailed { .. } + | ClusterError::Mirror(_) + | ClusterError::BspBarrier(_) + | ClusterError::VectorGather(_) + | ClusterError::SpatialGather(_) + | ClusterError::Bm25Gather(_) + | ClusterError::TsGather(_) + | ClusterError::RemoteUntyped { .. }) => crate::Error::Internal { + detail: format!("raft propose (async): {other}"), + }, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_missing_leader_is_retryable() { + let error = ClusterError::Raft(RaftError::NotLeader { leader_hint: None }); + assert!(matches!( + async_propose_error(3, error), + crate::Error::NoLeader { .. } + )); + } + + /// A moving vShard has no owner to take the proposal until the + /// cut-over, so the statement answers the retryable no-leader class. + #[test] + fn a_moving_vshard_is_retryable() { + let error = ClusterError::WrongOwner { + vshard_id: 3, + expected_owner_node: None, + }; + assert!(matches!( + async_propose_error(3, error), + crate::Error::NoLeader { .. } + )); + } + + #[test] + fn a_forward_timeout_is_a_deadline() { + let error = ClusterError::ShardTimeout { + vshard_id: 3, + elapsed_ms: 50, + }; + assert!(matches!( + async_propose_error(3, error), + crate::Error::DeadlineExceeded { .. } + )); + } +} diff --git a/nodedb/src/control/cluster/start_raft/proposer_wiring.rs b/nodedb/src/control/cluster/start_raft/proposer_wiring.rs index 449088cb9..d0023522a 100644 --- a/nodedb/src/control/cluster/start_raft/proposer_wiring.rs +++ b/nodedb/src/control/cluster/start_raft/proposer_wiring.rs @@ -15,6 +15,7 @@ use crate::control::distributed_applier::{ApplyBatch, ProposeTracker, run_apply_ use crate::control::state::SharedState; use super::loop_build::RaftLoopType; +use super::propose_error::async_propose_error; /// Install the sync `raft_proposer` / `raft_compactor` / /// `raft_applied_index_sink`, the async `async_raft_proposer`, and spawn the @@ -147,21 +148,27 @@ pub(super) fn wire_proposers( // Weak for the same cycle-breaking reason as `raft_proposer` above. let raft_loop_async = Arc::downgrade(raft_loop); let tracker_for_proposer = tracker.clone(); - let deadline_secs = shared.tuning.network.default_deadline_secs; + // Held weakly for the same cycle-breaking reason as `raft_proposer` above: + // the proposer lives on `SharedState`. + let state_for_proposer = Arc::downgrade(shared); let async_proposer: Arc = - Arc::new(move |vshard_id, idempotency_key, data| { + Arc::new(move |vshard_id, idempotency_key, data, deadline| { let rl_weak = raft_loop_async.clone(); let tk = tracker_for_proposer.clone(); + let state_weak = state_for_proposer.clone(); Box::pin(async move { let rl = rl_weak.upgrade().ok_or_else(|| crate::Error::Internal { detail: "raft propose (async): cluster not running".into(), })?; - let (group_id, log_index) = rl - .propose_via_data_leader(vshard_id, data) - .await - .map_err(|e| crate::Error::Internal { - detail: format!("raft propose (async): {e}"), - })?; + // The attempt gets only what remains of the caller's deadline. + if tokio::time::Instant::now() >= deadline { + return Err(propose_deadline_exceeded()); + } + let (group_id, log_index) = + tokio::time::timeout_at(deadline, rl.propose_via_data_leader(vshard_id, data)) + .await + .map_err(|_| propose_deadline_exceeded())? + .map_err(|e| async_propose_error(vshard_id, e))?; // Register the waiter with the proposer's idempotency // key. The apply path compares against the committed @@ -171,42 +178,62 @@ pub(super) fn wire_proposers( // `RetryableLeaderChange` instead of leaking a // not-our-payload back to the caller. let rx = tk.register(group_id, log_index, idempotency_key); - tokio::time::timeout(std::time::Duration::from_secs(deadline_secs), rx) - .await - .map_err(|_| crate::Error::Dispatch { - detail: format!( - "raft commit timeout for group {group_id} index {log_index}" - ), - })? - .map_err(|_| crate::Error::Dispatch { - detail: "propose waiter channel closed".into(), - })? - // Preserve `RetryableLeaderChange` so the gateway - // retry loop can re-propose against the new leader - // — wrapping it in `Dispatch` would hide the - // retryable signal and surface as silent INSERT - // success. Only machinery failures stay wrapped for - // diagnostics; a classified apply verdict keeps its - // client-visible classification. - .map_err(|e| { - if crate::error_classify::is_unclassified_failure(&e) { - crate::Error::Dispatch { - detail: format!("apply error: {e}"), - } - } else { - e + let applied = await_local_apply(LocalApplyWait { + state: &state_weak, + tracker: &tk, + group_id, + log_index, + vshard_id, + deadline, + rx, + }) + .await + // Preserve `RetryableLeaderChange` so the gateway + // retry loop can re-propose against the new leader + // — wrapping it in `Dispatch` would hide the + // retryable signal and surface as silent INSERT + // success. Only machinery failures stay wrapped for + // diagnostics; a classified apply verdict keeps its + // client-visible classification. + .map_err(|e| { + if crate::error_classify::is_unclassified_failure(&e) { + crate::Error::Dispatch { + detail: format!("apply error: {e}"), } - }) - // Carry out the write-version the APPLY side stamped, not - // `log_index`. The tracker resolves on the node that applied - // the entry locally, so `write_version` is this replica's own - // post-write `coll_write_lsn` — a WAL LSN, the same domain - // every other feed of that map records in, and the only - // domain the shard-local OCC read validator compares in. The - // raft log index is a per-group counter on a different scale - // entirely; publishing it here made reads validate a WAL LSN - // against a log index. - .map(|applied| (applied.payload, applied.write_version)) + } else { + e + } + }) + // Carry out the write-version the APPLY side stamped, not + // `log_index`. The tracker resolves on the node that applied + // the entry locally, so `write_version` is this replica's own + // post-write `coll_write_lsn` — a WAL LSN, the same domain + // every other feed of that map records in, and the only + // domain the shard-local OCC read validator compares in. The + // raft log index is a per-group counter on a different scale + // entirely; publishing it here made reads validate a WAL LSN + // against a log index. + .map(|applied| (applied.payload, applied.write_version)); + let applied = applied?; + // A write to a vShard homing a permission-tree source is + // acknowledged only once every lease holder covers it, or its + // lease expired. + if let Some(state) = state_weak.upgrade() + && state + .authorization_fence + .sources() + .is_source_vshard(vshard_id) + { + crate::control::security::auth_lease::authorization_barrier( + &state, + vec![nodedb_cluster::GroupCoverage { + group_id, + through: log_index, + }], + ) + .await?; + } + Ok(applied) }) }); crate::control::vshard_admission::install_async_raft_proposer(shared, async_proposer)?; @@ -246,3 +273,113 @@ pub(super) fn wire_proposers( ); Ok(()) } + +/// Where a group's pipeline stands, for a propose waiter that timed out: the +/// Raft commit index, the index handed to the apply loop, and the index the +/// apply loop applied. The first of the three that stops short of the waited +/// index names the stage that stalled. +fn apply_progress(state: Option<&SharedState>, group_id: u64) -> String { + let Some(state) = state else { + return "node is shutting down".to_owned(); + }; + let status = state + .raft_status_fn + .get() + .and_then(|status| status().into_iter().find(|g| g.group_id == group_id)); + let applied = state.applied_index_watcher(group_id).current(); + match status { + Some(group) => format!( + "commit_index={} handed_to_apply_loop={} applied={applied} role={} leader={}", + group.commit_index, group.last_applied, group.role, group.leader_id + ), + None => format!("group not hosted here, applied={applied}"), + } +} + +/// The error a proposal returns once the caller's statement deadline passed. +/// +/// The proposer carries no request id, so the error names request 0. +fn propose_deadline_exceeded() -> crate::Error { + crate::Error::DeadlineExceeded { + request_id: crate::types::RequestId::new(0), + } +} + +/// How often a waiting proposer checks that this node still replicates the +/// entry's group. +const MEMBERSHIP_CHECK: std::time::Duration = std::time::Duration::from_millis(100); + +/// One proposer's wait for this node's apply of its entry. +struct LocalApplyWait<'a> { + state: &'a std::sync::Weak, + tracker: &'a ProposeTracker, + group_id: u64, + log_index: u64, + vshard_id: u32, + deadline: tokio::time::Instant, + rx: tokio::sync::oneshot::Receiver, +} + +/// Wait until this node applied the proposer's entry, and return what the +/// apply produced. +/// +/// The wait ends early when this node leaves the entry's group: a removed +/// replica receives no further entries, so it never applies the index. That +/// ends as [`crate::Error::NotLeader`] naming no leader, which sends the +/// caller to the group's current members. +/// +/// `deadline` is the caller's statement deadline, shared by every attempt. +/// Once it passes, the wait ends as [`crate::Error::DeadlineExceeded`]. The +/// pipeline stage that stalled goes to the log first. +async fn await_local_apply( + wait: LocalApplyWait<'_>, +) -> crate::Result { + let LocalApplyWait { + state, + tracker, + group_id, + log_index, + vshard_id, + deadline, + mut rx, + } = wait; + loop { + let check = tokio::time::sleep( + MEMBERSHIP_CHECK.min(deadline.saturating_duration_since(tokio::time::Instant::now())), + ); + tokio::select! { + received = &mut rx => { + return received.map_err(|_| crate::Error::Dispatch { + detail: "propose waiter channel closed".into(), + })?; + } + _ = check => {} + } + let state = state.upgrade(); + if let Some(state) = state.as_deref() + && !crate::control::security::auth_fence::cluster::hosts_group(state, group_id) + { + tracker.abandon(group_id, log_index); + return Err(crate::Error::NotLeader { + vshard_id: crate::types::VShardId::new(vshard_id), + leader_node: 0, + leader_addr: format!( + "this node left raft group {group_id} before it applied index {log_index}" + ), + }); + } + if tokio::time::Instant::now() >= deadline { + tracker.abandon(group_id, log_index); + tracing::warn!( + group_id, + log_index, + progress = %apply_progress(state.as_deref(), group_id), + oldest_unfinished = %tracker + .applying(group_id) + .map_or_else(|| "nothing".to_owned(), |entry| entry.to_string()), + "raft proposal reached the statement deadline before this node applied it" + ); + return Err(propose_deadline_exceeded()); + } + } +} diff --git a/nodedb/src/control/cluster/start_raft_helpers.rs b/nodedb/src/control/cluster/start_raft_helpers.rs index 9626be8f8..ebcd74c32 100644 --- a/nodedb/src/control/cluster/start_raft_helpers.rs +++ b/nodedb/src/control/cluster/start_raft_helpers.rs @@ -9,9 +9,10 @@ use nodedb_cluster::vshard_handler::{DispatchTarget, dispatch_by_type}; use nodedb_cluster::wire::VShardEnvelope; use crate::control::cluster::calvin::scheduler::metrics::SchedulerMetrics; -use crate::control::cluster::calvin::scheduler::read_applied_recovery; +use crate::control::cluster::calvin::scheduler::recover_applied; use crate::control::cluster::calvin::{ - ReadResultEvent, Scheduler, SchedulerConfig, SchedulerParams, + RaftSequencerProposer, ReadResultEvent, Scheduler, SchedulerConfig, SchedulerParams, + SequencerProposer, }; use crate::control::cluster::handle::ClusterHandle; use crate::control::state::SharedState; @@ -114,6 +115,8 @@ struct ReconcileSchedulersParams<'a> { routing: &'a Arc>, shared: &'a Arc, raft_loop_handle: &'a Arc>, + /// One proposer per node, so its forward limit bounds the whole node. + sequencer_proposer: &'a Arc, sequencer_state_machine: &'a Arc>, calvin_read_result_senders: &'a ReadResultSenders, calvin_completion_registry: &'a Arc, @@ -138,6 +141,7 @@ fn reconcile_vshard_schedulers(params: ReconcileSchedulersParams<'_>) -> crate:: routing, shared, raft_loop_handle, + sequencer_proposer, sequencer_state_machine, calvin_read_result_senders, calvin_completion_registry, @@ -155,7 +159,10 @@ fn reconcile_vshard_schedulers(params: ReconcileSchedulersParams<'_>) -> crate:: continue; } - let recovery = read_applied_recovery(&shared.wal, vshard_id)?; + // The applied state the last checkpoint saved, with the markers the + // WAL still holds: a checkpoint deletes the segments that held older + // markers, and the sequencer log delivers their entries again. + let recovery = recover_applied(&shared.wal, shared.credentials.catalog(), vshard_id)?; let (sequenced_tx, sequenced_rx) = tokio::sync::mpsc::channel(scheduler_config.channel_capacity); @@ -197,13 +204,14 @@ fn reconcile_vshard_schedulers(params: ReconcileSchedulersParams<'_>) -> crate:: // The deterministic lock table is shared between this scheduler and the // Control-Plane write-admission gate: build it once and register the - // SAME `Arc` in `calvin_lock_managers` so a fast-path point write and + // SAME `Arc` in `CalvinLocalState::lock_managers` so a fast-path point write and // this scheduler's validation contend on one mutex. let lock_manager = Arc::new(Mutex::new( crate::control::cluster::calvin::scheduler::lock_manager::LockManager::new(), )); shared - .calvin_lock_managers + .calvin + .lock_managers .lock() .unwrap_or_else(|p| p.into_inner()) .insert(vshard_id, Arc::clone(&lock_manager)); @@ -217,7 +225,8 @@ fn reconcile_vshard_schedulers(params: ReconcileSchedulersParams<'_>) -> crate:: // synchronous `Drop` that must not block. let (promotion_tx, promotion_rx) = tokio::sync::mpsc::unbounded_channel(); shared - .calvin_promotion_senders + .calvin + .promotion_senders .lock() .unwrap_or_else(|p| p.into_inner()) .insert(vshard_id, promotion_tx); @@ -237,6 +246,7 @@ fn reconcile_vshard_schedulers(params: ReconcileSchedulersParams<'_>) -> crate:: receiver: sequenced_rx, shared: Arc::clone(shared), multi_raft: raft_loop_handle.clone(), + sequencer_proposer: Arc::clone(sequencer_proposer), sequencer_state_machine: Arc::clone(sequencer_state_machine), fully_applied_epoch: recovery.fully_applied_epoch, applied_tail: recovery.applied_tail, @@ -321,6 +331,20 @@ pub(super) fn spawn_vshard_schedulers( let node_id = handle.node_id; let routing = Arc::clone(&handle.routing); + let sequencer_proposer: Arc = Arc::new(RaftSequencerProposer::new( + node_id, + Arc::clone(&raft_loop_handle), + shared, + )); + // A backup's cut proposes its marker through the same proposer. + if shared + .calvin + .sequencer_proposer + .set(Arc::clone(&sequencer_proposer)) + .is_err() + { + tracing::warn!("calvin: the sequencer proposer was set already; keeping the first"); + } // Initial reconcile: schedulers for vShards this node already knows it hosts. reconcile_vshard_schedulers(ReconcileSchedulersParams { @@ -328,6 +352,7 @@ pub(super) fn spawn_vshard_schedulers( routing: &routing, shared, raft_loop_handle: &raft_loop_handle, + sequencer_proposer: &sequencer_proposer, sequencer_state_machine, calvin_read_result_senders, calvin_completion_registry, @@ -362,6 +387,7 @@ pub(super) fn spawn_vshard_schedulers( routing: &routing, shared: &shared_task, raft_loop_handle: &raft_loop_handle, + sequencer_proposer: &sequencer_proposer, sequencer_state_machine: &sm_task, calvin_read_result_senders: &rr_task, calvin_completion_registry: ®istry_task, diff --git a/nodedb/src/control/cluster/warm_peers/mod.rs b/nodedb/src/control/cluster/warm_peers/mod.rs index 438209544..02f35296c 100644 --- a/nodedb/src/control/cluster/warm_peers/mod.rs +++ b/nodedb/src/control/cluster/warm_peers/mod.rs @@ -3,10 +3,13 @@ //! Pre-warm the QUIC peer cache after `TransportBind` so the //! first replicated request after boot doesn't pay a cold //! connect. Slots into the startup sequencer between -//! `TransportBind` and `WarmPeers` phases. +//! `TransportBind` and `WarmPeers` phases. Also registers fan-out targets +//! with the transport before an RPC. +pub mod register; pub mod report; pub mod warm; +pub(crate) use register::register_peers_from_topology; pub use report::PeerWarmReport; pub use warm::warm_known_peers; diff --git a/nodedb/src/control/cluster/warm_peers/register.rs b/nodedb/src/control/cluster/warm_peers/register.rs new file mode 100644 index 000000000..cc99ef989 --- /dev/null +++ b/nodedb/src/control/cluster/warm_peers/register.rs @@ -0,0 +1,34 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Register target nodes' addresses with the transport before a fan-out. + +use std::collections::BTreeSet; + +use crate::control::state::SharedState; + +/// Register each target node's address with the transport from the live cluster +/// topology (idempotent). Makes a fan-out robust to a peer the transport has +/// not warmed yet — without it `send_rpc` to an unregistered (but +/// topology-known) node fails with `NodeUnreachable`. Self IS registered too: +/// a coordinator that also owns one of the targets dispatches to itself via +/// `send_rpc`, which loops back through the local QUIC endpoint and runs the +/// same handler (an extra local hop, functionally correct). Missing topology / +/// address for a node is left alone so the subsequent `send_rpc` surfaces the +/// typed `NodeUnreachable` rather than this silently masking it. +pub(crate) fn register_peers_from_topology( + state: &SharedState, + transport: &nodedb_cluster::NexarTransport, + nodes: &BTreeSet, +) { + let Some(topology) = state.cluster_topology.as_ref() else { + return; + }; + let topo = topology.read().unwrap_or_else(|p| p.into_inner()); + for &node in nodes { + if let Some(info) = topo.get_node(node) + && let Some(addr) = info.socket_addr() + { + transport.register_peer(node, addr); + } + } +} diff --git a/nodedb/src/control/crdt_admission.rs b/nodedb/src/control/crdt_admission.rs index 726333486..8dc7b0988 100644 --- a/nodedb/src/control/crdt_admission.rs +++ b/nodedb/src/control/crdt_admission.rs @@ -31,6 +31,9 @@ const FRONTIER_RETRY_LIMIT: usize = 8; pub struct AuthorizedCrdtApplyAdmissionRequest<'a> { pub authorized: AuthorizedTask, + /// The db-qualified collection (`QualifiedCollection::as_str`), the same + /// string the plan's `CrdtOp::Apply` carries. The preview and the apply + /// address the Data Plane by it, and the catalog lookup de-qualifies it. pub collection: &'a str, pub timeout: Duration, pub event_source: EventSource, @@ -40,6 +43,8 @@ pub struct AuthorizedCrdtApplyAdmissionRequest<'a> { pub struct CrdtApplyAdmissionRequest<'a> { pub tenant_id: TenantId, pub database_id: DatabaseId, + /// The db-qualified collection, equal to the plan's `CrdtOp::Apply` + /// collection. pub collection: &'a str, pub plan: PhysicalPlan, pub timeout: Duration, @@ -70,6 +75,7 @@ impl CrdtAdmissionOutcome { pub struct CrdtRestoreAdmissionRequest<'a> { pub tenant_id: TenantId, pub database_id: DatabaseId, + /// The db-qualified collection the generated restore ops address. pub collection: &'a str, pub document_id: &'a str, pub target_version_json: &'a str, @@ -147,31 +153,35 @@ pub async fn dispatch_authorized_crdt_apply_admitted_outcome( .await } +/// `collection` is db-qualified. The catalog keys collections by the bare +/// name, so the lookup de-qualifies it first. fn enforce_external_signing_policy( state: &SharedState, authorized: &AuthorizedTask, collection: &str, ) -> crate::Result<()> { + let bare = crate::control::target_identity::naming::bare_collection_name( + authorized.database_id(), + collection, + ); let stored = state .credentials .catalog() .get_collection( authorized.database_id(), authorized.tenant_id().as_u64(), - collection, + &bare, )? .ok_or_else(|| crate::Error::CollectionNotFound { tenant_id: authorized.tenant_id(), - collection: collection.to_owned(), + collection: bare.clone(), })?; if stored.crdt_signing_required && matches!(authorized.plan(), PhysicalPlan::Crdt(CrdtOp::Apply { .. })) { return Err(crate::Error::RejectedAuthz { tenant_id: authorized.tenant_id(), - resource: format!( - "collection:{collection}:unsigned_crdt_delta_requires_authenticated_sync" - ), + resource: format!("collection:{bare}:unsigned_crdt_delta_requires_authenticated_sync"), }); } Ok(()) @@ -224,7 +234,9 @@ pub(crate) async fn dispatch_crdt_apply_admitted_outcome( }); } }; - let vshard_id = VShardId::from_collection_in_database(database_id, collection); + // `collection` is the plan's database-qualified name. + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?.vshard(); let workflow = CrdtAdmissionWorkflow { state, tenant_id, @@ -286,8 +298,7 @@ async fn preview( crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( crate::control::server::shared::ddl::sync_dispatch::SystemReason::AdmittedContinuation, workflow.tenant_id, - workflow.database_id, - workflow.collection, + nodedb_types::CollectionKey::from_qualified_str(workflow.database_id, workflow.collection)?, PhysicalPlan::Crdt(CrdtOp::PreviewApply { collection: nodedb_types::QualifiedCollection::from_stored( workflow.collection.to_owned(), @@ -415,7 +426,9 @@ pub(crate) async fn dispatch_crdt_restore_admitted( event_source, policy, } = request; - let vshard_id = VShardId::from_collection_in_database(database_id, collection); + // `collection` is the plan's database-qualified name. + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?.vshard(); let workflow = CrdtAdmissionWorkflow { state, tenant_id, @@ -481,8 +494,7 @@ async fn generate_restore_delta( crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( crate::control::server::shared::ddl::sync_dispatch::SystemReason::AdmittedContinuation, workflow.tenant_id, - workflow.database_id, - workflow.collection, + nodedb_types::CollectionKey::from_qualified_str(workflow.database_id, workflow.collection)?, PhysicalPlan::Crdt(CrdtOp::RestoreToVersion { collection: nodedb_types::QualifiedCollection::from_stored( workflow.collection.to_owned(), @@ -515,7 +527,10 @@ async fn apply_fenced( )? .ok_or(crate::Error::CrdtAdmissionInvalidPlan { reason: "admitted CRDT Apply has no replicated form", - })?; + })? + // Every replica gives the write the source this node dispatches it + // with. + .with_event_source(workflow.event_source); let outcome = tokio::time::timeout( workflow.timeout, crate::control::wal_replication::propose_replicated_entry(workflow.state, raw, entry), @@ -525,9 +540,7 @@ async fn apply_fenced( vshard_id: workflow.vshard_id, timeout_ms: timeout_ms(workflow.timeout), })??; - workflow - .state - .advance_tenant_write_hlc(workflow.tenant_id.as_u64()); + // This node's apply of the entry recorded its commit HLC. return Ok(CrdtAdmissionOutcome { payload: outcome.0, write_version: outcome.1, @@ -777,12 +790,7 @@ mod tests { })); let seen = Arc::new(Mutex::new(Vec::new())); let policy = RecordingPolicy { seen, reject: true }; - let before_hlc = state - .tenant_write_hlc - .lock() - .expect("hlc lock") - .get(&1) - .copied(); + let before_hlc = state.tenant_write_mark(1); let result = dispatch_crdt_apply_admitted(&state, admission_request(&policy)).await; responder.await.expect("responder completes"); assert!(matches!( @@ -790,12 +798,7 @@ mod tests { Err(crate::Error::CrdtAdmissionCallerFence) )); assert_eq!( - state - .tenant_write_hlc - .lock() - .expect("hlc lock") - .get(&1) - .copied(), + state.tenant_write_mark(1), before_hlc, "policy rejection must not advance the tenant write HLC" ); @@ -827,7 +830,7 @@ mod tests { let fenced = Arc::new(Mutex::new(Vec::new())); let observed = Arc::clone(&fenced); let raw: Arc = - Arc::new(move |_shard, _key, bytes| { + Arc::new(move |_shard, _key, bytes, _deadline| { let observed = Arc::clone(&observed); Box::pin(async move { let entry = @@ -883,7 +886,7 @@ mod tests { let fences = Arc::new(AtomicUsize::new(0)); let count = Arc::clone(&fences); let raw: Arc = - Arc::new(move |_shard, _key, _bytes| { + Arc::new(move |_shard, _key, _bytes, _deadline| { let count = Arc::clone(&count); Box::pin(async move { if count.fetch_add(1, Ordering::SeqCst) == 0 { @@ -947,7 +950,7 @@ mod tests { let attempts = Arc::new(AtomicUsize::new(0)); let raw: Arc = { let attempts = Arc::clone(&attempts); - Arc::new(move |_shard, _key, _bytes| { + Arc::new(move |_shard, _key, _bytes, _deadline| { let attempts = Arc::clone(&attempts); Box::pin(async move { if attempts.fetch_add(1, Ordering::SeqCst) == 0 { @@ -1015,7 +1018,7 @@ mod tests { let fenced_count = Arc::new(AtomicUsize::new(0)); let count = Arc::clone(&fenced_count); let raw: Arc = - Arc::new(move |_shard, _key, _bytes| { + Arc::new(move |_shard, _key, _bytes, _deadline| { let count = Arc::clone(&count); Box::pin(async move { count.fetch_add(1, Ordering::SeqCst); @@ -1064,7 +1067,7 @@ mod tests { let task = nodedb_physical::physical_task::PhysicalTask { tenant_id, database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::from_collection_in_database(DatabaseId::DEFAULT, "docs"), + vshard_id: nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "docs").vshard(), plan: apply_plan(), post_set_op: nodedb_physical::physical_task::PostSetOp::None, txn_id: None, @@ -1114,7 +1117,7 @@ mod tests { let task = nodedb_physical::physical_task::PhysicalTask { tenant_id, database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::from_collection_in_database(DatabaseId::DEFAULT, "docs"), + vshard_id: nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "docs").vshard(), plan: apply_plan(), post_set_op: nodedb_physical::physical_task::PostSetOp::None, txn_id: None, diff --git a/nodedb/src/control/distributed_applier/applier.rs b/nodedb/src/control/distributed_applier/applier.rs index f2a4ffa96..28a61bc68 100644 --- a/nodedb/src/control/distributed_applier/applier.rs +++ b/nodedb/src/control/distributed_applier/applier.rs @@ -129,69 +129,37 @@ impl CommitApplier for DistributedApplier { } let fresh_last = fresh.last().map(|e| e.index).unwrap_or(last_index); - // Empty entries are Raft leader-transition no-ops, not user - // proposals. A waiter registered at (group_id, idx) was - // proposed by a previous leader at index `idx`; when that - // leader stepped down before the entry committed, the new - // leader's election no-op commits at the same index and - // overwrites it. The proposer's data is GONE — silently - // firing `tracker.complete(Ok([]))` here would tell the - // proposer their INSERT succeeded when in fact it was - // truncated, producing the classic "simple_query returned - // Ok but the row never appears" silent data-loss bug. - // - // Surface the truncation as an explicit error so the gateway - // / caller can retry. Idempotent re-propose is safe because - // the encoded payload carries enough identity (collection, - // PK, surrogate) for the apply path to be replayable. - for entry in &fresh { - if entry.data.is_empty() { - tracing::error!( - group_id, - log_index = entry.index, - "leader-change no-op committed at index where a proposer was waiting; \ - surfacing RetryableLeaderChange so the gateway re-proposes" - ); - // applied_key = 0 (no entry payload to derive a key - // from). The slot fires the explicit - // `RetryableLeaderChange` carried in `result`. - self.tracker.complete( - group_id, - entry.index, - 0, - Err(crate::Error::RetryableLeaderChange { - group_id, - log_index: entry.index, - }), - ); - } + // Empty entries are Raft leader-transition no-ops. They go to the apply + // loop with the rest of the batch: the loop resolves a waiter at a + // no-op's index with `RetryableLeaderChange`, and moves the applied + // watermark past the no-op only once every entry before it finished. + let batch: Vec = fresh.iter().map(|e| (*e).clone()).collect(); + let first_index = batch.first().map(|e| e.index).unwrap_or(fresh_last); + let count = batch.len(); + + // A group whose window is full waits: Raft delivers the batch again on + // a later tick. Every other group keeps its own window. + if !self.tracker.window().try_admit(group_id, count) { + debug!( + group_id, + outstanding = self.tracker.window().outstanding(group_id), + "apply window full, entries will be retried on next tick" + ); + self.release_claim(group_id, fresh_last, first_index.saturating_sub(1)); + return 0; } - let real_entries: Vec = fresh - .iter() - .filter(|e| !e.data.is_empty()) - .map(|e| (*e).clone()) - .collect(); - - let Some(first_real_index) = real_entries.first().map(|e| e.index) else { - // Nothing but no-ops, and they are now fully handled. The claim - // already covers them. - return last_index; - }; - // Push to background task. If the channel is full, log a warning // but don't block the tick loop. if let Err(e) = self.apply_tx.try_send(ApplyBatch { group_id, - entries: real_entries, + entries: batch, }) { warn!(group_id, error = %e, "apply queue full, entries will be retried on next tick"); - // Release the part of the claim that was never handed off. The - // watermark may only cover the no-ops strictly BELOW the first - // rejected entry — those were completed above and must not fire a - // second time. Everything from `first_real_index` up is re-collected - // on the next tick. - self.release_claim(group_id, fresh_last, first_real_index.saturating_sub(1)); + self.tracker.window().release(group_id, count); + // Release the claim: every entry of the batch is re-collected on + // the next tick. + self.release_claim(group_id, fresh_last, first_index.saturating_sub(1)); // Don't advance applied index — entries will be re-delivered. return 0; } @@ -270,24 +238,18 @@ mod tests { } #[test] - fn redelivered_leader_change_noop_does_not_resolve_a_waiter_twice() { - let tracker = Arc::new(ProposeTracker::new()); - let (applier, _rx) = create_distributed_applier(tracker.clone()); + fn a_leader_change_noop_is_handed_off_once() { + let (applier, mut rx) = create_distributed_applier(Arc::new(ProposeTracker::new())); let noop = vec![entry(1, b"")]; - let mut waiter = tracker.register(7, 1, 0); applier.apply_committed(7, &noop); - assert!(matches!( - waiter.try_recv(), - Ok(Err(crate::Error::RetryableLeaderChange { .. })) - )); + let batch = rx.try_recv().expect("the no-op reaches the apply loop"); + assert!(batch.entries[0].data.is_empty()); applier.apply_committed(7, &noop); - let mut probe = tracker.register(7, 1, 0); assert!( - probe.try_recv().is_err(), - "a second completion would park an orphan result on an index whose \ - waiter is already gone" + rx.try_recv().is_err(), + "a second hand-off would resolve the no-op's waiter twice" ); } @@ -312,33 +274,51 @@ mod tests { } #[test] - fn noops_below_a_rejected_entry_are_not_completed_twice() { - let tracker = Arc::new(ProposeTracker::new()); + fn a_rejected_batch_is_handed_off_whole_with_its_noops_on_retry() { let (tx, mut rx) = mpsc::channel(1); - let applier = DistributedApplier::new(tx, tracker.clone()); + let applier = DistributedApplier::new(tx, Arc::new(ProposeTracker::new())); let entries = vec![entry(2, b""), entry(3, b"x")]; applier.apply_committed(7, &[entry(1, b"a")]); - let mut waiter = tracker.register(7, 2, 0); assert_eq!(applier.apply_committed(7, &entries), 0); - assert!(matches!( - waiter.try_recv(), - Ok(Err(crate::Error::RetryableLeaderChange { .. })) - )); rx.try_recv().expect("first batch queued"); assert_eq!(applier.apply_committed(7, &entries), 3); + let batch = rx + .try_recv() + .expect("the rejected entries must be re-accepted"); + let indexes: Vec = batch.entries.iter().map(|e| e.index).collect(); + assert_eq!(indexes, vec![2, 3]); + } - let mut probe = tracker.register(7, 2, 0); - assert!( - probe.try_recv().is_err(), - "the no-op below the rejected entry was already handled" + /// A group past its apply window waits for Raft to deliver it again. A + /// different group still hands its entries off. + #[test] + fn a_group_past_its_window_waits_while_another_group_hands_off() { + let tracker = Arc::new(ProposeTracker::new()); + let (applier, mut rx) = create_distributed_applier(Arc::clone(&tracker)); + let limit = crate::control::distributed_applier::APPLY_WINDOW_PER_GROUP as u64; + let full: Vec = (1..=limit).map(|i| entry(i, b"w")).collect(); + + assert_eq!(applier.apply_committed(7, &full), limit); + rx.try_recv().expect("group 7 fills its window"); + assert_eq!( + applier.apply_committed(7, &[entry(limit + 1, b"w")]), + 0, + "a full window must not advance raft's applied index" + ); + assert_eq!(applier.apply_committed(8, &[entry(1, b"w")]), 1); + assert_eq!(rx.try_recv().expect("group 8 hands off").group_id, 8); + + tracker.window().release(7, 1); + assert_eq!( + applier.apply_committed(7, &[entry(limit + 1, b"w")]), + limit + 1 ); let batch = rx .try_recv() - .expect("the rejected entry must be re-accepted"); - let indexes: Vec = batch.entries.iter().map(|e| e.index).collect(); - assert_eq!(indexes, vec![3]); + .expect("group 7 hands off once a slot settles"); + assert_eq!(batch.entries[0].index, limit + 1); } /// Two concurrent deliveries of one group must not both claim the same diff --git a/nodedb/src/control/distributed_applier/apply_loop.rs b/nodedb/src/control/distributed_applier/apply_loop.rs deleted file mode 100644 index 1ec317e77..000000000 --- a/nodedb/src/control/distributed_applier/apply_loop.rs +++ /dev/null @@ -1,530 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Background apply loop — reads committed Raft entries from the mpsc channel, -//! submits them through the shared Control-Plane write funnel (which appends -//! each entry's redo record on THIS replica before the enqueue), and resolves -//! propose waiters with the result. -//! -//! Each batch advances the group's durable applied floor (see -//! [`super::applied_index`]) to its highest contiguous successfully-applied -//! entry, so the next boot replays only above it and no entry is applied by -//! both WAL replay and Raft log replay. - -use std::sync::Arc; - -use tokio::sync::mpsc; -use tracing::debug; - -use crate::bridge::envelope::{PhysicalPlan, Status}; -use crate::control::array_sync::raft_apply::{ - AppliedPosition, ArrayCellTarget, apply_array_cell_write, apply_array_op, apply_array_schema, -}; -use crate::control::cluster::calvin::ReadResultEvent; -use crate::control::server::dispatch_utils::{ - ChangeFeedOwner, SubmitWrite, WalDurability, WriteOrdering, submit_write, -}; -use crate::control::state::SharedState; -use crate::control::wal_replication::{ReplicatedEntry, ReplicatedWrite, from_replicated_entry}; -use crate::types::{DatabaseId, TenantId, TraceId}; -use nodedb_physical::physical_plan::ArrayOp; - -use super::applied_index::{AppliedPrefix, save_applied_index}; -use super::applier::ApplyBatch; -use super::propose_tracker::{AppliedWrite, ProposeTracker}; - -fn committed_response_result( - response: &crate::bridge::envelope::Response, -) -> crate::Result { - if response.status == Status::Ok { - return Ok(AppliedWrite::from_response(response)); - } - // The typed code carries the client's classification (constraint, authz, - // conflict); stringifying it here would leave the caller only XX000. - match response.error_code.as_deref() { - Some(code) => { - tracing::warn!(reason = ?code, "applying committed write failed"); - Err(crate::Error::DataPlane(code.clone())) - } - None => { - tracing::warn!( - reason = "execution error", - "applying committed write failed" - ); - Err(crate::Error::Internal { - detail: "execution error".to_owned(), - }) - } - } -} - -fn deterministic_crdt_fence_noop(result: &crate::Result) -> bool { - matches!( - result, - Err(crate::Error::DataPlane( - crate::bridge::envelope::ErrorCode::CrdtFrontierMismatch { .. } - )) - ) -} - -/// Run the background loop that applies committed Raft entries to the local Data Plane. -/// -/// This task reads from the apply channel, deserializes each entry, dispatches -/// the write to the Data Plane via SPSC, and notifies proposers. -pub async fn run_apply_loop( - mut apply_rx: mpsc::Receiver, - state: Arc, - tracker: Arc, - calvin_read_result_senders: Arc< - std::sync::Mutex>>, - >, -) { - while let Some(batch) = apply_rx.recv().await { - // The floor is saved ONCE per batch, after the loop — never per entry. - // `save_applied_index` lands a redb transaction, and redb commits at - // `Durability::Immediate`, so a per-entry save puts one synchronous - // fsync per applied entry directly on the raft apply path. That stalls - // the raft loop hard enough to delay heartbeats and keep elections from - // stabilizing under a multi-node write load. One fsync per batch - // amortizes the cost across every entry in it and keeps the critical - // path free. - // - // `AppliedPrefix` computes WHICH index is safe to save: the highest - // contiguous successfully-applied entry, stopping at the first failure - // and never advancing past it. Every branch below must therefore report - // its outcome — `record` for the ones whose success means a durable - // redo record, `skip` for the ones that apply no durable state at all. - let mut prefix = AppliedPrefix::new(); - for entry in &batch.entries { - // Decode once; reused for both the idempotency key and the - // Array/Calvin fast-path match below. Returns 0 for - // unparseable / pre-key entries; the tracker treats 0 as - // "no key" (no mismatch detection). - let replicated_opt = ReplicatedEntry::from_bytes(&entry.data); - let applied_key = replicated_opt - .as_ref() - .map(|e| e.idempotency_key) - .unwrap_or(0); - - // Database scope for the entry, read from the wire. `0` decodes to - // `DatabaseId::DEFAULT` (the pre-`database_id` legacy shape). The - // generic decode path (`from_replicated_entry`) returns only - // `(tenant, vshard, plan, resolved_now_ms)`, so the scope is taken - // from the entry itself — a WAL redo appended under the wrong - // database scope replays into the wrong catalog namespace. - let database_id = replicated_opt - .as_ref() - .map(|e| DatabaseId::new(e.database_id)) - .unwrap_or(DatabaseId::DEFAULT); - - // ── Array CRDT variants — handled on the Control Plane, bypass Data Plane ── - if let Some(replicated) = replicated_opt { - let target_vshard = replicated.vshard_id; - match replicated.write { - ReplicatedWrite::ArrayOp { - ref array, - ref op_bytes, - ref provenance, - .. - } => { - let applied_ok = apply_array_op( - &state, - &tracker, - AppliedPosition { - group_id: batch.group_id, - log_index: entry.index, - applied_key, - }, - crate::control::array_sync::ArrayOpTarget { - tenant_id: TenantId::new(replicated.tenant_id), - database_id: DatabaseId::new(replicated.database_id), - array, - }, - op_bytes, - provenance.as_deref(), - ) - .await; - // Advance the durable prefix only when the op durably - // applied — same safe-watermark rule as the Data Plane - // write path below, and the same funnel: the op path - // submits through `submit_write`, so its redo is fsynced - // before it reports success. A failure breaks the - // prefix: the entry must stay replayable. - prefix.record(entry.index, applied_ok); - continue; - } - ReplicatedWrite::ArraySchema { - ref array, - ref snapshot_payload, - schema_hlc_bytes, - } => { - let applied_ok = apply_array_schema( - &state, - &tracker, - AppliedPosition { - group_id: batch.group_id, - log_index: entry.index, - applied_key, - }, - crate::control::array_sync::raft_apply::ArraySchemaPayload { - tenant_id: TenantId::new(replicated.tenant_id), - database_id: DatabaseId::new(replicated.database_id), - array, - snapshot_payload, - schema_hlc_bytes, - }, - ); - // Advance the durable prefix only when the schema - // snapshot durably imported. - // - // This is the one applied branch that mints no WAL redo - // record, and it needs none: its entire effect is two - // fsync-committed redb transactions — the schema - // registry's snapshot row and the array catalog's entry - // — both written before it reports success. The floor's - // invariant ("this entry's state survives a restart, so - // Raft need not redeliver it") is therefore already met - // by the registries themselves. The cell paths have no - // such durable store behind them: their state lives in - // Data-Plane memtables and exists on disk only as the - // redo record the funnel appends, which is why they must - // route through `submit_write`. - prefix.record(entry.index, applied_ok); - continue; - } - ReplicatedWrite::CalvinReadResult { - epoch, - position, - passive_vshard, - tenant_id, - ref values, - } => { - let decoded_values: Vec<( - nodedb_physical::physical_plan::meta::PassiveReadKeyId, - nodedb_types::Value, - )> = match zerompk::from_msgpack(values) { - Ok(decoded) => decoded, - Err(e) => { - tracing::warn!( - group_id = batch.group_id, - index = entry.index, - error = %e, - "failed to decode CalvinReadResult payload" - ); - tracker.complete( - batch.group_id, - entry.index, - applied_key, - Err(crate::Error::Internal { - detail: format!("decode CalvinReadResult payload: {e}"), - }), - ); - // Prefix-neutral, like the forward below: a read - // result mints no durable state either way, so - // there is nothing a re-delivery could restore - // and nothing later entries must wait behind. - prefix.skip(); - continue; - } - }; - - let event = ReadResultEvent { - epoch, - position, - passive_vshard, - tenant_id: TenantId::new(tenant_id), - values: decoded_values, - }; - - let send_result = calvin_read_result_senders - .lock() - .unwrap_or_else(|p| p.into_inner()) - .get(&target_vshard) - .cloned() - .map(|sender| sender.try_send(event)); - - if let Some(Err(e)) = send_result { - tracing::warn!( - group_id = batch.group_id, - index = entry.index, - error = %e, - "failed to forward CalvinReadResult to scheduler" - ); - } - tracker.complete( - batch.group_id, - entry.index, - applied_key, - Ok(AppliedWrite::unversioned(Vec::new())), - ); - // A read result is forwarded to an in-memory Calvin - // scheduler and writes nothing durable, so it neither - // advances the prefix nor breaks it. Advancing on it - // would assert a redo record that does not exist; - // breaking on it would stall the floor behind an entry - // that a re-delivery could not usefully replay anyway — - // the epoch it belongs to does not survive a restart — - // and force every later write in the batch to be applied - // twice on the next boot. - prefix.skip(); - continue; - } - _ => {} - } - } - - let decoded = - from_replicated_entry(&entry.data, Some(state.surrogate_assigner.as_ref())); - let (tenant_id, vshard_id, plan, resolved_now_ms) = match decoded { - Ok(Some(t)) => t, - Ok(None) => { - // Couldn't deserialize — might be a different format or corrupted. - debug!( - group_id = batch.group_id, - index = entry.index, - "skipping non-ReplicatedEntry commit" - ); - tracker.complete( - batch.group_id, - entry.index, - applied_key, - Ok(AppliedWrite::unversioned(Vec::new())), - ); - // Prefix-neutral. This is a pure shape check over - // `entry.data`, so a re-delivery on the next boot decodes to - // `None` again and skips again — stalling the floor behind - // it buys nothing and costs a double-apply of every later - // write in the batch. It applied no state, so it must not - // advance the floor either. - prefix.skip(); - continue; - } - Err(e) => { - tracing::warn!( - group_id = batch.group_id, - index = entry.index, - error = %e, - "failed to decode replicated entry (surrogate bind error)" - ); - tracker.complete( - batch.group_id, - entry.index, - applied_key, - Err(crate::Error::Internal { - detail: format!("decode replicated entry: {e}"), - }), - ); - // Breaks the prefix, unlike the `Ok(None)` skip above: this - // IS a write, and it failed against live surrogate-assigner - // state rather than on its own bytes, so a re-delivery can - // legitimately succeed. Holding the floor below it is what - // keeps it replayable. - prefix.record(entry.index, false); - continue; - } - }; - - // Raft-native array cell writes (`ArrayCellPut` / `ArrayCellDelete`) - // decode to `PhysicalPlan::Array(Put | Delete)`. A follower must - // OPEN the array on the Data Plane before applying, so these route - // through the array-open bootstrap first — and then through the same - // write funnel as the generic branch below, which is what gives them - // a redo record and the fsync the applied floor asserts. No other - // `ReplicatedWrite` variant decodes to a `PhysicalPlan::Array`, so - // this match is exact. - if matches!( - plan, - PhysicalPlan::Array(ArrayOp::Put { .. } | ArrayOp::Delete { .. }) - ) { - let applied_ok = apply_array_cell_write( - &state, - &tracker, - AppliedPosition { - group_id: batch.group_id, - log_index: entry.index, - applied_key, - }, - ArrayCellTarget { - tenant_id, - database_id, - vshard: vshard_id, - resolved_now_ms, - }, - plan, - ) - .await; - prefix.record(entry.index, applied_ok); - continue; - } - - let submitted = submit_write( - &state, - SubmitWrite { - tenant_id, - database_id, - vshard_id, - plan, - trace_id: TraceId::generate(), - // Cluster mode has exactly ONE write-apply path — this loop; - // the proposing node does not execute locally before commit - // either. Tagging these `RaftFollower` would mean AFTER - // triggers, DML audit, and CRDT packaging never fire anywhere - // in cluster mode, so the committed write keeps the `User` - // source its proposer had. - event_source: crate::event::EventSource::User, - txn_id: None, - // Auth ran on the proposing node before the entry was - // proposed; the committed entry carries no session user. - user_id: None, - // The redo record is appended HERE, on this replica, from the - // committed plan — the leader's WAL LSN is deliberately not - // carried on the wire, and the memory-only engines have no - // other durability path. `now_override` pins a TTL-bearing KV - // write's `expire_at_ms` to the instant the proposing node - // resolved, so this replica's redo record and its live apply - // install the byte-identical value every other replica does. - durability: WalDurability::AppendHere { - now_override: resolved_now_ms, - }, - // Raft committed this entry at a fixed log index; every - // replica applies it in that order. Re-entering the - // write-admission gate would re-decide an ordering that is - // already final. - ordering: WriteOrdering::AlreadyOrdered, - // This loop runs on EVERY replica, so it must not publish: - // the node that proposed this entry already published the - // write's change event once, after commit + apply. Emitting - // here would give each subscriber one copy per replica plus - // a NOTIFY fan-out from each. See [`ChangeFeedOwner`]. - change_feed: ChangeFeedOwner::Unowned, - }, - ) - .await - .map(|outcome| outcome.response); - - // The funnel returns an error-status response as `Ok`; a committed - // entry that failed to apply must surface to the propose waiter as a - // failure, not as an empty success. - let result = match submitted { - // The response carries this replica's post-write - // `coll_write_lsn` for the written collection, which the - // proposer needs as its read-your-writes floor: the version is - // minted here (the funnel's WAL append) and never travels on the - // wire, so the propose waiter is the only place it can be - // handed back. - Ok(resp) if resp.status == Status::Ok => Ok(AppliedWrite::from_response(&resp)), - Ok(resp) => committed_response_result(&resp), - Err(e) => { - tracing::warn!( - group_id = batch.group_id, - index = entry.index, - error = %e, - "applying committed write failed" - ); - // Passed through: the typed error already carries the - // caller's classification. - Err(e) - } - }; - - let applied_ok = result.is_ok() || deterministic_crdt_fence_noop(&result); - tracker.complete(batch.group_id, entry.index, applied_key, result); - - // Extend the batch's durable prefix. On success `submit_write`'s - // durable-at-ack barrier has already fsynced this entry's redo, - // which is exactly the fact the floor asserts — `entry.index` is - // the data-plane applied watermark here, NOT raft's commit index. - // On failure the engines did not persist this index, so it is - // neither a safe compaction boundary nor a safe restart floor; - // breaking the prefix is what keeps a genuinely failed apply - // replayable rather than silently skipped. - prefix.record(entry.index, applied_ok); - } - - // One save + one compaction check per batch, against the contiguous - // prefix. Compaction is deliberately driven by the same index the floor - // was just saved at — never the batch's last delivered index — so it can - // never discard an entry the next boot still has to replay. Compacting - // on the raft commit index while the SPSC apply lags would likewise let - // the `SnapshotBuilder` serialize incomplete engine state and corrupt a - // lagging follower's snapshot. - if let Some(applied_index) = prefix.floor() { - record_durable_apply(&state, batch.group_id, applied_index); - } - } -} - -/// Record entry `applied_index` of `group_id` as durably applied: persist the -/// group's durable applied floor, then fire the compaction trigger against it. -/// -/// `applied_index` MUST be the highest CONTIGUOUS successfully-applied entry — -/// [`AppliedPrefix::floor`] — not merely some entry that happened to succeed. -/// Everything at and below it must have applied with its redo record already -/// WAL-fsync-durable, because that is the fact the floor asserts and the next -/// boot resumes Raft delivery above it on the strength of it. -/// -/// Called once per apply batch: each call is a redb commit and therefore an -/// fsync, and one per entry is slow enough to stall the raft loop it runs on. -/// -/// Order is load-bearing: the floor lands first because compaction is itself -/// gated on the floor (it may only discard entries the engines can no longer -/// need the log for). Compacting first would either be refused or, if the gate -/// used the delivery watermark, discard entries whose redo is not yet fsynced. -fn record_durable_apply(state: &Arc, group_id: u64, applied_index: u64) { - save_applied_index(state, group_id, applied_index); - maybe_compact_log(state, group_id, applied_index); -} - -/// Fire the Raft log-compaction trigger for `group_id` up to the -/// data-plane applied index `applied_index`, if a compactor is wired. -/// -/// Gated by the caller on data-plane apply completion. A no-op when no -/// compactor is installed (single-node mode) or when the group's -/// `log_compaction_threshold` is `None`. -fn maybe_compact_log(state: &Arc, group_id: u64, applied_index: u64) { - let Some(compactor) = state.raft_compactor.get() else { - return; - }; - match compactor(group_id, applied_index) { - Ok(true) => { - debug!( - group_id, - applied_index, "raft log compacted past data-plane applied watermark" - ); - } - Ok(false) => {} - Err(e) => { - tracing::warn!( - group_id, - applied_index, - error = %e, - "raft log compaction failed" - ); - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn fenced_frontier_mismatch_completes_retry_and_advances_durable_prefix() { - let result: crate::Result = Err(crate::Error::DataPlane( - crate::bridge::envelope::ErrorCode::CrdtFrontierMismatch { - expected: [1; 32], - actual: [2; 32], - }, - )); - assert!(deterministic_crdt_fence_noop(&result)); - assert!(matches!( - result, - Err(crate::Error::DataPlane( - crate::bridge::envelope::ErrorCode::CrdtFrontierMismatch { .. } - )) - )); - - let mut prefix = AppliedPrefix::new(); - prefix.record(17, deterministic_crdt_fence_noop(&result)); - assert_eq!(prefix.floor(), Some(17)); - } -} diff --git a/nodedb/src/control/distributed_applier/apply_loop/bookkeeping.rs b/nodedb/src/control/distributed_applier/apply_loop/bookkeeping.rs new file mode 100644 index 000000000..1de50ce95 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/bookkeeping.rs @@ -0,0 +1,62 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Applied-prefix bookkeeping: persist the group's durable applied floor, +//! then fire the Raft log-compaction trigger against it. + +use std::sync::Arc; + +use tracing::debug; + +use crate::control::distributed_applier::applied_index::save_applied_index; +use crate::control::state::SharedState; + +/// Record entry `applied_index` of `group_id` as durably applied: persist the +/// group's durable applied floor, then fire the compaction trigger against it. +/// +/// `applied_index` MUST be the highest CONTIGUOUS successfully-applied entry — +/// [`crate::control::distributed_applier::applied_index::AppliedPrefix::floor`] +/// — not merely some entry that happened to succeed. Everything at and below +/// it must have applied with its redo record already WAL-fsync-durable, +/// because that is the fact the floor asserts and the next boot resumes Raft +/// delivery above it on the strength of it. +/// +/// Called once per apply batch: each call is a redb commit and therefore an +/// fsync, and one per entry is slow enough to stall the raft loop it runs on. +/// +/// Order is load-bearing: the floor lands first because compaction is itself +/// gated on the floor (it may only discard entries the engines can no longer +/// need the log for). Compacting first would either be refused or, if the gate +/// used the delivery watermark, discard entries whose redo is not yet fsynced. +pub(super) fn record_durable_apply(state: &Arc, group_id: u64, applied_index: u64) { + save_applied_index(state, group_id, applied_index); + maybe_compact_log(state, group_id, applied_index); +} + +/// Fire the Raft log-compaction trigger for `group_id` up to the +/// data-plane applied index `applied_index`, if a compactor is wired. +/// +/// Gated by the caller on data-plane apply completion. A no-op when no +/// compactor is installed (single-node mode) or when the group's +/// `log_compaction_threshold` is `None`. +fn maybe_compact_log(state: &Arc, group_id: u64, applied_index: u64) { + let Some(compactor) = state.raft_compactor.get() else { + return; + }; + match compactor(group_id, applied_index) { + Ok(true) => { + debug!( + group_id, + applied_index, "raft log compacted past data-plane applied watermark" + ); + } + Ok(false) => {} + Err(e) => { + tracing::warn!( + group_id, + applied_index, + error = %e, + "raft log compaction failed" + ); + } + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/calvin_read_result.rs b/nodedb/src/control/distributed_applier/apply_loop/calvin_read_result.rs new file mode 100644 index 000000000..998e8aa74 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/calvin_read_result.rs @@ -0,0 +1,106 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Forward a committed `ReplicatedWrite::CalvinReadResult` entry to the local +//! Calvin scheduler's read-result channel for its target vShard. +//! +//! A read result is forwarded to an in-memory Calvin scheduler and writes +//! nothing durable, so the caller neither advances the applied prefix nor +//! breaks it for this entry. Advancing on it would assert a redo record that +//! does not exist; breaking on it would stall the floor behind an entry that a +//! re-delivery could not usefully replay anyway — the epoch it belongs to does +//! not survive a restart — and force every later write in the batch to be +//! applied twice on the next boot. + +use std::collections::BTreeMap; +use std::sync::{Arc, Mutex}; + +use tokio::sync::mpsc; + +use crate::control::array_sync::raft_apply::AppliedPosition; +use crate::control::cluster::calvin::ReadResultEvent; +use crate::control::distributed_applier::propose_tracker::{AppliedWrite, ProposeTracker}; +use crate::types::TenantId; + +/// Fields extracted from a `ReplicatedWrite::CalvinReadResult` entry. +pub(super) struct CalvinReadResultFields<'a> { + pub target_vshard: u32, + pub epoch: u64, + pub position: u32, + pub passive_vshard: u32, + pub tenant_id: u64, + pub values: &'a [u8], +} + +/// Decode `fields.values`, forward the resulting [`ReadResultEvent`] to the +/// scheduler registered for `fields.target_vshard`, and complete the propose +/// waiter. +pub(super) fn forward_calvin_read_result( + tracker: &Arc, + calvin_read_result_senders: &Arc>>>, + pos: AppliedPosition, + fields: CalvinReadResultFields<'_>, +) { + let AppliedPosition { + group_id, + log_index, + applied_key, + .. + } = pos; + + let decoded_values: Vec<( + nodedb_physical::physical_plan::meta::PassiveReadKeyId, + nodedb_types::Value, + )> = match zerompk::from_msgpack(fields.values) { + Ok(decoded) => decoded, + Err(e) => { + tracing::warn!( + group_id, + index = log_index, + error = %e, + "failed to decode CalvinReadResult payload" + ); + tracker.complete( + group_id, + log_index, + applied_key, + Err(crate::Error::Internal { + detail: format!("decode CalvinReadResult payload: {e}"), + }), + ); + // Prefix-neutral, like the forward below: a read result mints no + // durable state either way, so there is nothing a re-delivery + // could restore and nothing later entries must wait behind. + return; + } + }; + + let event = ReadResultEvent { + epoch: fields.epoch, + position: fields.position, + passive_vshard: fields.passive_vshard, + tenant_id: TenantId::new(fields.tenant_id), + values: decoded_values, + }; + + let send_result = calvin_read_result_senders + .lock() + .unwrap_or_else(|p| p.into_inner()) + .get(&fields.target_vshard) + .cloned() + .map(|sender| sender.try_send(event)); + + if let Some(Err(e)) = send_result { + tracing::warn!( + group_id, + index = log_index, + error = %e, + "failed to forward CalvinReadResult to scheduler" + ); + } + tracker.complete( + group_id, + log_index, + applied_key, + Ok(AppliedWrite::unversioned(Vec::new())), + ); +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/context.rs b/nodedb/src/control/distributed_applier/apply_loop/context.rs new file mode 100644 index 000000000..21b27a800 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/context.rs @@ -0,0 +1,83 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! What every entry's apply borrows from the loop, and the futures the loop +//! collects while it prepares later entries. + +use std::collections::BTreeMap; +use std::future::Future; +use std::pin::Pin; +use std::sync::{Arc, Mutex}; + +use tokio::sync::mpsc; + +use crate::control::cluster::calvin::ReadResultEvent; +use crate::control::distributed_applier::propose_tracker::ProposeTracker; +use crate::control::state::SharedState; + +use super::proposal_gate::EntryOutcome; + +/// Senders to each local Calvin scheduler's read-result channel, by vShard. +pub(super) type CalvinReadResultSenders = Arc>>>; + +/// The loop-owned handles an entry's apply borrows. +#[derive(Clone, Copy)] +pub(super) struct ApplyContext<'a> { + pub state: &'a Arc, + pub tracker: &'a Arc, + pub calvin_read_result_senders: &'a CalvinReadResultSenders, +} + +/// An entry whose apply finished: its waiter is resolved, and its outcome +/// waits for every earlier entry of its group before it settles. +pub(super) struct FinishedApply { + pub group_id: u64, + pub log_index: u64, + pub outcome: EntryOutcome, +} + +/// An apply the loop collects while it starts later entries. +pub(super) type ApplyFuture<'a> = Pin + Send + 'a>>; + +/// How a write continues once its enqueue returned. +pub(super) enum Started<'a> { + /// The write is on its core. The apply collects its outcome. + Running(ApplyFuture<'a>), + /// The write concluded without reaching a core. + Concluded(EntryOutcome), +} + +/// A write past its enqueue, the collection its plan named, and whether its +/// plan writes user data: only such a write raises its tenant's write mark, +/// as the write funnel decides for every write it records. +pub(super) struct StartedEntry<'a> { + pub started: Started<'a>, + pub collection: Option, + pub user_write: bool, +} + +impl StartedEntry<'_> { + pub fn concluded(outcome: EntryOutcome) -> Self { + Self { + started: Started::Concluded(outcome), + collection: None, + user_write: false, + } + } +} + +/// A write's enqueue. The next entry of its group starts once it returns. +pub(super) type EnqueueFuture<'a> = Pin> + Send + 'a>>; + +/// What the loop collects: an enqueue that returned, or an apply that +/// finished. +pub(super) enum LoopEvent<'a> { + Enqueued { + group_id: u64, + log_index: u64, + entry: StartedEntry<'a>, + }, + Finished(FinishedApply), +} + +/// A future the loop collects. +pub(super) type LoopFuture<'a> = Pin> + Send + 'a>>; diff --git a/nodedb/src/control/distributed_applier/apply_loop/driver.rs b/nodedb/src/control/distributed_applier/apply_loop/driver.rs new file mode 100644 index 000000000..1d515a0a6 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/driver.rs @@ -0,0 +1,89 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Loop driver: takes batches off the apply channel into the pipeline, and +//! collects the enqueues and applies that finish, until the channel closes +//! and every started entry concluded. + +use std::sync::Arc; + +use tokio::sync::mpsc; + +use crate::control::cluster::calvin::ReadResultEvent; +use crate::control::distributed_applier::applier::ApplyBatch; +use crate::control::distributed_applier::proposal_ledger::{ + PROPOSAL_LEDGER_CAPACITY, ProposalLedger, +}; +use crate::control::distributed_applier::propose_tracker::ProposeTracker; +use crate::control::state::SharedState; + +use super::context::ApplyContext; +use super::pipeline::Pipeline; + +/// Run the background loop that applies committed Raft entries to the local Data Plane. +/// +/// This task reads from the apply channel, deserializes each entry, dispatches +/// the write to the Data Plane via SPSC, and notifies proposers. +pub async fn run_apply_loop( + mut apply_rx: mpsc::Receiver, + state: Arc, + tracker: Arc, + calvin_read_result_senders: Arc< + std::sync::Mutex>>, + >, +) { + // Proposals this node already applied, recovered from its WAL before any + // entry is delivered: every record an entry's apply appended carries the + // entry's idempotency key in its header. + let records = match state.wal.replay() { + Ok(records) => records, + Err(error) => { + // Without the keys an entry re-delivered above the durable floor, + // or a second committed copy of a proposal, would apply a second + // time. Refuse to apply anything rather than risk it: the loop + // stops, and every propose waiter surfaces the stall. + tracing::error!( + %error, + "data-group apply loop cannot read its WAL to recover applied proposals; \ + refusing to apply committed entries" + ); + return; + } + }; + let ledger = ProposalLedger::from_records(&records, PROPOSAL_LEDGER_CAPACITY); + drop(records); + + let ctx = ApplyContext { + state: &state, + tracker: &tracker, + calvin_read_result_senders: &calvin_read_result_senders, + }; + let mut pipeline = Pipeline::new(ctx, ledger); + let mut accepting = true; + loop { + if !accepting && !pipeline.has_running() { + // The channel closed and every started entry concluded. The + // pump started every entry that can start, so none is queued. + return; + } + let running = pipeline.has_running(); + tokio::select! { + biased; + Some(event) = pipeline.next_event(), if running => { + pipeline.handle(event); + } + batch = apply_rx.recv(), if accepting => match batch { + Some(batch) => { + pipeline.accept(batch); + // Take every batch already queued, so one pass starts + // them all and one floor save covers them. + while let Ok(batch) = apply_rx.try_recv() { + pipeline.accept(batch); + } + } + None => accepting = false, + }, + } + pipeline.pump(); + pipeline.settle(); + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/group_watch.rs b/nodedb/src/control/distributed_applier/apply_loop/group_watch.rs new file mode 100644 index 000000000..4bb3367d9 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/group_watch.rs @@ -0,0 +1,82 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-group state the apply loop keeps across batches: the highest index it +//! applied, to report a second apply of a committed entry, and the backup cut +//! floor, which raises the commit HLC of every entry after a cut barrier. + +use std::collections::HashMap; + +/// Per-group apply state. +#[derive(Debug, Default)] +pub(super) struct GroupWatch { + highest_applied: HashMap, + /// Lowest commit HLC an entry after the group's latest cut barrier + /// records: one above that barrier's watermark. + cut_floor: HashMap, +} + +impl GroupWatch { + /// Note that the apply loop applies `(group_id, log_index)`. Reports an + /// index at or below one it already applied as a second apply. + pub(super) fn note_apply(&mut self, group_id: u64, log_index: u64) { + let highest = self.highest_applied.entry(group_id).or_insert(0); + if log_index <= *highest { + crate::diag::raft_entry_reapplied(group_id, log_index, *highest); + return; + } + *highest = log_index; + } + + /// Raise `group_id`'s cut floor above the watermark `cut_hlc` of a + /// backup's cut barrier. + pub(super) fn raise_cut(&mut self, group_id: u64, cut_hlc: u64) { + let floor = self.cut_floor.entry(group_id).or_insert(0); + *floor = (*floor).max(cut_hlc.saturating_add(1)); + } + + /// The commit HLC an entry of `group_id` stamped `write_hlc` records. + /// + /// An entry the log places after a cut barrier records at least the cut + /// floor, however early its proposer stamped it: the backup that placed + /// the barrier did not contain it, so a restore of that backup refuses + /// it. `0` means the entry carries no stamp; its apply stamps its own + /// append, which already follows every barrier before it. + pub(super) fn commit_hlc(&self, group_id: u64, write_hlc: u64) -> u64 { + if write_hlc == 0 { + return 0; + } + self.cut_floor + .get(&group_id) + .map_or(write_hlc, |floor| write_hlc.max(*floor)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn an_entry_after_a_cut_records_above_the_cut() { + let mut watch = GroupWatch::default(); + assert_eq!(watch.commit_hlc(1, 50), 50); + watch.raise_cut(1, 100); + assert_eq!(watch.commit_hlc(1, 50), 101); + assert_eq!(watch.commit_hlc(1, 200), 200); + assert_eq!(watch.commit_hlc(2, 50), 50, "a cut binds only its group"); + assert_eq!( + watch.commit_hlc(1, 0), + 0, + "an unstamped entry stamps its own append" + ); + } + + #[test] + fn a_second_apply_of_an_index_is_counted() { + let before = crate::diag::raft_entries_reapplied(); + let mut watch = GroupWatch::default(); + watch.note_apply(3, 5); + watch.note_apply(3, 6); + watch.note_apply(3, 6); + assert!(crate::diag::raft_entries_reapplied() > before); + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/helpers.rs b/nodedb/src/control/distributed_applier/apply_loop/helpers.rs new file mode 100644 index 000000000..0918a69b6 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/helpers.rs @@ -0,0 +1,60 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Shared classification helpers for committed-write apply results. + +use crate::control::distributed_applier::propose_tracker::AppliedWrite; + +pub(super) fn committed_response_result( + response: &crate::bridge::envelope::Response, +) -> crate::Result { + if response.status == crate::bridge::envelope::Status::Ok { + return Ok(AppliedWrite::from_response(response)); + } + // The typed code carries the client's classification (constraint, authz, + // conflict); stringifying it here would leave the caller only XX000. + match response.error_code.as_deref() { + Some(code) => { + tracing::warn!(reason = ?code, "applying committed write failed"); + Err(crate::Error::DataPlane(code.clone())) + } + None => { + tracing::warn!( + reason = "execution error", + "applying committed write failed" + ); + Err(crate::Error::Internal { + detail: "execution error".to_owned(), + }) + } + } +} + +pub(super) fn deterministic_crdt_fence_noop(result: &crate::Result) -> bool { + matches!( + result, + Err(crate::Error::DataPlane( + crate::bridge::envelope::ErrorCode::CrdtFrontierMismatch { .. } + )) + ) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::distributed_applier::applied_index::AppliedPrefix; + + #[test] + fn fenced_frontier_mismatch_completes_retry_and_advances_durable_prefix() { + let result: crate::Result = Err(crate::Error::DataPlane( + crate::bridge::envelope::ErrorCode::CrdtFrontierMismatch { + expected: [1; 32], + actual: [2; 32], + }, + )); + assert!(deterministic_crdt_fence_noop(&result)); + + let mut prefix = AppliedPrefix::new(); + prefix.record(17, deterministic_crdt_fence_noop(&result)); + assert_eq!(prefix.floor(), Some(17)); + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/lane.rs b/nodedb/src/control/distributed_applier/apply_loop/lane.rs new file mode 100644 index 000000000..be896f36f --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/lane.rs @@ -0,0 +1,458 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! One group's place in the apply pipeline: the entries not yet started, in +//! log order, and the started entries not yet settled, in log order. +//! +//! Entries start in log order, and a write's enqueue returns before the next +//! entry of its group starts, so every core receives a group's writes in the +//! order the log fixed. They finish in any order. They settle in log +//! order: the group's applied index is the highest index with every earlier +//! entry finished, and its durable floor is the highest index with every +//! earlier entry durable. + +use std::collections::VecDeque; + +use nodedb_raft::message::LogEntry; + +use crate::control::distributed_applier::applied_index::AppliedPrefix; +use crate::control::distributed_applier::propose_tracker::{ApplyingEntry, ProposeTracker}; +use crate::control::server::shared::write_admission::plan_writes_user_data; +use crate::control::state::tenant_marks::{MarkSite, TenantMarks}; +use crate::control::wal_replication::{ReplicatedEntry, ReplicatedWrite, from_replicated_entry}; + +use super::proposal_gate::PrefixStep; + +/// A committed entry handed to the loop and not yet started. +pub(super) struct QueuedEntry { + pub entry: LogEntry, + /// The entry decoded once on arrival. `None` for a leader-change no-op or + /// bytes that do not decode as a replicated entry. + pub decoded: Option, +} + +impl QueuedEntry { + pub fn new(entry: LogEntry) -> Self { + let decoded = if entry.data.is_empty() { + None + } else { + ReplicatedEntry::from_bytes(&entry.data) + }; + Self { entry, decoded } + } + + /// The proposal's idempotency key, `0` when the entry carries none. + pub fn proposal_key(&self) -> u64 { + self.decoded.as_ref().map_or(0, |e| e.idempotency_key) + } + + /// The metadata index the entry's proposer had applied, `0` when the + /// entry carries none. + pub fn metadata_floor(&self) -> u64 { + self.decoded.as_ref().map_or(0, |e| e.metadata_floor) + } + + /// `(tenant_id, write_hlc)` of an entry that writes a tenant's data. A + /// cut barrier and a Calvin read result write nothing, and an entry with + /// no proposer stamp has no commit HLC to record. + pub fn write_stamp(&self) -> Option<(u64, u64)> { + let decoded = self.decoded.as_ref()?; + if decoded.write_hlc == 0 + || matches!( + decoded.write, + ReplicatedWrite::CutBarrier { .. } | ReplicatedWrite::CalvinReadResult { .. } + ) + { + return None; + } + Some((decoded.tenant_id, decoded.write_hlc)) + } + + /// Whether the entry's plan writes user data, as the write funnel decides + /// for the writes it records. Decoded here only for an entry that never + /// reaches the funnel: a second copy of an applied proposal. + pub fn plan_writes_user_data(&self) -> bool { + let Some(decoded) = self.decoded.as_ref() else { + return false; + }; + match decoded.write { + ReplicatedWrite::ArrayOp { .. } + | ReplicatedWrite::ArrayCellPut { .. } + | ReplicatedWrite::ArrayCellDelete { .. } + | ReplicatedWrite::TransactionRedo { .. } => true, + ReplicatedWrite::ArraySchema { .. } + | ReplicatedWrite::CutBarrier { .. } + | ReplicatedWrite::CalvinReadResult { .. } => false, + _ => matches!( + from_replicated_entry(&self.entry.data, None), + Ok(Some((_, _, plan, _))) if plan_writes_user_data(&plan) + ), + } + } + + /// Whether the entry must apply with nothing else of its group in + /// flight. The array paths await their own write inside the apply, so + /// the loop cannot fix their arrival order at the core any other way, + /// and a schema import must follow every earlier entry's apply. + pub fn is_exclusive(&self) -> bool { + self.decoded.as_ref().is_some_and(|e| { + matches!( + e.write, + ReplicatedWrite::ArrayOp { .. } + | ReplicatedWrite::ArraySchema { .. } + | ReplicatedWrite::ArrayCellPut { .. } + | ReplicatedWrite::ArrayCellDelete { .. } + ) + }) + } +} + +/// Where a started entry stands. +pub(super) enum SlotState { + /// The write's enqueue runs. + Starting, + /// The apply runs; its outcome arrives as a finished apply. + Running, + /// A backup's cut barrier. It completes once every earlier entry of its + /// group settled. + Barrier, + /// The entry concluded. + Concluded(PrefixStep), +} + +/// A started entry not yet settled. +pub(super) struct Slot { + pub log_index: u64, + pub proposal_key: u64, + /// The collection the entry writes, when its apply named one. + pub collection: Option, + /// `(tenant_id, commit_hlc)` the entry records on its tenant's mark in + /// this group once it settles, when it carries a proposer stamp. + pub write_mark: Option<(u64, u64)>, + /// Whether the entry's plan writes user data. Only such an entry raises + /// its tenant's mark. + pub user_write: bool, + pub state: SlotState, +} + +/// One group's apply pipeline. +pub(super) struct Lane { + group_id: u64, + pub backlog: VecDeque, + slots: VecDeque, + /// The entry the group's next start waits on: a write whose enqueue + /// runs, or an exclusive entry that runs. + pub blocking: Option, + /// The durable prefix over every entry this process settled for the + /// group. A break holds for the life of the process: an index saved past + /// a non-durable entry would let the next boot skip it. + prefix: AppliedPrefix, + /// The floor last saved for the group. + saved_floor: Option, +} + +impl Lane { + pub fn new(group_id: u64) -> Self { + Self { + group_id, + backlog: VecDeque::new(), + slots: VecDeque::new(), + blocking: None, + prefix: AppliedPrefix::new(), + saved_floor: None, + } + } + + /// Whether any started entry of the group has not concluded. + pub fn has_running(&self) -> bool { + self.slots + .iter() + .any(|slot| matches!(slot.state, SlotState::Starting | SlotState::Running)) + } + + /// Record that the enqueue of the entry at `log_index` returned: its + /// state is `state` now, its apply named `collection`, and `user_write` + /// says whether its plan writes user data. Returns `false` when no entry + /// at that index is starting. + pub fn enqueued( + &mut self, + log_index: u64, + state: SlotState, + collection: Option, + user_write: bool, + ) -> bool { + let Some(slot) = self.slot_mut(log_index) else { + return false; + }; + if !matches!(slot.state, SlotState::Starting) { + return false; + } + slot.state = state; + slot.user_write = user_write; + if collection.is_some() { + slot.collection = collection; + } + if self.blocking == Some(log_index) { + self.blocking = None; + } + true + } + + fn slot_mut(&mut self, log_index: u64) -> Option<&mut Slot> { + let position = self + .slots + .binary_search_by_key(&log_index, |slot| slot.log_index) + .ok()?; + self.slots.get_mut(position) + } + + pub fn push(&mut self, slot: Slot) { + self.slots.push_back(slot); + } + + /// Conclude the running entry at `log_index`: `conclude` receives its + /// proposal key and returns how it moves the prefix. `wrote_rows` says + /// whether the apply wrote the entry's rows; an entry that wrote none + /// raises no tenant write mark. Returns `false` when no running entry + /// has that index. + pub fn conclude( + &mut self, + log_index: u64, + wrote_rows: bool, + conclude: impl FnOnce(u64) -> PrefixStep, + ) -> bool { + let Some(slot) = self.slot_mut(log_index) else { + return false; + }; + if !matches!(slot.state, SlotState::Running) { + return false; + } + slot.user_write &= wrote_rows; + slot.state = SlotState::Concluded(conclude(slot.proposal_key)); + if self.blocking == Some(log_index) { + self.blocking = None; + } + true + } + + /// Settle every concluded entry at the front, in log order: complete a + /// barrier's waiter, raise the tenant's write mark in this group, extend + /// or break the durable prefix, and advance the applied watermark. The + /// mark rises first, so a reader that waited for the applied index sees + /// it. Returns how many entries settled. + pub fn settle(&mut self, tracker: &ProposeTracker, marks: &TenantMarks) -> usize { + let mut settled = 0; + while let Some(front) = self.slots.front() { + let step = match front.state { + SlotState::Starting | SlotState::Running => break, + SlotState::Barrier => { + // Every entry before the barrier finished; a waiting + // backup may snapshot this group now. + tracker.complete( + self.group_id, + front.log_index, + front.proposal_key, + Ok( + crate::control::distributed_applier::AppliedWrite::unversioned( + Vec::new(), + ), + ), + ); + PrefixStep::Neutral + } + SlotState::Concluded(step) => step, + }; + let log_index = front.log_index; + if front.user_write + && let Some((tenant_id, commit_hlc)) = front.write_mark + { + marks.raise( + self.group_id, + tenant_id, + commit_hlc, + MarkSite::ReplicatedApply, + front.collection.as_deref(), + ); + } + self.slots.pop_front(); + match step { + PrefixStep::Neutral => self.prefix.skip(), + PrefixStep::Record(durable) => self.prefix.record(log_index, durable), + } + tracker.note_applied(self.group_id, log_index); + settled += 1; + } + tracker.note_applying( + self.group_id, + self.slots.front().map(|slot| ApplyingEntry { + group_id: self.group_id, + log_index: slot.log_index, + collection: slot.collection.clone(), + }), + ); + settled + } + + /// Whether the durable floor moved past the floor last saved. + pub fn floor_pending(&self) -> bool { + self.prefix + .floor() + .is_some_and(|floor| self.saved_floor.is_none_or(|saved| saved < floor)) + } + + /// The durable floor to save, when it moved past the floor last saved. + pub fn take_floor_to_save(&mut self) -> Option { + let floor = self.prefix.floor()?; + if self.saved_floor.is_some_and(|saved| saved >= floor) { + return None; + } + self.saved_floor = Some(floor); + Some(floor) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn slot(log_index: u64, state: SlotState) -> Slot { + Slot { + log_index, + proposal_key: 0, + collection: None, + write_mark: None, + user_write: false, + state, + } + } + + #[test] + fn entries_settle_in_log_order_whatever_order_they_finish() { + let tracker = ProposeTracker::new(); + let mut lane = Lane::new(1); + lane.push(slot(5, SlotState::Running)); + lane.push(slot(6, SlotState::Running)); + lane.push(slot(7, SlotState::Running)); + + assert!(lane.conclude(7, true, |_| PrefixStep::Record(true))); + assert!(lane.conclude(6, true, |_| PrefixStep::Record(true))); + assert!( + !lane.conclude(6, true, |_| PrefixStep::Record(true)), + "6 concluded already" + ); + assert_eq!( + lane.settle(&tracker, &TenantMarks::default()), + 0, + "entry 5 still runs" + ); + assert_eq!(lane.take_floor_to_save(), None); + assert_eq!( + tracker.applying(1).map(|entry| entry.log_index), + Some(5), + "the timeout diagnostic names the entry the group waits behind" + ); + + assert!(lane.conclude(5, true, |_| PrefixStep::Record(true))); + assert_eq!(lane.settle(&tracker, &TenantMarks::default()), 3); + assert_eq!(lane.take_floor_to_save(), Some(7)); + assert_eq!(lane.take_floor_to_save(), None, "a saved floor saves once"); + assert!(tracker.applying(1).is_none()); + } + + #[test] + fn a_non_durable_entry_holds_the_floor_across_later_settles() { + let tracker = ProposeTracker::new(); + let mut lane = Lane::new(1); + lane.push(slot(1, SlotState::Concluded(PrefixStep::Record(true)))); + lane.push(slot(2, SlotState::Concluded(PrefixStep::Record(false)))); + lane.settle(&tracker, &TenantMarks::default()); + assert_eq!(lane.take_floor_to_save(), Some(1)); + + lane.push(slot(3, SlotState::Concluded(PrefixStep::Record(true)))); + lane.settle(&tracker, &TenantMarks::default()); + assert_eq!( + lane.take_floor_to_save(), + None, + "a floor past entry 2 would let the next boot skip it" + ); + } + + #[test] + fn an_enqueue_that_returns_unblocks_the_next_start() { + let tracker = ProposeTracker::new(); + let mut lane = Lane::new(2); + lane.push(slot(4, SlotState::Starting)); + lane.blocking = Some(4); + assert!(lane.has_running()); + + assert!(lane.enqueued(4, SlotState::Running, Some("docs".to_owned()), true)); + assert_eq!(lane.blocking, None); + lane.settle(&tracker, &TenantMarks::default()); + assert_eq!( + tracker.applying(2).and_then(|entry| entry.collection), + Some("docs".to_owned()) + ); + assert!( + !lane.enqueued(4, SlotState::Running, None, false), + "4 left its enqueue" + ); + } + + #[test] + fn a_user_write_raises_its_mark_before_the_applied_index_moves() { + let tracker = ProposeTracker::new(); + let marks = TenantMarks::default(); + let mut lane = Lane::new(4); + let mut write = slot(1, SlotState::Concluded(PrefixStep::Record(true))); + write.write_mark = Some((7, 500)); + write.user_write = true; + write.collection = Some("docs".to_owned()); + let mut index_change = slot(2, SlotState::Concluded(PrefixStep::Record(true))); + index_change.write_mark = Some((7, 900)); + lane.push(write); + lane.push(index_change); + + lane.settle(&tracker, &marks); + let mark = marks.get(4, 7).expect("the write's mark"); + assert_eq!( + mark.hlc, 500, + "an entry that writes no user data raises no mark" + ); + assert_eq!(mark.collection.as_deref(), Some("docs")); + } + + #[test] + fn a_refused_user_write_raises_no_mark() { + let tracker = ProposeTracker::new(); + let marks = TenantMarks::default(); + let mut lane = Lane::new(4); + let mut refused = slot(1, SlotState::Running); + refused.write_mark = Some((7, 500)); + refused.user_write = true; + refused.collection = Some("docs".to_owned()); + lane.push(refused); + + assert!(lane.conclude(1, false, |_| PrefixStep::Record(true))); + lane.settle(&tracker, &marks); + assert_eq!( + marks.get(4, 7), + None, + "a write that changed no row must not refuse a restore" + ); + } + + #[test] + fn a_barrier_completes_only_after_every_earlier_entry() { + let tracker = ProposeTracker::new(); + let mut lane = Lane::new(3); + lane.push(slot(1, SlotState::Running)); + lane.push(slot(2, SlotState::Barrier)); + let mut barrier = tracker.register(3, 2, 0); + + lane.settle(&tracker, &TenantMarks::default()); + assert!(barrier.try_recv().is_err(), "entry 1 still runs"); + + assert!(lane.conclude(1, true, |_| PrefixStep::Record(true))); + lane.settle(&tracker, &TenantMarks::default()); + assert!(matches!(barrier.try_recv(), Ok(Ok(_)))); + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/metadata_floor.rs b/nodedb/src/control/distributed_applier/apply_loop/metadata_floor.rs new file mode 100644 index 000000000..d182027b2 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/metadata_floor.rs @@ -0,0 +1,130 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Hold a replicated write until this node's catalog reached the one its +//! proposer planned it against. +//! +//! A data group and the metadata group apply on independent loops. Without +//! this hold, a replica can apply a write to a collection before it applied +//! the metadata entries its proposer had already applied: +//! +//! - a same-name collection's purge, whose storage reclaim then removes the +//! write's rows; +//! - the collection's creation, whose registration the write needs. +//! +//! The proposer stamps its applied metadata index on the entry +//! (`ReplicatedEntry::metadata_floor`). The hold runs before the write's +//! enqueue, so the group's later entries wait behind it in log order. + +use std::sync::Arc; +use std::time::Duration; + +use nodedb_cluster::{METADATA_GROUP_ID, WaitOutcome}; + +use crate::control::distributed_applier::propose_tracker::ProposeTracker; +use crate::control::state::SharedState; + +use super::context::{FinishedApply, StartedEntry}; +use super::proposal_gate::{EntryOutcome, ledger_outcome}; +use super::start::Prepared; + +/// How long one wait slice lasts before the hold logs that it still waits. +const WAIT_SLICE: Duration = Duration::from_secs(5); + +/// The entry a hold belongs to. +#[derive(Debug, Clone, Copy)] +pub(super) struct HeldEntry { + pub group_id: u64, + pub log_index: u64, + pub proposal_key: u64, + /// The metadata index the write waits for. `0` holds nothing. + pub metadata_floor: u64, +} + +/// Put the metadata hold in front of `prepared`'s enqueue or apply. +pub(super) fn hold_for_metadata<'a>( + state: &'a Arc, + tracker: &'a Arc, + held: HeldEntry, + prepared: Prepared<'a>, +) -> Prepared<'a> { + if held.metadata_floor == 0 + || state.applied_index_watcher(METADATA_GROUP_ID).current() >= held.metadata_floor + { + return prepared; + } + match prepared { + Prepared::Enqueue(enqueue) => Prepared::Enqueue(Box::pin(async move { + match await_metadata_floor(state, held).await { + Ok(()) => enqueue.await, + Err(error) => StartedEntry::concluded(conclude_unapplied(tracker, held, error)), + } + })), + Prepared::Exclusive(apply) => Prepared::Exclusive(Box::pin(async move { + match await_metadata_floor(state, held).await { + Ok(()) => apply.await, + Err(error) => FinishedApply { + group_id: held.group_id, + log_index: held.log_index, + outcome: conclude_unapplied(tracker, held, error), + }, + } + })), + other @ (Prepared::Concluded(_) | Prepared::Barrier) => other, + } +} + +/// Wait until this node applied the metadata group through the entry's +/// floor. The wait has no deadline: applying the write earlier breaks the +/// order it holds. It fails only when the metadata group left this node. +async fn await_metadata_floor(state: &SharedState, held: HeldEntry) -> crate::Result<()> { + let watcher = state.applied_index_watcher(METADATA_GROUP_ID); + loop { + let waiting = Arc::clone(&watcher); + let floor = held.metadata_floor; + let outcome = tokio::task::spawn_blocking(move || waiting.wait_for(floor, WAIT_SLICE)) + .await + .map_err(|e| crate::Error::Internal { + detail: format!( + "raft group {} entry {}: the metadata catch-up wait did not finish: {e}", + held.group_id, held.log_index + ), + })?; + match outcome { + WaitOutcome::Reached => return Ok(()), + WaitOutcome::TimedOut => tracing::warn!( + group_id = held.group_id, + log_index = held.log_index, + metadata_floor = floor, + metadata_applied = watcher.current(), + "a replicated write waits for this node's metadata apply to reach the \ + catalog its proposer planned it against" + ), + WaitOutcome::GroupGone => { + return Err(crate::Error::Internal { + detail: format!( + "raft group {} entry {}: the metadata group left this node before it \ + applied index {floor}, the catalog the write was planned against; \ + the write stays unapplied and replays on the next boot", + held.group_id, held.log_index + ), + }); + } + } + } +} + +/// Resolve the entry's waiter with `error`. The entry is not durable, so it +/// holds the group's applied floor and replays on the next boot. +fn conclude_unapplied( + tracker: &ProposeTracker, + held: HeldEntry, + error: crate::Error, +) -> EntryOutcome { + let result = Err(error); + let applied = ledger_outcome(&result); + tracker.complete(held.group_id, held.log_index, held.proposal_key, result); + EntryOutcome::Applied { + durable: false, + result: Some(applied), + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/mod.rs b/nodedb/src/control/distributed_applier/apply_loop/mod.rs new file mode 100644 index 000000000..f6566ef51 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/mod.rs @@ -0,0 +1,51 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Background apply loop — reads committed Raft entries from the mpsc channel, +//! enqueues each through the shared Control-Plane write funnel (which appends +//! each entry's redo record on THIS replica before the enqueue), and resolves +//! propose waiters with the result. +//! +//! Entries of a group start in log order, so every core receives them in the +//! order the log fixed. Their outcomes are collected independently: a write +//! the core parks holds only its own position. Each group's applied index and +//! durable floor (see [`super::applied_index`]) advance in log order, to the +//! highest entry with every earlier entry finished and durable, so the next +//! boot replays only above the floor and no entry is applied by both WAL +//! replay and Raft log replay. +//! +//! Split by concern: +//! - [`driver`]: takes batches off the apply channel and collects finished +//! applies. +//! - [`pipeline`]: every group's lane and the applies that run. +//! - [`lane`]: one group's queued and started entries, settled in log order. +//! - [`start`]: prepares one entry and routes it to its apply path. +//! - [`context`]: the handles an apply borrows, and the futures the loop +//! collects. +//! - [`calvin_read_result`]: forwards a committed `CalvinReadResult` entry to +//! the local Calvin scheduler. +//! - [`write_dispatch`]: the generic decode + write-funnel enqueue path. +//! - [`transaction_redo`]: a committed transaction's redo, stamped with its +//! Raft entry and applied through the WAL replay arms. +//! - [`proposal_gate`]: skips a second committed copy of an applied proposal +//! and records each applied proposal in the ledger. +//! - [`group_watch`]: per-group second-apply detection and backup cut floors. +//! - [`bookkeeping`]: applied-floor persistence + Raft log compaction trigger. +//! - [`helpers`]: shared response/result classification helpers. +//! - [`metadata_floor`]: holds a write until this node's catalog reached the +//! one its proposer planned it against. + +mod bookkeeping; +mod calvin_read_result; +mod context; +mod driver; +mod group_watch; +mod helpers; +mod lane; +mod metadata_floor; +mod pipeline; +mod proposal_gate; +mod start; +mod transaction_redo; +mod write_dispatch; + +pub use driver::run_apply_loop; diff --git a/nodedb/src/control/distributed_applier/apply_loop/pipeline.rs b/nodedb/src/control/distributed_applier/apply_loop/pipeline.rs new file mode 100644 index 000000000..7f22690fe --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/pipeline.rs @@ -0,0 +1,285 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The apply pipeline over every group this node applies. +//! +//! Each group's entries start in log order and settle in log order. A write +//! starts with its enqueue, and the next entry of its group waits for that +//! enqueue to return. Once enqueued, the write runs on its core and finishes +//! whenever its core answers. A write the core parks holds its own position +//! and no other: later entries of its group start and finish, and other +//! groups never wait on it. Only the group's applied index and durable floor +//! wait for it to finish. + +use std::collections::HashMap; + +use futures::FutureExt; +use futures::stream::{FuturesUnordered, StreamExt}; + +use crate::control::distributed_applier::applier::ApplyBatch; +use crate::control::distributed_applier::proposal_ledger::ProposalLedger; + +use super::bookkeeping::record_durable_apply; +use super::context::{ApplyContext, ApplyFuture, LoopEvent, LoopFuture, Started, StartedEntry}; +use super::group_watch::GroupWatch; +use super::lane::{Lane, QueuedEntry, Slot, SlotState}; +use super::metadata_floor::{HeldEntry, hold_for_metadata}; +use super::proposal_gate::{EntryOutcome, ProposalGate}; +use super::start::{Prepared, prepare_entry}; + +/// Every group's lane, and the enqueues and applies that run. +pub(super) struct Pipeline<'a> { + ctx: ApplyContext<'a>, + lanes: HashMap, + gate: ProposalGate, + watch: GroupWatch, + running: FuturesUnordered>, +} + +impl<'a> Pipeline<'a> { + pub fn new(ctx: ApplyContext<'a>, ledger: ProposalLedger) -> Self { + Self { + ctx, + lanes: HashMap::new(), + gate: ProposalGate::new(ledger), + watch: GroupWatch::default(), + running: FuturesUnordered::new(), + } + } + + /// Queue a batch the applier handed off, behind its group's earlier + /// entries. + pub fn accept(&mut self, batch: ApplyBatch) { + let lane = self + .lanes + .entry(batch.group_id) + .or_insert_with(|| Lane::new(batch.group_id)); + lane.backlog + .extend(batch.entries.into_iter().map(QueuedEntry::new)); + } + + /// Whether any enqueue or apply runs. + pub fn has_running(&self) -> bool { + !self.running.is_empty() + } + + /// The next enqueue to return or apply to finish. `None` when nothing + /// runs. + pub async fn next_event(&mut self) -> Option> { + self.running.next().await + } + + /// Handle `event`, then every other event that is ready already. + pub fn handle(&mut self, event: LoopEvent<'a>) { + self.handle_one(event); + while let Some(Some(event)) = self.running.next().now_or_never() { + self.handle_one(event); + } + } + + fn handle_one(&mut self, event: LoopEvent<'a>) { + let (group_id, log_index, handled) = match event { + LoopEvent::Enqueued { + group_id, + log_index, + entry, + } => ( + group_id, + log_index, + self.enqueued(group_id, log_index, entry), + ), + LoopEvent::Finished(finished) => { + let gate = &mut self.gate; + let wrote_rows = finished.outcome.wrote_rows(); + let handled = self.lanes.get_mut(&finished.group_id).is_some_and(|lane| { + lane.conclude(finished.log_index, wrote_rows, |proposal_key| { + gate.conclude(proposal_key, true, finished.outcome) + }) + }); + (finished.group_id, finished.log_index, handled) + } + }; + if !handled { + // Every enqueue and apply the pipeline runs has a slot in its + // group's lane in the matching state: the pump pushes both + // together and only this call moves them on. + tracing::error!( + group_id, + log_index, + "an apply-loop event has no matching entry in its group's lane" + ); + } + } + + fn enqueued(&mut self, group_id: u64, log_index: u64, entry: StartedEntry<'a>) -> bool { + let StartedEntry { + started, + collection, + user_write, + } = entry; + let Some(lane) = self.lanes.get_mut(&group_id) else { + return false; + }; + match started { + Started::Running(apply) => { + self.running.push(finished_event(apply)); + lane.enqueued(log_index, SlotState::Running, collection, user_write) + } + Started::Concluded(outcome) => { + // The write concluded without reaching its core. It leaves + // its enqueue and concludes in one step. + let gate = &mut self.gate; + let wrote_rows = outcome.wrote_rows(); + lane.enqueued(log_index, SlotState::Running, collection, user_write) + && lane.conclude(log_index, wrote_rows, |proposal_key| { + gate.conclude(proposal_key, true, outcome) + }) + } + } + } + + /// Start every entry each group can start now, in log order. + pub fn pump(&mut self) { + let groups: Vec = self + .lanes + .iter() + .filter(|(_, lane)| !lane.backlog.is_empty()) + .map(|(group_id, _)| *group_id) + .collect(); + for group_id in groups { + self.pump_group(group_id); + } + } + + fn pump_group(&mut self, group_id: u64) { + while let Some(queued) = self.next_startable(group_id) { + let log_index = queued.entry.index; + let proposal_key = queued.proposal_key(); + // Stamped before the entry is prepared: a barrier raises the cut + // floor only for the entries after it. + let write_mark = queued.write_stamp().map(|(tenant_id, write_hlc)| { + (tenant_id, self.watch.commit_hlc(group_id, write_hlc)) + }); + // A second copy of an applied proposal never reaches the funnel, + // so its plan is classified here. Its first copy may sit above + // the saved floor, and this copy then carries the mark again. + let repeat_writes = + self.gate.prior_wrote_rows(proposal_key) && queued.plan_writes_user_data(); + let held = HeldEntry { + group_id, + log_index, + proposal_key, + metadata_floor: queued.metadata_floor(), + }; + let prepared = hold_for_metadata( + self.ctx.state, + self.ctx.tracker, + held, + prepare_entry(self.ctx, &mut self.watch, &self.gate, group_id, queued), + ); + let (state, blocks, user_write) = match prepared { + Prepared::Concluded(outcome) => { + let user_write = matches!(outcome, EntryOutcome::Repeat) && repeat_writes; + ( + SlotState::Concluded(self.gate.conclude(proposal_key, false, outcome)), + false, + user_write, + ) + } + Prepared::Barrier => (SlotState::Barrier, false, false), + Prepared::Enqueue(enqueue) => { + self.gate.open(proposal_key); + self.running + .push(Box::pin(enqueue.map(move |entry| LoopEvent::Enqueued { + group_id, + log_index, + entry, + }))); + // The enqueue reports whether the plan writes user data. + (SlotState::Starting, true, false) + } + Prepared::Exclusive(apply) => { + // An array op or cell write: user data. + self.gate.open(proposal_key); + self.running.push(finished_event(apply)); + (SlotState::Running, true, true) + } + }; + let Some(lane) = self.lanes.get_mut(&group_id) else { + return; + }; + if blocks { + lane.blocking = Some(log_index); + } + lane.push(Slot { + log_index, + proposal_key, + collection: None, + write_mark, + user_write, + state, + }); + } + } + + /// Take the next entry of `group_id` when it may start now. + /// + /// It waits while the group's previous write is in its enqueue or an + /// exclusive entry of the group runs, while it is exclusive and an + /// earlier entry of the group has not concluded, and while a copy of its + /// proposal runs: the ledger decides it once that copy concludes. + fn next_startable(&mut self, group_id: u64) -> Option { + let lane = self.lanes.get_mut(&group_id)?; + if lane.blocking.is_some() { + return None; + } + let front = lane.backlog.front()?; + if front.is_exclusive() && lane.has_running() { + return None; + } + if self.gate.in_flight(front.proposal_key()) { + return None; + } + lane.backlog.pop_front() + } + + /// Settle every group's concluded entries in log order, release their + /// window, and save each durable floor that moved. + /// + /// The tenant write marks of the settled entries are persisted before any + /// floor that covers them. An entry above the saved floor is delivered + /// again after a restart and records its mark again, so every committed + /// write keeps a durable mark. + pub fn settle(&mut self) { + let state = self.ctx.state; + let tracker = self.ctx.tracker; + for (group_id, lane) in &mut self.lanes { + let settled = lane.settle(tracker, &state.tenant_marks); + if settled > 0 { + tracker.window().release(*group_id, settled); + } + if !lane.floor_pending() { + continue; + } + // One save per pass that moved the floor, never one per entry: + // each save is an fsync, and the pass coalesces every apply that + // finished before it. + if let Err(error) = state.tenant_marks.persist(state.credentials.catalog()) { + // The floor stays where it is, so every entry above it keeps + // its place in the log. The next pass persists and saves again. + tracing::error!( + group_id = *group_id, + %error, + "apply loop: tenant write marks did not persist; the applied floor waits" + ); + continue; + } + if let Some(floor) = lane.take_floor_to_save() { + record_durable_apply(state, *group_id, floor); + } + } + } +} + +fn finished_event(apply: ApplyFuture<'_>) -> LoopFuture<'_> { + Box::pin(apply.map(LoopEvent::Finished)) +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/proposal_gate.rs b/nodedb/src/control/distributed_applier/apply_loop/proposal_gate.rs new file mode 100644 index 000000000..2d1325ee5 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/proposal_gate.rs @@ -0,0 +1,224 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-entry proposal identity gate: skip a second committed copy of a +//! proposal, and record every durably applied proposal in the ledger. See +//! [`crate::control::distributed_applier::proposal_ledger`]. +//! +//! Entries finish out of order, so the gate also knows which proposals are +//! in flight. A second copy of an in-flight proposal waits until the first +//! copy concludes, then the ledger decides it like any other copy. + +use std::collections::HashMap; + +use crate::control::distributed_applier::proposal_ledger::{ + AppliedOutcome, PriorApply, ProposalLedger, +}; +use crate::control::distributed_applier::propose_tracker::{AppliedWrite, ProposeTracker}; + +/// What applying one committed entry produced. +pub(super) enum EntryOutcome { + /// The entry carries no durable state and no proposal to record: it + /// neither advances nor breaks the applied prefix. + Skipped, + /// A second committed copy of a proposal this node already applied. Its + /// first copy's outcome is durable, so it extends the prefix. The ledger + /// already holds the proposal. + Repeat, + /// The entry was applied. `durable` says its outcome survives a restart. + /// `result` is what its waiter received, when the apply produced one. + Applied { + durable: bool, + result: Option, + }, +} + +impl EntryOutcome { + /// Whether the apply wrote the entry's rows. A refused or failed apply + /// wrote none, so it raises no tenant write mark. A second copy writes + /// nothing itself: its mark follows its first copy (see + /// [`ProposalGate::prior_wrote_rows`]). + pub fn wrote_rows(&self) -> bool { + match self { + Self::Skipped | Self::Repeat => false, + Self::Applied { + result: Some(outcome), + .. + } => outcome.is_ok(), + // An apply with no waiter result reports its success as `durable`. + Self::Applied { + durable, + result: None, + } => *durable, + } + } +} + +/// How a concluded entry moves its group's durable applied prefix. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) enum PrefixStep { + /// Neither advances nor breaks the prefix. + Neutral, + /// Extends the prefix when `true`, breaks it when `false`. + Record(bool), +} + +/// The outcome a waiter received, in the form the ledger keeps. A refusal is +/// kept as its typed code; an error with no code keeps its message. +pub(super) fn ledger_outcome(result: &crate::Result) -> AppliedOutcome { + match result { + Ok(applied) => Ok(applied.clone()), + Err(crate::Error::DataPlane(code)) => Err(code.clone()), + Err(other) => Err(crate::bridge::envelope::ErrorCode::Internal { + detail: other.to_string(), + }), + } +} + +/// The proposal ledger, plus the proposals whose apply has not concluded. +pub(super) struct ProposalGate { + ledger: ProposalLedger, + /// Keyed proposals started and not concluded, with their copy counts. + in_flight: HashMap, +} + +impl ProposalGate { + pub fn new(ledger: ProposalLedger) -> Self { + Self { + ledger, + in_flight: HashMap::new(), + } + } + + /// Whether a copy of `proposal_key` is started and not concluded. Key `0` + /// names no proposal and is never in flight. + pub fn in_flight(&self, proposal_key: u64) -> bool { + proposal_key != 0 && self.in_flight.contains_key(&proposal_key) + } + + /// Whether the first copy of `proposal_key` applied on this node and + /// wrote its rows. A key recovered from the WAL keeps no outcome, so it + /// counts as written: a mark too high only refuses a restore that FORCE + /// can override, and a mark too low lets a restore overwrite the write. + pub fn prior_wrote_rows(&self, proposal_key: u64) -> bool { + match self.ledger.prior(proposal_key) { + Some(PriorApply::Outcome(outcome)) => outcome.is_ok(), + Some(PriorApply::NoOutcome) => true, + None => false, + } + } + + /// When `proposal_key` already applied on this node, resolve the entry's + /// waiter with the first copy's outcome and return `true`: the caller + /// skips the entry. The first copy's outcome is durable: the record that + /// carries its key was durable before the floor passed it, or it applied + /// in this process. + pub fn skip_duplicate( + &self, + tracker: &ProposeTracker, + group_id: u64, + log_index: u64, + proposal_key: u64, + ) -> bool { + let Some(prior) = self.ledger.prior(proposal_key) else { + return false; + }; + let result = match prior { + PriorApply::Outcome(Ok(applied)) => Ok(applied.clone()), + PriorApply::Outcome(Err(code)) => Err(crate::Error::DataPlane(code.clone())), + PriorApply::NoOutcome => Ok(AppliedWrite::unversioned(Vec::new())), + }; + tracing::debug!( + group_id, + log_index, + proposal_key, + "skipping a second committed copy of an applied proposal" + ); + tracker.complete(group_id, log_index, proposal_key, result); + true + } + + /// Note that a copy of `proposal_key` started and has not concluded. + pub fn open(&mut self, proposal_key: u64) { + if proposal_key != 0 { + *self.in_flight.entry(proposal_key).or_insert(0) += 1; + } + } + + /// Conclude an entry of `proposal_key` with `outcome`: note a durable + /// apply in the ledger, close the copy `open` noted when `opened`, and + /// return how the entry moves its group's prefix. + /// + /// No separate marker is written: the entry's own records carry its key, + /// durable in the same write as its effect. + pub fn conclude( + &mut self, + proposal_key: u64, + opened: bool, + outcome: EntryOutcome, + ) -> PrefixStep { + if opened + && proposal_key != 0 + && let Some(copies) = self.in_flight.get_mut(&proposal_key) + { + *copies -= 1; + if *copies == 0 { + self.in_flight.remove(&proposal_key); + } + } + match outcome { + EntryOutcome::Skipped => PrefixStep::Neutral, + EntryOutcome::Repeat => PrefixStep::Record(true), + EntryOutcome::Applied { durable, result } => { + if durable { + self.ledger.note(proposal_key, result); + } + PrefixStep::Record(durable) + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::distributed_applier::proposal_ledger::PROPOSAL_LEDGER_CAPACITY; + + #[test] + fn a_copy_of_an_in_flight_proposal_is_decided_once_the_first_concludes() { + let tracker = ProposeTracker::new(); + let mut gate = ProposalGate::new(ProposalLedger::new(PROPOSAL_LEDGER_CAPACITY)); + gate.open(42); + assert!(gate.in_flight(42)); + assert!(!gate.skip_duplicate(&tracker, 1, 9, 42)); + + let step = gate.conclude( + 42, + true, + EntryOutcome::Applied { + durable: true, + result: Some(Ok(AppliedWrite::unversioned(Vec::new()))), + }, + ); + assert_eq!(step, PrefixStep::Record(true)); + assert!(!gate.in_flight(42)); + assert!(gate.skip_duplicate(&tracker, 1, 9, 42)); + } + + #[test] + fn a_failed_first_copy_leaves_the_second_copy_to_apply() { + let tracker = ProposeTracker::new(); + let mut gate = ProposalGate::new(ProposalLedger::new(PROPOSAL_LEDGER_CAPACITY)); + gate.open(7); + let step = gate.conclude( + 7, + true, + EntryOutcome::Applied { + durable: false, + result: None, + }, + ); + assert_eq!(step, PrefixStep::Record(false)); + assert!(!gate.in_flight(7)); + assert!(!gate.skip_duplicate(&tracker, 1, 9, 7)); + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/start.rs b/nodedb/src/control/distributed_applier/apply_loop/start.rs new file mode 100644 index 000000000..2edb3efe2 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/start.rs @@ -0,0 +1,216 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Prepare one committed entry, in log order: note it, skip a second copy of +//! an applied proposal, and route it to its apply path. +//! +//! Preparing never waits. A write leaves here as its enqueue, and the next +//! entry of its group waits for that enqueue to return, so every core +//! receives a group's writes in the order the log fixed. Other groups never +//! wait on it. + +use crate::control::array_sync::ArrayOpTarget; +use crate::control::array_sync::raft_apply::{ + AppliedPosition, ArraySchemaPayload, apply_array_op, apply_array_schema, +}; +use crate::control::wal_replication::ReplicatedWrite; +use crate::types::{DatabaseId, TenantId}; + +use super::calvin_read_result::{CalvinReadResultFields, forward_calvin_read_result}; +use super::context::{ApplyContext, ApplyFuture, EnqueueFuture, FinishedApply}; +use super::group_watch::GroupWatch; +use super::lane::QueuedEntry; +use super::proposal_gate::{EntryOutcome, ProposalGate}; +use super::transaction_redo::prepare_transaction_redo_entry; +use super::write_dispatch::{EntryScope, prepare_generic_entry}; + +/// How a prepared entry continues. +pub(super) enum Prepared<'a> { + /// The entry concluded while it was prepared. + Concluded(EntryOutcome), + /// A backup's cut barrier: it completes once every earlier entry of its + /// group settled. + Barrier, + /// The write's enqueue. The next entry of the group starts once it + /// returns. + Enqueue(EnqueueFuture<'a>), + /// An apply that awaits its own write. It starts with nothing else of its + /// group running, and the next entry starts once it finishes. + Exclusive(ApplyFuture<'a>), +} + +/// Prepare `queued`, the next entry of `group_id` in log order. +pub(super) fn prepare_entry<'a>( + ctx: ApplyContext<'a>, + watch: &mut GroupWatch, + gate: &ProposalGate, + group_id: u64, + queued: QueuedEntry, +) -> Prepared<'a> { + let QueuedEntry { entry, decoded } = queued; + let log_index = entry.index; + watch.note_apply(group_id, log_index); + + // A leader-change no-op committed where a proposer may wait. The + // proposer's data is gone; firing an empty success would tell it the + // write applied. `RetryableLeaderChange` makes the gateway re-propose. + if entry.data.is_empty() { + tracing::error!( + group_id, + log_index, + "leader-change no-op committed at index where a proposer was waiting; \ + surfacing RetryableLeaderChange so the gateway re-proposes" + ); + ctx.tracker.complete( + group_id, + log_index, + 0, + Err(crate::Error::RetryableLeaderChange { + group_id, + log_index, + }), + ); + return Prepared::Concluded(EntryOutcome::Skipped); + } + + // `0` for unparseable / pre-key entries; the tracker treats 0 as "no key" + // (no mismatch detection). + let applied_key = decoded.as_ref().map_or(0, |e| e.idempotency_key); + // The proposer's commit stamp, raised above any backup cut the log placed + // before this entry. The entry's mark carries it, not the instant this + // replica applies, so a late apply never records a write as newer than a + // backup taken after its ack. + let commit_hlc = watch.commit_hlc(group_id, decoded.as_ref().map_or(0, |e| e.write_hlc)); + // Database scope for the entry, read from the wire. The generic decode + // path returns no scope, so it is taken from the entry itself: a redo + // appended under the wrong scope replays into the wrong namespace. + let database_id = decoded + .as_ref() + .map_or(DatabaseId::DEFAULT, |e| DatabaseId::new(e.database_id)); + // The source the proposer stamped. An entry that does not decode applies + // nothing, so its source is never read. + let event_source = decoded + .as_ref() + .map_or(crate::event::EventSource::User, |e| e.event_source.into()); + let scope = EntryScope { + database_id, + event_source, + }; + + // A second committed copy of a proposal this node already applied (a + // re-proposal after a leader change whose first copy also committed) + // resolves its waiter with the first copy's result and applies nothing. + if gate.skip_duplicate(ctx.tracker, group_id, log_index, applied_key) { + return Prepared::Concluded(EntryOutcome::Repeat); + } + + let pos = AppliedPosition { + group_id, + log_index, + applied_key, + commit_hlc, + }; + let Some(replicated) = decoded else { + return prepare_generic_entry(ctx, pos, entry, scope, false); + }; + let tenant_id = TenantId::new(replicated.tenant_id); + let entry_database = DatabaseId::new(replicated.database_id); + match replicated.write { + ReplicatedWrite::ArrayOp { + array, + op_bytes, + provenance, + .. + } => { + // The op path submits through the write funnel, so its redo is + // durable before it reports success. A failure breaks the + // prefix: the entry must stay replayable. + Prepared::Exclusive(Box::pin(async move { + let applied_ok = apply_array_op( + ctx.state, + ctx.tracker, + pos, + ArrayOpTarget { + tenant_id, + database_id: entry_database, + array: &array, + }, + &op_bytes, + provenance.as_deref(), + ) + .await; + FinishedApply { + group_id, + log_index, + outcome: EntryOutcome::Applied { + durable: applied_ok, + result: None, + }, + } + })) + } + ReplicatedWrite::ArraySchema { + ref array, + ref snapshot_payload, + schema_hlc_bytes, + } => { + // The one applied branch that mints no WAL redo record, and it + // needs none: its whole effect is two fsync-committed redb + // transactions, the schema registry's snapshot row and the array + // catalog's entry, both written before it reports success. + let applied_ok = apply_array_schema( + ctx.state, + ctx.tracker, + pos, + ArraySchemaPayload { + tenant_id, + database_id: entry_database, + array, + snapshot_payload, + schema_hlc_bytes, + }, + ); + Prepared::Concluded(EntryOutcome::Applied { + durable: applied_ok, + result: None, + }) + } + ReplicatedWrite::ArrayCellPut { .. } | ReplicatedWrite::ArrayCellDelete { .. } => { + prepare_generic_entry(ctx, pos, entry, scope, true) + } + ReplicatedWrite::TransactionRedo { .. } => { + prepare_transaction_redo_entry(ctx, pos, &replicated) + } + ReplicatedWrite::CutBarrier { hlc } => { + // Every entry after the barrier records above the cut. + watch.raise_cut(group_id, hlc); + Prepared::Barrier + } + ReplicatedWrite::CalvinReadResult { + epoch, + position, + passive_vshard, + tenant_id, + ref values, + } => { + forward_calvin_read_result( + ctx.tracker, + ctx.calvin_read_result_senders, + pos, + CalvinReadResultFields { + target_vshard: replicated.vshard_id, + epoch, + position, + passive_vshard, + tenant_id, + values, + }, + ); + // A read result is forwarded to an in-memory Calvin scheduler and + // writes nothing durable, so it neither advances the prefix nor + // breaks it. The epoch it belongs to does not survive a restart, + // so a re-delivery could not usefully replay it. + Prepared::Concluded(EntryOutcome::Skipped) + } + _ => prepare_generic_entry(ctx, pos, entry, scope, false), + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/transaction_redo.rs b/nodedb/src/control/distributed_applier/apply_loop/transaction_redo.rs new file mode 100644 index 000000000..c33e7d669 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/transaction_redo.rs @@ -0,0 +1,128 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Apply path for a committed `ReplicatedWrite::TransactionRedo` entry. +//! +//! Every replica, the proposer included, applies the entry through +//! [`enqueue_transaction_redo`]: the redo record is appended to this node's WAL, +//! its header carrying the entry's idempotency key, and installed through the +//! WAL replay arms. The key makes the record the entry's applied-marker, so an +//! entry re-delivered after a restart is recognised by the proposal ledger and +//! skipped before it reaches here. +//! +//! A refusal the Data Plane proves applied nothing (a constraint verdict) is +//! final: every replica reaches it at the same log position against the same +//! state, and the funnel cancels the record in the WAL before it returns. The +//! cancelling marker carries the entry's key, so the ledger counts the +//! refusal as the entry's outcome. It advances the durable prefix like a +//! success, because replaying the entry can only refuse it again. + +use crate::bridge::envelope::Status; +use crate::control::array_sync::raft_apply::AppliedPosition; +use crate::control::distributed_applier::propose_tracker::{AppliedWrite, ProposeTracker}; +use crate::control::server::dispatch_utils::{SubmitOutcome, refusal_is_final}; +use crate::control::wal_replication::ReplicatedEntry; +use crate::control::wal_replication::decode::transaction_redo_payload; +use crate::control::wal_replication::transaction_redo::{RedoTarget, enqueue_transaction_redo}; +use crate::types::{DatabaseId, TenantId, VShardId}; + +use super::context::{ApplyContext, FinishedApply, Started, StartedEntry}; +use super::helpers::committed_response_result; +use super::proposal_gate::{EntryOutcome, ledger_outcome}; +use super::start::Prepared; + +/// Prepare one committed `TransactionRedo` entry. Its enqueue appends the +/// record and hands it to its core. The apply that follows resolves the +/// propose waiter. Its outcome says whether the entry's effect is durable on +/// this node, which is what the group's applied prefix records. +pub(super) fn prepare_transaction_redo_entry<'a>( + ctx: ApplyContext<'a>, + pos: AppliedPosition, + entry: &ReplicatedEntry, +) -> Prepared<'a> { + let ApplyContext { state, tracker, .. } = ctx; + let payload = match transaction_redo_payload(&entry.write) { + Ok(payload) => payload, + Err(error) => { + tracker.complete(pos.group_id, pos.log_index, pos.applied_key, Err(error)); + // The entry's own bytes are malformed; a re-delivery decodes the + // same bytes and fails the same way, so it holds the floor rather + // than skipping a committed transaction. + return Prepared::Concluded(EntryOutcome::Applied { + durable: false, + result: None, + }); + } + }; + let target = RedoTarget { + tenant_id: TenantId::new(entry.tenant_id), + database_id: DatabaseId::new(entry.database_id), + vshard_id: VShardId::new(entry.vshard_id), + }; + Prepared::Enqueue(Box::pin(async move { + let collection = payload.collections.first().cloned(); + let enqueued = enqueue_transaction_redo( + state, + target, + &payload, + pos.applied_key, + pos.carried_commit_hlc(), + ) + .await; + let started = match enqueued { + Ok(pending) => Started::Running(Box::pin(async move { + let submitted = pending.finish(state).await; + FinishedApply { + group_id: pos.group_id, + log_index: pos.log_index, + outcome: conclude_transaction_redo(tracker, pos, submitted), + } + })), + Err(error) => Started::Concluded(conclude_transaction_redo(tracker, pos, Err(error))), + }; + // A committed transaction's redo carries the rows it wrote. + StartedEntry { + started, + collection, + user_write: true, + } + })) +} + +/// Resolve a transaction redo's waiter from what the funnel returned. +fn conclude_transaction_redo( + tracker: &ProposeTracker, + pos: AppliedPosition, + submitted: crate::Result, +) -> EntryOutcome { + let (result, durable) = match submitted { + Ok(outcome) if outcome.response.status == Status::Ok => { + (Ok(AppliedWrite::from_response(&outcome.response)), true) + } + Ok(outcome) => { + let refused_finally = outcome + .response + .error_code + .as_deref() + .is_some_and(refusal_is_final); + ( + committed_response_result(&outcome.response), + refused_finally, + ) + } + Err(error) => { + tracing::warn!( + group_id = pos.group_id, + index = pos.log_index, + error = %error, + "applying committed transaction redo failed" + ); + (Err(error), false) + } + }; + let applied = ledger_outcome(&result); + tracker.complete(pos.group_id, pos.log_index, pos.applied_key, result); + EntryOutcome::Applied { + durable, + result: Some(applied), + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs b/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs new file mode 100644 index 000000000..4f55e6686 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs @@ -0,0 +1,307 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Generic per-entry apply path: decode the replicated entry, route +//! Raft-native array cell writes through the array-open bootstrap, and +//! enqueue everything else through the shared Control-Plane write funnel. +//! +//! The enqueue runs in log order when the entry starts. The outcome is +//! collected by the returned apply, in any order. + +use tracing::debug; + +use nodedb_physical::physical_plan::ArrayOp; +use nodedb_raft::message::LogEntry; + +use crate::bridge::envelope::{PhysicalPlan, Status}; +use crate::control::array_sync::raft_apply::{ + AppliedPosition, ArrayCellTarget, apply_array_cell_write, +}; +use crate::control::distributed_applier::propose_tracker::{AppliedWrite, ProposeTracker}; +use crate::control::server::dispatch_utils::{ + ChangeFeedOwner, SubmitWrite, WalDurability, WriteOrdering, enqueue_write, + error_is_final_refusal, +}; +use crate::control::wal_replication::from_replicated_entry; +use crate::types::{DatabaseId, TraceId}; + +use crate::control::server::shared::write_admission::plan_writes_user_data; + +use super::context::{ApplyContext, FinishedApply, Started, StartedEntry}; +use super::helpers::{committed_response_result, deterministic_crdt_fence_noop}; +use super::proposal_gate::{EntryOutcome, ledger_outcome}; +use super::start::Prepared; + +/// What a generic entry's apply takes from its decoded envelope. +#[derive(Debug, Clone, Copy)] +pub(super) struct EntryScope { + /// Database scope of the entry. + pub database_id: DatabaseId, + /// The source the proposer stamped. Every replica gives the write's + /// events this source. + pub event_source: crate::event::EventSource, +} + +/// Prepare a generic entry. `exclusive` marks a Raft-native array cell write: +/// its apply awaits the array-open bootstrap and its own write, so it runs +/// with nothing else of its group in flight. Every other entry leaves as its +/// enqueue. +pub(super) fn prepare_generic_entry<'a>( + ctx: ApplyContext<'a>, + pos: AppliedPosition, + entry: LogEntry, + scope: EntryScope, + exclusive: bool, +) -> Prepared<'a> { + if !exclusive { + return Prepared::Enqueue(Box::pin(enqueue_generic_entry(ctx, pos, entry, scope))); + } + Prepared::Exclusive(Box::pin(async move { + let outcome = match enqueue_generic_entry(ctx, pos, entry, scope).await.started { + Started::Running(apply) => return apply.await, + Started::Concluded(outcome) => outcome, + }; + FinishedApply { + group_id: pos.group_id, + log_index: pos.log_index, + outcome, + } + })) +} + +/// Decode `entry` and enqueue it: Raft-native array cell writes route through +/// the array-open bootstrap and the write funnel, and conclude here; everything +/// else is enqueued through the write funnel directly. +async fn enqueue_generic_entry<'a>( + ctx: ApplyContext<'a>, + pos: AppliedPosition, + entry: LogEntry, + scope: EntryScope, +) -> StartedEntry<'a> { + let EntryScope { + database_id, + event_source, + } = scope; + let ApplyContext { state, tracker, .. } = ctx; + let AppliedPosition { + group_id, + log_index, + applied_key, + .. + } = pos; + let decoded = from_replicated_entry(&entry.data, Some(state.surrogate_assigner.as_ref())); + let (tenant_id, vshard_id, plan, resolved_now_ms) = match decoded { + Ok(Some(t)) => t, + Ok(None) => { + // Couldn't deserialize — might be a different format or corrupted. + debug!( + group_id, + index = entry.index, + "skipping non-ReplicatedEntry commit" + ); + tracker.complete( + group_id, + entry.index, + applied_key, + Ok(AppliedWrite::unversioned(Vec::new())), + ); + // Prefix-neutral. This is a pure shape check over + // `entry.data`, so a re-delivery on the next boot decodes to + // `None` again and skips again — stalling the floor behind + // it buys nothing and costs a double-apply of every later + // write in the batch. It applied no state, so it must not + // advance the floor either. + return StartedEntry::concluded(EntryOutcome::Skipped); + } + Err(e) => { + tracing::warn!( + group_id, + index = entry.index, + error = %e, + "failed to decode replicated entry (surrogate bind error)" + ); + tracker.complete( + group_id, + entry.index, + applied_key, + Err(crate::Error::Internal { + detail: format!("decode replicated entry: {e}"), + }), + ); + // Breaks the prefix, unlike the `Ok(None)` skip above: this + // IS a write, and it failed against live surrogate-assigner + // state rather than on its own bytes, so a re-delivery can + // legitimately succeed. Holding the floor below it is what + // keeps it replayable. + return StartedEntry::concluded(EntryOutcome::Applied { + durable: false, + result: None, + }); + } + }; + + // Raft-native array cell writes (`ArrayCellPut` / `ArrayCellDelete`) + // decode to `PhysicalPlan::Array(Put | Delete)`. A follower must + // OPEN the array on the Data Plane before applying, so these route + // through the array-open bootstrap first — and then through the same + // write funnel as the generic branch below, which is what gives them + // a redo record and the fsync the applied floor asserts. No other + // `ReplicatedWrite` variant decodes to a `PhysicalPlan::Array`, so + // this match is exact, and the caller runs them as exclusive entries. + if matches!( + plan, + PhysicalPlan::Array(ArrayOp::Put { .. } | ArrayOp::Delete { .. }) + ) { + let applied_ok = apply_array_cell_write( + state, + tracker, + pos, + ArrayCellTarget { + tenant_id, + database_id, + vshard: vshard_id, + resolved_now_ms, + }, + plan, + ) + .await; + return StartedEntry::concluded(EntryOutcome::Applied { + durable: applied_ok, + result: None, + }); + } + + let collection = plan + .named_collections() + .first() + .map(|collection| (*collection).to_owned()); + let user_write = plan_writes_user_data(&plan); + debug!( + group_id, + log_index, + tenant_id = tenant_id.as_u64(), + vshard_id = vshard_id.as_u32(), + collection = collection.as_deref().unwrap_or(""), + user_write, + "applying a committed write entry" + ); + let enqueued = enqueue_write( + state, + SubmitWrite { + tenant_id, + database_id, + vshard_id, + plan, + trace_id: TraceId::generate(), + // Cluster mode has exactly ONE write-apply path: this loop. The + // proposing node does not execute locally before commit either. + // So the committed write keeps the source its proposer stamped + // on the entry. A client write stays `User`, and a restored row + // stays `Restore`, on every replica. + event_source, + txn_id: None, + // Auth ran on the proposing node before the entry was + // proposed; the committed entry carries no session user. + user_id: None, + // The redo record is appended HERE, on this replica, from the + // committed plan — the leader's WAL LSN is deliberately not + // carried on the wire, and the memory-only engines have no + // other durability path. `now_override` pins a TTL-bearing KV + // write's `expire_at_ms` to the instant the proposing node + // resolved, so this replica's redo record and its live apply + // install the byte-identical value every other replica does. + durability: WalDurability::AppendHere { + now_override: resolved_now_ms, + apply_key: applied_key, + commit_hlc: pos.carried_commit_hlc(), + }, + // Raft committed this entry at a fixed log index; every + // replica applies it in that order. Re-entering the + // write-admission gate would re-decide an ordering that is + // already final. + ordering: WriteOrdering::AlreadyOrdered, + // This loop runs on EVERY replica, so it must not publish: + // the node that proposed this entry already published the + // write's change event once, after commit + apply. Emitting + // here would give each subscriber one copy per replica plus + // a NOTIFY fan-out from each. See [`ChangeFeedOwner`]. + change_feed: ChangeFeedOwner::Unowned, + }, + ) + .await; + let started = match enqueued { + Ok(pending) => Started::Running(Box::pin(async move { + let submitted = pending.finish(state).await.map(|outcome| outcome.response); + FinishedApply { + group_id, + log_index, + outcome: conclude_generic_entry(tracker, pos, submitted), + } + })), + Err(error) => Started::Concluded(conclude_generic_entry(tracker, pos, Err(error))), + }; + StartedEntry { + started, + collection, + user_write, + } +} + +/// Resolve a generic entry's waiter from what the funnel returned, and report +/// the outcome its group's prefix records. +fn conclude_generic_entry( + tracker: &ProposeTracker, + pos: AppliedPosition, + submitted: crate::Result, +) -> EntryOutcome { + let AppliedPosition { + group_id, + log_index, + applied_key, + .. + } = pos; + + // The funnel returns an error-status response as `Ok`; a committed + // entry that failed to apply must surface to the propose waiter as a + // failure, not as an empty success. + let result = match submitted { + // The response carries this replica's post-write + // `coll_write_lsn` for the written collection, which the + // proposer needs as its read-your-writes floor: the version is + // minted here (the funnel's WAL append) and never travels on the + // wire, so the propose waiter is the only place it can be + // handed back. + Ok(resp) if resp.status == Status::Ok => Ok(AppliedWrite::from_response(&resp)), + Ok(resp) => committed_response_result(&resp), + Err(e) => { + tracing::warn!( + group_id, + index = log_index, + error = %e, + "applying committed write failed" + ); + // Passed through: the typed error already carries the + // caller's classification. + Err(e) + } + }; + + // A final refusal is the entry's outcome: its marker carries the key. + let applied_ok = result.is_ok() + || deterministic_crdt_fence_noop(&result) + || result.as_ref().is_err_and(error_is_final_refusal); + let applied = ledger_outcome(&result); + tracker.complete(group_id, log_index, applied_key, result); + + // Extend the group's durable prefix. On success the funnel's + // durable-at-ack barrier has already fsynced this entry's redo, + // which is exactly the fact the floor asserts — `log_index` is + // the data-plane applied watermark here, NOT raft's commit index. + // On failure the engines did not persist this index, so it is + // neither a safe compaction boundary nor a safe restart floor; + // breaking the prefix is what keeps a genuinely failed apply + // replayable rather than silently skipped. + EntryOutcome::Applied { + durable: applied_ok, + result: Some(applied), + } +} diff --git a/nodedb/src/control/distributed_applier/apply_window.rs b/nodedb/src/control/distributed_applier/apply_window.rs new file mode 100644 index 000000000..4e68802ca --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_window.rs @@ -0,0 +1,101 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-group bound on committed entries between hand-off and settle. +//! +//! The apply loop enqueues a group's entries in log order and collects each +//! outcome independently, so a parked write holds its own position and no +//! other. The window bounds how many of a group's entries the loop holds at +//! once. The applier refuses a batch that would pass the bound, and Raft +//! delivers it again on a later tick. Only the saturated group waits: every +//! other group keeps its own window. + +use std::collections::HashMap; +use std::sync::Mutex; + +use crate::bridge::dispatch::DATA_PLANE_QUEUE_CAPACITY; + +/// Entries of one group the apply loop may hold between hand-off and settle. +/// +/// One core's request queue. A group's writes spread over the cores that own +/// its vShards, so a window of one queue keeps every one of those cores busy. +/// A larger window only adds writes parked behind a full queue. +pub const APPLY_WINDOW_PER_GROUP: usize = DATA_PLANE_QUEUE_CAPACITY; + +/// Outstanding entry counts per group. +#[derive(Debug)] +pub struct ApplyWindow { + limit: usize, + outstanding: Mutex>, +} + +impl Default for ApplyWindow { + fn default() -> Self { + Self::new(APPLY_WINDOW_PER_GROUP) + } +} + +impl ApplyWindow { + /// A window of `limit` entries per group. + pub fn new(limit: usize) -> Self { + Self { + limit, + outstanding: Mutex::new(HashMap::new()), + } + } + + /// Take `count` entries of `group_id` into the window. Refuses when the + /// group holds entries and `count` more would pass the limit. A group that + /// holds none always takes the batch, so a batch longer than the limit + /// still applies. + pub fn try_admit(&self, group_id: u64, count: usize) -> bool { + let mut outstanding = self.outstanding.lock().unwrap_or_else(|p| p.into_inner()); + let held = outstanding.entry(group_id).or_insert(0); + if *held > 0 && *held + count > self.limit { + return false; + } + *held += count; + true + } + + /// Release `count` entries of `group_id`: settled by the loop, or never + /// handed to it. + pub fn release(&self, group_id: u64, count: usize) { + let mut outstanding = self.outstanding.lock().unwrap_or_else(|p| p.into_inner()); + if let Some(held) = outstanding.get_mut(&group_id) { + *held = held.saturating_sub(count); + } + } + + /// Entries of `group_id` the window holds. + pub fn outstanding(&self, group_id: u64) -> usize { + self.outstanding + .lock() + .unwrap_or_else(|p| p.into_inner()) + .get(&group_id) + .copied() + .unwrap_or(0) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_full_group_refuses_while_another_group_admits() { + let window = ApplyWindow::new(4); + assert!(window.try_admit(1, 3)); + assert!(!window.try_admit(1, 2), "group 1 would pass its bound"); + assert!(window.try_admit(2, 4), "group 2 keeps its own window"); + window.release(1, 1); + assert!(window.try_admit(1, 2)); + assert_eq!(window.outstanding(1), 4); + } + + #[test] + fn an_empty_group_takes_a_batch_longer_than_the_limit() { + let window = ApplyWindow::new(2); + assert!(window.try_admit(7, 5)); + assert!(!window.try_admit(7, 1)); + } +} diff --git a/nodedb/src/control/distributed_applier/mod.rs b/nodedb/src/control/distributed_applier/mod.rs index 8b9f1fd9d..5b808d1fd 100644 --- a/nodedb/src/control/distributed_applier/mod.rs +++ b/nodedb/src/control/distributed_applier/mod.rs @@ -8,9 +8,13 @@ pub mod applied_index; pub mod applier; pub mod apply_loop; +pub mod apply_window; +pub mod proposal_ledger; pub mod propose_tracker; pub use applied_index::{AppliedPrefix, save_applied_index}; pub use applier::{ApplyBatch, DistributedApplier, create_distributed_applier}; pub use apply_loop::run_apply_loop; -pub use propose_tracker::{AppliedWrite, ProposeResult, ProposeTracker}; +pub use apply_window::{APPLY_WINDOW_PER_GROUP, ApplyWindow}; +pub use proposal_ledger::{AppliedOutcome, PROPOSAL_LEDGER_CAPACITY, PriorApply, ProposalLedger}; +pub use propose_tracker::{AppliedWrite, ApplyingEntry, ProposeResult, ProposeTracker}; diff --git a/nodedb/src/control/distributed_applier/proposal_ledger.rs b/nodedb/src/control/distributed_applier/proposal_ledger.rs new file mode 100644 index 000000000..85374ffea --- /dev/null +++ b/nodedb/src/control/distributed_applier/proposal_ledger.rs @@ -0,0 +1,271 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Which replicated proposals this node already applied, keyed by proposal +//! identity. +//! +//! Every `ReplicatedEntry` carries an `idempotency_key` minted once by its +//! proposer. A re-proposal after `RetryableLeaderChange` sends the same bytes, +//! so both copies carry the same key. When the first copy committed at a log +//! index the proposer was not waiting on, both copies commit, and applying +//! both double-counts every non-idempotent effect (a materialized sum fold, a +//! columnar append). The apply loop checks this ledger before it applies an +//! entry and skips a copy whose key already applied. +//! +//! ## Durability +//! +//! Every WAL record an entry's apply appends carries the entry's key in its +//! header (`RecordHeader::apply_key`). The key is durable in the same write as +//! the effect it names, so no separate marker and no extra fsync is needed. +//! After a restart, [`ProposalLedger::from_records`] recovers the keys from +//! the replayed WAL: +//! +//! - A forward record the WAL cancelled with `WriteAborted` is absent from the +//! replayed records, so a refused entry stays replayable. +//! - A final refusal's `WriteAborted` marker carries the key, so the refusal +//! counts as the entry's outcome. +//! - An apply that writes no record of its own (a `wal=false` timeseries +//! ingest) appends a payload-free `ProposalApplied` record in its place. +//! +//! A duplicate delivered after a restart is recognised as long as the WAL +//! still retains a record of its original. +//! +//! ## Bound +//! +//! The ledger keeps the most recent [`PROPOSAL_LEDGER_CAPACITY`] keys across +//! every group and evicts the oldest past that. Keys are random 64-bit values, +//! so one set serves every group. A re-proposal commits within the proposer's +//! retry budget, which is orders of magnitude fewer entries than the +//! capacity. A checkpoint truncates the WAL, which bounds the recovered set +//! too. +//! +//! The ledger also keeps the outcome of every proposal this process applied, +//! so the waiter of a skipped copy receives what the first copy's waiter +//! received: the same payload and write version, or the same refusal. A key +//! recovered from the WAL has no outcome: no waiter from before the restart +//! survives it. + +use std::collections::{HashMap, VecDeque}; + +use nodedb_wal::WalRecord; + +use super::propose_tracker::AppliedWrite; +use crate::bridge::envelope::ErrorCode; + +/// Keys the ledger keeps before it evicts the oldest. +pub const PROPOSAL_LEDGER_CAPACITY: usize = 1 << 20; + +/// What an applied proposal's waiter received: the applied write, or the +/// typed refusal a durable refusal answered with. +pub type AppliedOutcome = Result; + +/// Applied proposal keys, in apply order. +#[derive(Debug)] +pub struct ProposalLedger { + results: HashMap>, + order: VecDeque, + capacity: usize, +} + +/// A proposal key the ledger already holds. +#[derive(Debug)] +pub enum PriorApply<'a> { + /// Applied in this process: the outcome its first copy produced. + Outcome(&'a AppliedOutcome), + /// Recovered from the WAL, or applied with no outcome to share. + NoOutcome, +} + +impl ProposalLedger { + /// An empty ledger that keeps `capacity` keys. + pub fn new(capacity: usize) -> Self { + Self { + results: HashMap::new(), + order: VecDeque::new(), + capacity: capacity.max(1), + } + } + + /// Recover the ledger from the keyed records in `records`, the node's + /// replayed WAL in WAL order, with cancelled forward records already + /// removed. A record with key `0` belongs to no proposal. + pub fn from_records(records: &[WalRecord], capacity: usize) -> Self { + let mut ledger = Self::new(capacity); + for record in records { + ledger.note(record.apply_key(), None); + } + ledger + } + + /// The prior apply of `proposal_key`, if any. Key `0` is the "no key" + /// sentinel of a legacy or synthetic entry and never matches. + pub fn prior(&self, proposal_key: u64) -> Option> { + if proposal_key == 0 { + return None; + } + Some(match self.results.get(&proposal_key)? { + Some(outcome) => PriorApply::Outcome(outcome), + None => PriorApply::NoOutcome, + }) + } + + /// Record that `proposal_key` applied, with the outcome its waiter + /// received when there is one. Key `0` is ignored. A key already held + /// keeps its place in the eviction order. + pub fn note(&mut self, proposal_key: u64, result: Option) { + if proposal_key == 0 { + return; + } + if self.results.insert(proposal_key, result).is_none() { + self.order.push_back(proposal_key); + } + while self.order.len() > self.capacity { + if let Some(oldest) = self.order.pop_front() { + self.results.remove(&oldest); + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; + use crate::wal::manager::NO_APPLY_KEY; + + fn applied(payload: &[u8]) -> AppliedWrite { + AppliedWrite { + payload: payload.to_vec(), + write_version: Lsn::new(9), + } + } + + #[test] + fn a_second_copy_of_a_proposal_finds_the_first_copys_outcome() { + let mut ledger = ProposalLedger::new(8); + assert!(ledger.prior(77).is_none()); + ledger.note(77, Some(Ok(applied(b"first")))); + match ledger.prior(77) { + Some(PriorApply::Outcome(Ok(result))) => assert_eq!(result.payload, b"first"), + other => panic!("expected the first copy's result, got {other:?}"), + } + assert!(ledger.prior(78).is_none()); + } + + #[test] + fn the_no_key_sentinel_never_deduplicates() { + let mut ledger = ProposalLedger::new(8); + ledger.note(0, None); + assert!(ledger.prior(0).is_none()); + } + + #[test] + fn the_oldest_key_is_evicted_past_capacity() { + let mut ledger = ProposalLedger::new(2); + ledger.note(10, None); + ledger.note(11, None); + ledger.note(12, None); + assert!(ledger.prior(10).is_none()); + assert!(ledger.prior(11).is_some()); + assert!(ledger.prior(12).is_some()); + } + + #[test] + fn a_key_noted_twice_holds_one_eviction_slot() { + let mut ledger = ProposalLedger::new(2); + ledger.note(10, None); + ledger.note(10, None); + ledger.note(11, None); + assert!(ledger.prior(10).is_some()); + assert!(ledger.prior(11).is_some()); + } + + fn open_wal(dir: &tempfile::TempDir) -> crate::wal::WalManager { + crate::wal::WalManager::open_for_testing(&dir.path().join("test.wal")).expect("open wal") + } + + #[test] + fn a_redelivered_key_is_skipped_after_the_ledger_rebuilds_from_keyed_records() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let (tid, vs, db) = (TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT); + wal.appender(0xAB) + .with_event_source(crate::event::EventSource::User) + .append_put(tid, vs, db, b"keyed") + .expect("append keyed put"); + wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) + .append_put(tid, vs, db, b"unkeyed") + .expect("append unkeyed put"); + wal.sync().expect("sync wal"); + + let ledger = ProposalLedger::from_records( + &wal.replay().expect("replay wal"), + PROPOSAL_LEDGER_CAPACITY, + ); + assert!(matches!(ledger.prior(0xAB), Some(PriorApply::NoOutcome))); + assert!(ledger.prior(0xCD).is_none()); + } + + #[test] + fn a_cancelled_forward_record_leaves_its_proposal_replayable() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let (tid, vs, db) = (TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT); + let forward = wal + .appender(0xAB) + .with_event_source(crate::event::EventSource::User) + .append_put(tid, vs, db, b"refused") + .expect("append keyed put"); + wal.appender(NO_APPLY_KEY) + .append_write_aborted(tid, vs, db, forward) + .expect("append unkeyed abort"); + let final_forward = wal + .appender(0xCD) + .with_event_source(crate::event::EventSource::User) + .append_put(tid, vs, db, b"refused for good") + .expect("append keyed put"); + wal.appender(0xCD) + .append_write_aborted(tid, vs, db, final_forward) + .expect("append keyed abort"); + wal.sync().expect("sync wal"); + + let ledger = ProposalLedger::from_records( + &wal.replay().expect("replay wal"), + PROPOSAL_LEDGER_CAPACITY, + ); + assert!( + ledger.prior(0xAB).is_none(), + "a non-final refusal stays replayable" + ); + assert!( + ledger.prior(0xCD).is_some(), + "a final refusal is the proposal's outcome" + ); + } + + #[test] + fn a_proposal_applied_marker_is_appended_only_under_an_apply_key() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let (tid, vs, db) = (TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT); + assert!( + wal.appender(NO_APPLY_KEY) + .append_proposal_applied(tid, vs, db) + .expect("append with no apply key") + .is_none() + ); + assert!( + wal.appender(0xEF) + .append_proposal_applied(tid, vs, db) + .expect("append under an apply key") + .is_some() + ); + wal.sync().expect("sync wal"); + + let ledger = ProposalLedger::from_records( + &wal.replay().expect("replay wal"), + PROPOSAL_LEDGER_CAPACITY, + ); + assert!(ledger.prior(0xEF).is_some()); + } +} diff --git a/nodedb/src/control/distributed_applier/propose_tracker.rs b/nodedb/src/control/distributed_applier/propose_tracker.rs index 50351f6fb..8d06a918f 100644 --- a/nodedb/src/control/distributed_applier/propose_tracker.rs +++ b/nodedb/src/control/distributed_applier/propose_tracker.rs @@ -8,6 +8,8 @@ use std::collections::HashMap; use std::collections::hash_map::Entry; use std::sync::{Arc, Mutex}; +use super::apply_window::ApplyWindow; + use tokio::sync::oneshot; use nodedb_cluster::GroupAppliedWatchers; @@ -23,7 +25,7 @@ use crate::types::Lsn; /// tracker resolves on the very node that applied locally — so the version this /// carries is that node's own, which is exactly what shard-local OCC validates /// against. -#[derive(Debug)] +#[derive(Debug, Clone)] pub struct AppliedWrite { /// The Data Plane's response payload, verbatim. pub payload: Vec, @@ -98,14 +100,41 @@ enum TrackerSlot { /// immediately if `complete()` already fired. pub struct ProposeTracker { slots: Mutex>, - /// Per-Raft-group apply watermark registry. Bumped on every - /// [`Self::complete`] so the watcher reflects "data applied on - /// this node up to index N" — the only semantic that's useful - /// for cross-node visibility waits. Tick-loop bumps cover the - /// metadata group (sync redb apply); this tracker covers data - /// groups (async SPSC dispatch through `run_apply_loop`). - /// `None` only in tests that don't exercise the watcher. + /// Per-Raft-group apply watermark registry. Bumped by + /// [`Self::note_applied`] once every entry of the group up to the index + /// finished, so the watcher reflects "data applied on this node up to + /// index N" — the only semantic that's useful for cross-node visibility + /// waits. Tick-loop bumps cover the metadata group (sync redb apply); + /// this tracker covers data groups (async SPSC dispatch through + /// `run_apply_loop`). `None` only in tests that don't exercise the + /// watcher. group_watchers: Option>, + /// Per group, the oldest committed entry the apply loop has not finished. + applying: Mutex>, + /// Per-group bound on entries between hand-off and settle, shared by the + /// applier that hands entries off and the loop that settles them. + window: Arc, +} + +/// The oldest committed entry of a group the apply loop has not finished: +/// what a propose waiter that timed out names as the entry its group's +/// applied index waits behind. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct ApplyingEntry { + pub group_id: u64, + pub log_index: u64, + /// The collection the entry's plan writes, once the apply decoded it. + pub collection: Option, +} + +impl std::fmt::Display for ApplyingEntry { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "group {} index {}", self.group_id, self.log_index)?; + if let Some(collection) = &self.collection { + write!(f, " writing '{collection}'")?; + } + Ok(()) + } } impl Default for ProposeTracker { @@ -119,6 +148,8 @@ impl ProposeTracker { Self { slots: Mutex::new(HashMap::new()), group_watchers: None, + applying: Mutex::new(HashMap::new()), + window: Arc::new(ApplyWindow::default()), } } @@ -129,6 +160,42 @@ impl ProposeTracker { self } + /// The per-group apply window. + pub fn window(&self) -> &Arc { + &self.window + } + + /// Record the oldest entry of `group_id` the apply loop has not finished, + /// or `None` once every entry it holds for the group finished. + pub fn note_applying(&self, group_id: u64, entry: Option) { + let mut applying = self.applying.lock().unwrap_or_else(|p| p.into_inner()); + match entry { + Some(entry) => { + applying.insert(group_id, entry); + } + None => { + applying.remove(&group_id); + } + } + } + + /// The oldest entry of `group_id` the apply loop has not finished, if any. + pub fn applying(&self, group_id: u64) -> Option { + self.applying + .lock() + .unwrap_or_else(|p| p.into_inner()) + .get(&group_id) + .cloned() + } + + /// Advance `group_id`'s applied watermark to `log_index`. The apply loop + /// calls it once every entry of the group up to `log_index` finished. + pub fn note_applied(&self, group_id: u64, log_index: u64) { + if let Some(w) = &self.group_watchers { + w.bump(group_id, log_index); + } + } + /// Register a waiter for a proposed entry. Returns a receiver that /// resolves when the entry is committed and executed. /// @@ -167,12 +234,27 @@ impl ProposeTracker { rx } + /// Drop the waiter a proposer registered at `(group_id, log_index)` and + /// stopped waiting on: its deadline passed, or this node left the group + /// and will never apply the index. A result already stored there stays. + pub fn abandon(&self, group_id: u64, log_index: u64) { + let mut slots = self.slots.lock().unwrap_or_else(|p| p.into_inner()); + if let Entry::Occupied(e) = slots.entry((group_id, log_index)) + && matches!(e.get(), TrackerSlot::Waiting { .. }) + { + e.remove(); + } + } + /// Complete a waiter after the entry has been committed and executed. /// /// If the proposer has already called `register()`, the result is sent /// immediately. If not, the result is stored so the next `register()` /// call picks it up without waiting. /// + /// Entries of one group complete in any order. The applied watermark + /// moves only through [`Self::note_applied`], in log order. + /// /// Returns true if a live waiter was found and notified, false otherwise. pub fn complete( &self, @@ -181,17 +263,6 @@ impl ProposeTracker { applied_key: u64, result: ProposeResult, ) -> bool { - // Bump the per-group apply watermark. Bumping unconditionally - // (success and error) keeps the watcher monotonic with raft's - // commit progression — a data-plane error means "the entry - // could not be applied" but the entry IS committed and Raft - // has advanced its applied index. Tests waiting on - // visibility care about the success path; liveness on the - // error path requires the bump too. - if let Some(w) = &self.group_watchers { - w.bump(group_id, log_index); - } - let mut slots = self.slots.lock().unwrap_or_else(|p| p.into_inner()); match slots.entry((group_id, log_index)) { Entry::Vacant(e) => { diff --git a/nodedb/src/control/event_action_error.rs b/nodedb/src/control/event_action_error.rs index 4f976f4e4..7e801f648 100644 --- a/nodedb/src/control/event_action_error.rs +++ b/nodedb/src/control/event_action_error.rs @@ -75,9 +75,9 @@ impl From for crate::Error { TriggerActionError::Plan { source } | TriggerActionError::LeaseAdmission { source } => { source } - TriggerActionError::Transaction { source } => crate::Error::Internal { - detail: source.to_string(), - }, + // The transaction error keeps its class: its statement or commit + // error, or the Data-Plane verdict that aborted the commit. + TriggerActionError::Transaction { source } => source.into(), } } } @@ -108,6 +108,7 @@ mod tests { TriggerActionError::Transaction { source: SystemTxnError::Commit { detail: "serialization failure against a concurrent write".to_owned(), + code: None, }, } } @@ -163,6 +164,37 @@ mod tests { } } + /// A commit abort keeps the Data-Plane verdict that decided it. + #[test] + fn a_failed_transaction_keeps_its_abort_code() { + let error = TriggerActionError::Transaction { + source: SystemTxnError::Commit { + detail: "serialization failure against a concurrent write".to_owned(), + code: Some(Box::new(crate::bridge::envelope::ErrorCode::ConflictRetry)), + }, + }; + match crate::Error::from(error) { + crate::Error::DataPlane(crate::bridge::envelope::ErrorCode::ConflictRetry) => {} + other => panic!("expected the abort code to survive, got {other:?}"), + } + } + + /// A commit that failed to dispatch keeps the dispatch error. + #[test] + fn a_failed_commit_dispatch_keeps_its_error() { + let error = TriggerActionError::Transaction { + source: SystemTxnError::CommitFailed { + source: crate::Error::DeadlineExceeded { + request_id: crate::types::RequestId::new(1), + }, + }, + }; + assert!(matches!( + crate::Error::from(error), + crate::Error::DeadlineExceeded { .. } + )); + } + #[test] fn a_render_failure_reports_as_a_bad_request() { let error = TriggerActionError::Rejected { diff --git a/nodedb/src/control/event_trigger.rs b/nodedb/src/control/event_trigger.rs index b9347ce83..717f1198c 100644 --- a/nodedb/src/control/event_trigger.rs +++ b/nodedb/src/control/event_trigger.rs @@ -37,8 +37,16 @@ pub async fn process_write_event( // An action's own writes come back through the Event Plane. Firing event // definitions on them lets an action that writes to the collection it // watches re-trigger itself without bound, so only the same sources that - // fire triggers fire event definitions. - if !matches!(event.source, EventSource::User | EventSource::Deferred) { + // fire triggers fire event definitions. A restored row fired its event + // definitions when it was first written. + let fires = match event.source { + EventSource::User | EventSource::Deferred => true, + EventSource::Trigger + | EventSource::RaftFollower + | EventSource::CrdtSync + | EventSource::Restore => false, + }; + if !fires { trace!( source = %event.source, collection = %event.collection, @@ -47,22 +55,17 @@ pub async fn process_write_event( return; } - let catalog = shared.credentials.catalog(); - let coll = match catalog.get_collection( + // The committed definitions, from memory: the Event Plane reads no redb. + let Some(event_defs) = shared.credentials.catalog().event_definitions( event.database_id, event.tenant_id.as_u64(), &event.collection, - ) { - Ok(Some(collection)) => collection, - _ => return, - }; - - if coll.event_defs.is_empty() { + ) else { return; - } + }; let op_str = event_operation(event.op); - for (index, event_def) in coll.event_defs.iter().enumerate() { + for (index, event_def) in event_defs.iter().enumerate() { let when_upper = event_def.when_condition.to_uppercase(); let matches = match when_upper.as_str() { "INSERT" => matches!(event.op, WriteOp::Insert | WriteOp::BulkInsert { .. }), @@ -413,6 +416,7 @@ mod tests { "doc'; DELETE FROM audit; --", )), lsn: Lsn::new(1), + record: None, database_id: DatabaseId::DEFAULT, tenant_id: TenantId::new(1), vshard_id: VShardId::new(0), diff --git a/nodedb/src/control/exec_receiver/backup_cut.rs b/nodedb/src/control/exec_receiver/backup_cut.rs new file mode 100644 index 000000000..1a44bb81a --- /dev/null +++ b/nodedb/src/control/exec_receiver/backup_cut.rs @@ -0,0 +1,38 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A backup's consistent cut on a remote source node. +//! +//! The coordinator of a backup picks the envelope watermark `W` and takes the +//! cut on its own replicas. A snapshot it sends to another node carries `W`. +//! That node takes the same cut on its replicas before it snapshots, so every +//! write committed below `W` has its final outcome in the snapshot, and every +//! write at or above `W` refuses a restore of the envelope. + +use nodedb_cluster::rpc_codec::TypedClusterError; +use nodedb_physical::physical_plan::{MetaOp, PhysicalPlan}; + +use crate::control::state::SharedState; + +use super::support::execution_error_to_typed; + +/// Take the cut a tenant snapshot plan asks for, then return the plan with +/// the request cleared. Every other plan passes through unchanged. +pub(super) async fn take_backup_cut( + state: &std::sync::Arc, + plan: PhysicalPlan, +) -> Result { + let PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { + tenant_id, + cut_watermark: Some(watermark), + }) = plan + else { + return Ok(plan); + }; + crate::control::backup::cut::cut_at(state, tenant_id, watermark) + .await + .map_err(execution_error_to_typed)?; + Ok(PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { + tenant_id, + cut_watermark: None, + })) +} diff --git a/nodedb/src/control/exec_receiver/executor.rs b/nodedb/src/control/exec_receiver/executor.rs index c7d32f5ea..e1dfe540a 100644 --- a/nodedb/src/control/exec_receiver/executor.rs +++ b/nodedb/src/control/exec_receiver/executor.rs @@ -21,6 +21,7 @@ use crate::control::state::SharedState; use crate::control::trace_export::EmitSpanParams; use crate::types::DatabaseId; +use super::backup_cut::take_backup_cut; use super::plan_decode::decode_plan; use super::request_validation::validate_request; use super::support::{PLAN_DECODE_FAILED, SinkOutcome, execution_error_to_typed}; @@ -108,7 +109,7 @@ impl LocalPlanExecutor { /// paths: validate deadline + descriptor versions, decode the plan, reject /// unresolved Exchange nodes. Returns `(plan, database_id, deadline)` on /// success or a typed cluster error to surface to the caller. - fn validate_and_decode( + async fn validate_and_decode( &self, req: &ExecuteRequest, ) -> Result< @@ -121,13 +122,15 @@ impl LocalPlanExecutor { > { let (deadline, database_id) = validate_request(&self.state, req)?; let plan = decode_plan(&self.state, database_id, req.tenant_id, &req.plan_bytes)?; + // A backup's snapshot plan takes the backup's cut on this node first. + let plan = take_backup_cut(&self.state, plan).await?; Ok((plan, database_id, deadline)) } /// One-shot execution: validate + decode, fan across all local cores, /// merge, and return the merged payload. async fn execute_plan_inner(&self, req: ExecuteRequest) -> ExecuteResponse { - let (plan, database_id, deadline) = match self.validate_and_decode(&req) { + let (plan, database_id, deadline) = match self.validate_and_decode(&req).await { Ok(t) => t, Err(e) => return ExecuteResponse::err(e), }; @@ -135,6 +138,22 @@ impl LocalPlanExecutor { let tenant_id = crate::types::TenantId::new(req.tenant_id); let trace_id = nodedb_types::TraceId(req.trace_id); + if let PhysicalPlan::ClusterEvent( + nodedb_physical::physical_plan::ClusterEventOp::TenantWriteMarks { + tenant_id: marks_tenant, + group_ids, + }, + ) = &plan + { + return super::tenant_marks::answer_tenant_marks( + &self.state, + *marks_tenant, + group_ids, + deadline, + ) + .await; + } + if let PhysicalPlan::ClusterEvent( nodedb_physical::physical_plan::ClusterEventOp::PublishTopic { database_id: topic_database_id, @@ -260,14 +279,23 @@ impl LocalPlanExecutor { // // The vshard is not carried on the wire; re-derive it as a pure // function of the plan's primary collection, matching the gateway - // router's `CollectionHomed` arm (`vshard_for_collection`). - let vshard_id = crate::types::VShardId::new( - crate::control::gateway::version_set::touched_collections(&plan) - .into_iter() - .next() - .map(|name| nodedb_cluster::routing::vshard_for_collection(database_id, &name)) - .unwrap_or(0), - ); + // router's `CollectionHomed` arm (`vshard_for_collection`). The plan + // carries the database-qualified name, de-qualified into the + // canonical key before hashing. + let vshard_raw = match crate::control::gateway::version_set::touched_collections(&plan) + .into_iter() + .next() + { + Some(name) => match nodedb_types::CollectionKey::from_qualified_str(database_id, &name) + { + Ok(key) => nodedb_cluster::routing::vshard_for_collection(key), + Err(error) => { + return ExecuteResponse::err(execution_error_to_typed(error.into())); + } + }, + None => 0, + }; + let vshard_id = crate::types::VShardId::new(vshard_raw); if let Err(error) = reject_unadmitted_crdt_apply(&plan) { return ExecuteResponse::err(error); } @@ -307,7 +335,21 @@ impl LocalPlanExecutor { { // Replicated writes carry no read watermark → 0: it floors a // session's later reads, and this RPC seam has no session. - Ok((payload, _write_version)) => ExecuteResponse::ok(vec![payload], 0, 0), + Ok((payload, write_version)) => { + // Replicas apply with `ChangeFeedOwner::Unowned`. This + // node proposed the write once, so it publishes the + // change event. + crate::control::server::dispatch_utils::publish_change_set_with_lsn( + &self.state, + tenant_id, + database_id, + crate::control::server::dispatch_utils::extract_write_change_set( + &plan, tenant_id, + ), + write_version, + ); + ExecuteResponse::ok(vec![payload], 0, 0) + } // A replicated write's apply verdict is a Data-Plane // verdict: carry its code, never flatten to internal. Err(e) => ExecuteResponse::err(execution_error_to_typed(e)), @@ -351,7 +393,7 @@ impl LocalPlanExecutor { req: ExecuteRequest, mut sink: impl ChunkSink, ) -> Option { - let (plan, database_id, deadline) = match self.validate_and_decode(&req) { + let (plan, database_id, deadline) = match self.validate_and_decode(&req).await { Ok(t) => t, Err(e) => return Some(e), }; diff --git a/nodedb/src/control/exec_receiver/mod.rs b/nodedb/src/control/exec_receiver/mod.rs index 7f1b62daa..477f8e6b8 100644 --- a/nodedb/src/control/exec_receiver/mod.rs +++ b/nodedb/src/control/exec_receiver/mod.rs @@ -2,9 +2,11 @@ //! Local execution of incoming `ExecuteRequest` / `ExecuteStreamRequest` RPCs. +mod backup_cut; pub mod executor; mod plan_decode; mod request_validation; mod support; +mod tenant_marks; pub use executor::LocalPlanExecutor; diff --git a/nodedb/src/control/exec_receiver/plan_decode.rs b/nodedb/src/control/exec_receiver/plan_decode.rs index dd764e5f1..83f22d209 100644 --- a/nodedb/src/control/exec_receiver/plan_decode.rs +++ b/nodedb/src/control/exec_receiver/plan_decode.rs @@ -59,12 +59,9 @@ pub(super) fn decode_plan( ) = &mut plan && *surrogate == nodedb_types::Surrogate::ZERO && !pk_bytes.is_empty() - && let Ok(Some(resolved)) = catalog_ref.get_surrogate_for_pk( - database_id, - crate::types::TenantId::new(tenant_id), - collection.as_str(), - pk_bytes, - ) + && let Ok(key) = nodedb_types::CollectionKey::from_qualified(database_id, collection) + && let Ok(Some(resolved)) = + catalog_ref.get_surrogate_for_pk(key, crate::types::TenantId::new(tenant_id), pk_bytes) { *surrogate = resolved; } diff --git a/nodedb/src/control/exec_receiver/tenant_marks.rs b/nodedb/src/control/exec_receiver/tenant_marks.rs new file mode 100644 index 000000000..49a769d63 --- /dev/null +++ b/nodedb/src/control/exec_receiver/tenant_marks.rs @@ -0,0 +1,63 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Answer a restoring node's request for this node's tenant write marks. +//! +//! The restoring node asks a replica of every data group it does not +//! replicate itself. This node answers only for groups it replicates, and only +//! once it applied every entry the groups committed before the request. + +use std::sync::Arc; +use std::time::Duration; + +use nodedb_cluster::rpc_codec::{ExecuteResponse, TypedClusterError}; + +use crate::control::backup::restore::guard::{encode_marks, local_tenant_marks}; +use crate::control::security::auth_fence::cluster::hosts_group; +use crate::control::state::SharedState; + +use super::support::execution_error_to_typed; + +/// This node's marks of `tenant_id` in `group_ids`, encoded for the wire. +/// +/// A group this node does not replicate is refused with +/// [`TypedClusterError::NotLeader`] naming the group. The asking node then +/// picks another replica. The refusal carries the replica this node's routing +/// table names for the group, when it names one other than this node. +/// `budget` is what remains of the asking statement's deadline. +pub(super) async fn answer_tenant_marks( + state: &Arc, + tenant_id: u64, + group_ids: &[u64], + budget: Duration, +) -> ExecuteResponse { + if let Some(&group_id) = group_ids.iter().find(|group| !hosts_group(state, **group)) { + return ExecuteResponse::err(TypedClusterError::NotLeader { + group_id, + leader_node_id: replica_hint(state, group_id), + leader_addr: None, + term: 0, + }); + } + let deadline = tokio::time::Instant::now() + budget; + let marks = match local_tenant_marks(state, tenant_id, group_ids, deadline).await { + Ok(marks) => marks, + Err(error) => return ExecuteResponse::err(execution_error_to_typed(error)), + }; + match encode_marks(&marks) { + Ok(payload) => ExecuteResponse::ok(vec![payload], 0, 0), + Err(error) => ExecuteResponse::err(execution_error_to_typed(error)), + } +} + +/// A replica of `group_id` other than this node, from this node's routing +/// table: the leader when it knows one, else the first voter, else the first +/// learner. +fn replica_hint(state: &SharedState, group_id: u64) -> Option { + let routing = state.cluster_routing.as_ref()?; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + let info = routing.group_info(group_id)?; + std::iter::once(info.leader) + .chain(info.members.iter().copied()) + .chain(info.learners.iter().copied()) + .find(|&node| node != 0 && node != state.node_id) +} diff --git a/nodedb/src/control/fail_gate.rs b/nodedb/src/control/fail_gate.rs new file mode 100644 index 000000000..eb7bb6dec --- /dev/null +++ b/nodedb/src/control/fail_gate.rs @@ -0,0 +1,71 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Async fail-point gates for Control-Plane code. +//! +//! A crash test sometimes needs one request parked at a precise point while +//! every other request proceeds. A synchronous fail point cannot do that: it +//! blocks the thread, and every task that shares the thread stalls with it. +//! The gate here awaits a file with a Tokio timer, so only the task that +//! reached it waits. The test creates the file to release it. +//! +//! Compiled only with the `failpoints` feature, and called only from async +//! Control-Plane code, never from the Data Plane. + +use std::time::Duration; + +use crate::bridge::envelope::PhysicalPlan; +use crate::fail_point::{FailAction, lookup}; +use crate::types::Lsn; + +/// How often a parked task checks for its release file. +const GATE_POLL: Duration = Duration::from_millis(20); + +/// Park until the file armed for `name` with `wait_file()` exists. No-op +/// when nothing is armed for `name`. +pub(crate) async fn wait(name: &str) { + let Some(FailAction::WaitForFile(path)) = lookup(name) else { + return; + }; + while !path.exists() { + tokio::time::sleep(GATE_POLL).await; + } +} + +/// The write funnel's gate: after a logged write's WAL record is appended and +/// before any core holds the request. Named per collection, +/// `funnel::before_dispatch::`, and only a write carrying a WAL +/// LSN reaches it. A committed redo parks on the gate of each collection it +/// writes. +/// +/// `funnel::before_dispatch::node::` parks the write only on +/// node `N`. An in-process cluster test shares one fail-point registry across +/// its nodes, so it names the node to hold one replica's apply. +pub(crate) async fn before_dispatch(node_id: u64, plan: &PhysicalPlan, wal_lsn: Option) { + if wal_lsn.is_none() { + return; + } + for collection in plan.named_collections() { + wait(&format!("funnel::before_dispatch::{collection}")).await; + wait(&format!( + "funnel::before_dispatch::node{node_id}::{collection}" + )) + .await; + } +} + +/// The Calvin scheduler's flush gate: after a committed transaction's redo +/// record is appended and before its flush reaches a core. Named per +/// collection, `calvin::before_flush::`. +/// +/// The scheduler cannot park on a timer, so the gate answers without waiting: +/// `true` while the file armed for one of the flush's collections is absent. +/// The scheduler then keeps the flush in its re-send queue and asks again on +/// its next pass. +pub(crate) fn holds_flush(plan: &PhysicalPlan) -> bool { + plan.named_collections().iter().any(|collection| { + matches!( + lookup(&format!("calvin::before_flush::{collection}")), + Some(FailAction::WaitForFile(path)) if !path.exists() + ) + }) +} diff --git a/nodedb/src/control/gateway/colocation_guard.rs b/nodedb/src/control/gateway/colocation_guard.rs index 408865832..7a62c119d 100644 --- a/nodedb/src/control/gateway/colocation_guard.rs +++ b/nodedb/src/control/gateway/colocation_guard.rs @@ -31,20 +31,21 @@ //! and never assume co-residence. use nodedb_physical::physical_plan::PhysicalPlan; -use nodedb_types::DatabaseId; +use nodedb_types::{CollectionKey, DatabaseId}; use crate::control::router::vshard::VShardRouter; use crate::control::state::SharedState; -use crate::types::VShardId; /// Resolve the Data-Plane core that owns `collection`'s vShard on THIS node, /// using the same `VShardRouter` the dispatch path uses so the guard can never /// drift from the real vShard→core mapping. +/// +/// `collection` is the plan's database-qualified name, de-qualified into the +/// canonical key. A name that does not de-qualify resolves to no core, so the +/// guard refuses the write. fn owning_core(router: &VShardRouter, database_id: DatabaseId, collection: &str) -> Option { - router.resolve(VShardId::from_collection_in_database( - database_id, - collection, - )) + let key = CollectionKey::from_qualified_str(database_id, collection).ok()?; + router.resolve(key.vshard()) } /// True when `coll_a` and `coll_b` resolve to DIFFERENT cores on `router`. diff --git a/nodedb/src/control/gateway/core.rs b/nodedb/src/control/gateway/core.rs index 02e5d201e..506257fcf 100644 --- a/nodedb/src/control/gateway/core.rs +++ b/nodedb/src/control/gateway/core.rs @@ -34,6 +34,7 @@ use nodedb_physical::physical_plan::PhysicalPlan; use super::dispatcher::{DispatchRouteParams, dispatch_route, statement_deadline_ms}; use super::fuser::fuse_payloads; use super::key_extractor::UnwiredKeyExtractor; +use super::outcome::GatewayOutcome; use super::plan_cache::PlanCache; use super::retry::retry_not_leader; use super::route::TaskRoute; @@ -163,7 +164,9 @@ impl Gateway { ctx: &QueryContext, plan: PhysicalPlan, ) -> Result<(Vec>, Vec<(VShardId, Lsn)>, Lsn), Error> { - self.execute_plan_with_watermarks(ctx, plan).await + self.execute_plan_outcome(ctx, plan) + .await + .map(GatewayOutcome::into_parts) } /// Execute a pre-planned `PhysicalPlan`, returning both the raw payloads and @@ -180,14 +183,17 @@ impl Gateway { checked: CloneCheckedTask, ) -> Result<(Vec>, Vec<(VShardId, Lsn)>, Lsn), Error> { let plan = authorized_plan_for_context(ctx, checked)?; - self.execute_plan_with_watermarks(ctx, plan).await + self.execute_plan_outcome(ctx, plan) + .await + .map(GatewayOutcome::into_parts) } - async fn execute_plan_with_watermarks( + /// Execute an authorized plan and keep every route's result detail. + pub(super) async fn execute_plan_outcome( &self, ctx: &QueryContext, plan: PhysicalPlan, - ) -> Result<(Vec>, Vec<(VShardId, Lsn)>, Lsn), Error> { + ) -> Result { let shared = self.shared()?; let span = info_span!( "gateway.execute", @@ -216,17 +222,6 @@ impl Gateway { status_ok: result.is_ok(), }); - // Advance per-tenant observed write-HLC high-water on any - // successful cluster dispatch (local or remote). Used by - // RESTORE staleness gate. Tracking on success of every - // gateway.execute is intentional: backup captures its - // envelope watermark AFTER its own fan-out, so a fresh - // backup's watermark always dominates the tenant_wm it - // itself advanced. - if result.is_ok() { - shared.advance_tenant_write_hlc(ctx.tenant_id.as_u64()); - } - result } @@ -240,9 +235,13 @@ impl Gateway { ctx: &QueryContext, plan: PhysicalPlan, version_set: GatewayVersionSet, - ) -> Result<(Vec>, Vec<(VShardId, Lsn)>, Lsn), Error> { + ) -> Result { let shared = self.shared()?; let routes = self.compute_routes(plan, ctx)?; + // A fan-out reads a `NotFound` route as a shard with no slice. A + // single-route plan keeps it as the Data Plane's verdict. + let single_route = routes.len() == 1; + let mut not_found = false; let deadline_ms = statement_deadline_ms(&shared); // Gateway-level byte ceiling: per-route `dispatch_to_data_plane` @@ -341,6 +340,7 @@ impl Gateway { // participating shard, never collapsed to a scalar, so a multi-route // read produces one read-set entry per shard. all_shard_watermarks.extend(outcome.shard_watermarks); + not_found = single_route && outcome.not_found; if outcome.read_version_lsn > max_read_version { max_read_version = outcome.read_version_lsn; } @@ -362,12 +362,17 @@ impl Gateway { // For broadcast scans, fuse all shard payloads into one. The per-shard // watermarks are NOT fused — each participating shard keeps its own // read-set entry. - if all_payloads.len() > 1 { - let fused = fuse_payloads(all_payloads)?; - Ok((vec![fused.payload], all_shard_watermarks, max_read_version)) + let payloads = if all_payloads.len() > 1 { + vec![fuse_payloads(all_payloads)?.payload] } else { - Ok((all_payloads, all_shard_watermarks, max_read_version)) - } + all_payloads + }; + Ok(GatewayOutcome { + payloads, + shard_watermarks: all_shard_watermarks, + read_version_lsn: max_read_version, + not_found, + }) } /// Compute routing decisions for a plan. diff --git a/nodedb/src/control/gateway/dispatch_remote.rs b/nodedb/src/control/gateway/dispatch_remote.rs index d572ed4aa..bdf0d253e 100644 --- a/nodedb/src/control/gateway/dispatch_remote.rs +++ b/nodedb/src/control/gateway/dispatch_remote.rs @@ -11,6 +11,7 @@ use std::sync::Arc; use futures::StreamExt; +use nodedb_cluster::ClusterError; use nodedb_cluster::rpc_codec::{ExecuteRequest, RaftRpc}; use tracing::debug; @@ -97,6 +98,7 @@ pub(super) async fn dispatch_remote( // and stamped it on this response, so carry it through. This // route did not observe a version of its own to report. read_version_lsn: resp.read_version_lsn, + not_found: false, }); } crate::control::server::exchange::Resolved::Plan(p) => *p, @@ -117,6 +119,7 @@ pub(super) async fn dispatch_remote( // serves an in-transaction read and no read-set entry consumes // this value. read_version_lsn: Lsn::ZERO, + not_found: false, }); } }; @@ -184,6 +187,7 @@ pub(super) async fn dispatch_remote( )], payloads: resp.payloads, read_version_lsn: Lsn::new(resp.read_version_lsn), + not_found: false, }) } } @@ -342,20 +346,58 @@ pub(super) async fn dispatch_remote_stream( Ok(Box::pin(head.chain(rest))) } -/// Map a pre-row [`nodedb_cluster::ClusterError`] from a streaming dispatch to a -/// retryable internal [`Error`]. +/// Map a pre-row [`nodedb_cluster::ClusterError`] from a streaming dispatch to +/// an [`Error`]. /// -/// A `StreamTerminal` carrying a typed `NotLeader` / `DescriptorMismatch` maps -/// through the same [`map_typed_cluster_error`] used by the one-shot path so the -/// gateway retry loop handles it identically. Any other cluster error becomes a -/// transport-style `NotLeader` (leader_node = 0) so the next attempt re-resolves -/// routing rather than re-entrenching an unreachable node. -fn map_stream_cluster_error(err: nodedb_cluster::ClusterError, vshard_id: u64) -> Error { +/// A typed error (`StreamTerminal`, `ShardExecution`) maps through the same +/// [`map_typed_cluster_error`] used by the one-shot path, so the gateway retry +/// loop handles it identically. A Data-Plane verdict keeps its code. Any other +/// cluster error becomes a transport-style `NotLeader` (leader_node = 0), so +/// the next attempt re-resolves routing rather than re-entrenching an +/// unreachable node. +fn map_stream_cluster_error(err: ClusterError, vshard_id: u64) -> Error { match err { - nodedb_cluster::ClusterError::StreamTerminal { error, .. } => { - map_typed_cluster_error(error, vshard_id) + ClusterError::StreamTerminal { error, .. } | ClusterError::ShardExecution { error, .. } => { + map_typed_cluster_error(*error, vshard_id) } - other => Error::NotLeader { + // A verdict from a shard that answered. Retrying it on another route + // repeats it, so it keeps its SQLSTATE. + ClusterError::DataPlane { code } => Error::DataPlane(code.into()), + other @ (ClusterError::Raft(_) + | ClusterError::VShardNotMapped { .. } + | ClusterError::GroupNotFound { .. } + | ClusterError::LearnerNotCaughtUp { .. } + | ClusterError::MigrationInProgress { .. } + | ClusterError::MigrationPauseBudgetExceeded { .. } + | ClusterError::NodeUnreachable { .. } + | ClusterError::GhostNotFound { .. } + | ClusterError::Transport { .. } + | ClusterError::ShardTimeout { .. } + | ClusterError::Storage { .. } + | ClusterError::Codec { .. } + | ClusterError::UnsupportedWireVersion { .. } + | ClusterError::CircuitOpen { .. } + | ClusterError::JoinGroupDisappeared { .. } + | ClusterError::JoinCommitTimeout { .. } + | ClusterError::ReadIndexNotLeader { .. } + | ClusterError::ReadIndexTimeout { .. } + | ClusterError::Config { .. } + | ClusterError::MigrationCheckpoint(_) + | ClusterError::MigrationRecovery(_) + | ClusterError::WrongOwner { .. } + | ClusterError::Calvin(_) + | ClusterError::SnapshotCrcMismatch { .. } + | ClusterError::SnapshotOffsetRegression { .. } + | ClusterError::PartialSnapshotCorrupt { .. } + | ClusterError::PartialSnapshotCleanupFailed { .. } + | ClusterError::SnapshotApplyFailed { .. } + | ClusterError::Mirror(_) + | ClusterError::BspBarrier(_) + | ClusterError::VectorGather(_) + | ClusterError::SpatialGather(_) + | ClusterError::Bm25Gather(_) + | ClusterError::TsGather(_) + | ClusterError::RemoteUntyped { .. }) => Error::NotLeader { vshard_id: VShardId::new((vshard_id % VShardId::COUNT as u64) as u32), leader_node: 0, leader_addr: format!("stream dispatch error: {other}"), diff --git a/nodedb/src/control/gateway/dispatcher.rs b/nodedb/src/control/gateway/dispatcher.rs index 3c818c405..915fcfc63 100644 --- a/nodedb/src/control/gateway/dispatcher.rs +++ b/nodedb/src/control/gateway/dispatcher.rs @@ -12,16 +12,20 @@ use std::sync::Arc; use nodedb_cluster::rpc_codec::TypedClusterError; use crate::Error; -use crate::bridge::envelope::PhysicalPlan; +use crate::bridge::envelope::{ErrorCode, PhysicalPlan, Response, Status}; +use crate::control::local_dispatch::reject_data_plane_error; use crate::control::server::dispatch_utils::{ - dispatch_to_data_plane_with_txn, reject_data_plane_error, + AutocommitWrite, dispatch_autocommit_write, dispatch_to_data_plane_with_txn, + extract_write_change_set, publish_change_set_with_lsn, }; use crate::control::server::result_stream::ResultStream; +use crate::control::server::shared::write_admission::plan_is_write; use crate::control::state::SharedState; use crate::types::{DatabaseId, Lsn, TenantId, TraceId, TxnId, VShardId}; use super::dispatch_remote::{RemoteDispatchArgs, dispatch_remote, dispatch_remote_stream}; use super::route::{RouteDecision, TaskRoute}; +use super::router::is_task_vshard_scoped; use super::version_check::check_descriptor_versions; use super::version_set::GatewayVersionSet; @@ -40,6 +44,11 @@ pub struct DispatchOutcome { /// read targets one collection, so one non-zero value survives — for /// cross-shard OCC read validation. pub read_version_lsn: Lsn, + /// The owning core refused the task with `ErrorCode::NotFound`. + /// + /// A fan-out reads it as a shard that holds no slice. A single-route + /// task reports it as the Data Plane's verdict on that task. + pub not_found: bool, } /// Parameters for [`dispatch_route`]. `txn_id` is session-transaction @@ -270,15 +279,15 @@ async fn dispatch_local( let resp = shared .vshard_admission_sequencer .run(vshard_id, || async { - dispatch_to_data_plane_with_txn( + dispatch_local_plan(LocalPlan { shared, tenant_id, database_id, vshard_id, - route.plan, + plan: route.plan, trace_id, - None, - ) + txn_id: None, + }) .await }) .await?; @@ -287,6 +296,7 @@ async fn dispatch_local( payloads: vec![resp.payload.to_vec()], shard_watermarks: vec![(vshard_id, resp.watermark_lsn)], read_version_lsn: resp.read_version_lsn, + not_found: is_not_found(&resp), }); } @@ -302,24 +312,34 @@ async fn dispatch_local( let (payload, write_version) = crate::control::wal_replication::propose_replicated_entry(shared, proposer, entry) .await?; + // Replicas apply with `ChangeFeedOwner::Unowned`. This node proposed + // the write once, so it publishes the change event. + publish_change_set_with_lsn( + shared, + tenant_id, + database_id, + extract_write_change_set(&route.plan, tenant_id), + write_version, + ); return Ok(DispatchOutcome { payloads: vec![payload], // A write carries no read watermark (Lsn::ZERO); its post-write // `coll_write_lsn` is surfaced via `read_version_lsn` instead. shard_watermarks: vec![(vshard_id, Lsn::ZERO)], read_version_lsn: write_version, + not_found: false, }); } - let resp = dispatch_to_data_plane_with_txn( + let resp = dispatch_local_plan(LocalPlan { shared, tenant_id, database_id, vshard_id, - route.plan, + plan: route.plan, trace_id, txn_id, - ) + }) .await?; // The remote sibling turns `ExecuteResponse.error` into `Err`; the local // route must reject its own error status the same way. Keeping only the @@ -330,9 +350,74 @@ async fn dispatch_local( payloads: vec![resp.payload.to_vec()], shard_watermarks: vec![(vshard_id, resp.watermark_lsn)], read_version_lsn: resp.read_version_lsn, + not_found: is_not_found(&resp), }) } +/// One plan this node applies on its own cores. +struct LocalPlan<'a> { + shared: &'a Arc, + tenant_id: TenantId, + database_id: DatabaseId, + vshard_id: VShardId, + plan: PhysicalPlan, + trace_id: TraceId, + txn_id: Option, +} + +/// Dispatch a plan to this node's cores on the route its class needs. +/// +/// A base-state write enters the funnel with `AppendHere`, which appends its +/// redo record under the write-admission guard. It reaches here when no Raft +/// proposal carries it: a standalone node, a plan with no replicated +/// encoding, or a write a transaction cannot buffer. The transaction meta-ops +/// own their durability, and a staged write is logged at COMMIT, so both take +/// the read route with everything else. +async fn dispatch_local_plan(local: LocalPlan<'_>) -> Result { + let LocalPlan { + shared, + tenant_id, + database_id, + vshard_id, + plan, + trace_id, + txn_id, + } = local; + if plan_is_write(&plan) && !is_task_vshard_scoped(&plan) { + return dispatch_autocommit_write( + shared, + AutocommitWrite { + tenant_id, + database_id, + vshard_id, + plan, + trace_id, + event_source: crate::event::EventSource::User, + txn_id, + }, + ) + .await; + } + dispatch_to_data_plane_with_txn( + shared, + tenant_id, + database_id, + vshard_id, + plan, + trace_id, + txn_id, + ) + .await +} + +/// Whether the core refused the task with `ErrorCode::NotFound`. +/// +/// `reject_data_plane_error` passes this refusal as an empty success. +/// The flag keeps the verdict for a caller that needs it. +fn is_not_found(resp: &Response) -> bool { + resp.status == Status::Error && resp.error_code.as_deref() == Some(&ErrorCode::NotFound) +} + /// Map a [`TypedClusterError`] to an internal [`Error`]. /// /// `NotLeader` is mapped such that the gateway retry loop can extract the @@ -383,7 +468,10 @@ pub(super) fn map_typed_cluster_error(err: TypedClusterError, vshard_id: u64) -> constraint, detail, }, - TypedClusterError::Internal { message, .. } => Error::Internal { detail: message }, + // A numeric class crosses as `Error::RemoteTyped`, so the client sees + // the SQLSTATE the executing node gave it. Only a code of 0 (no class) + // decodes as `Error::Internal`. + internal @ TypedClusterError::Internal { .. } => Error::from(internal), } } @@ -431,6 +519,21 @@ mod tests { } } + /// A remote error with a numeric class keeps it, never `Internal`. + #[test] + fn map_internal_keeps_its_numeric_class() { + let err = TypedClusterError::Internal { + code: u32::from(nodedb_types::error::ErrorCode::AUTHORIZATION_DENIED.0), + message: "permission denied on orders".into(), + }; + match map_typed_cluster_error(err, 0) { + Error::RemoteTyped { code, .. } => { + assert_eq!(code, nodedb_types::error::ErrorCode::AUTHORIZATION_DENIED); + } + other => panic!("expected RemoteTyped, got {other:?}"), + } + } + #[test] fn map_deadline_exceeded() { let err = TypedClusterError::DeadlineExceeded { elapsed_ms: 100 }; diff --git a/nodedb/src/control/gateway/error_map/class_parity.rs b/nodedb/src/control/gateway/error_map/class_parity.rs new file mode 100644 index 000000000..a6fd98a39 --- /dev/null +++ b/nodedb/src/control/gateway/error_map/class_parity.rs @@ -0,0 +1,1186 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Every Data-Plane `ErrorCode`, and each Control-Plane error a client acts +//! on, answers one class on native, pgwire and HTTP. +//! +//! pgwire renders a Data-Plane verdict as its SQLSTATE. A native client reads +//! the numeric `nodedb_types` code on the frame. The two agree when the +//! numeric code renders, through the numeric-code SQLSTATE table, in the same +//! SQLSTATE class (the first two characters) as the pgwire SQLSTATE. That same +//! table renders a code that crossed a node as a bare number, so agreement +//! also keeps a verdict's class across nodes. + +use nodedb_types::error::sqlstate; +use nodedb_types::sync::violation::ViolationType; +use nodedb_types::sync::wire::SyncProvenance; + +use crate::bridge::envelope::{CounterFault, ErrorCode, SyncHold}; +use crate::control::server::native::dispatch::{error_code_to_native, native_error_fields}; +use crate::control::server::pgwire::types::error_map::numeric_code_to_sqlstate; +use crate::control::server::pgwire::types::error_to_sqlstate; + +use super::gateway_map::GatewayErrorMap; + +/// A named encoder that turns an `Error` into its node-hop wire form. +type HopEncoder = ( + &'static str, + fn(crate::Error) -> nodedb_cluster::rpc_codec::TypedClusterError, +); + +/// An error builder, with the SQLSTATE and public code it must answer. +type ClassCase = ( + fn() -> crate::Error, + &'static str, + nodedb_types::error::ErrorCode, +); + +/// The number of `ErrorCode` variants [`variant_index`] numbers. +const VARIANT_COUNT: usize = 44; + +/// A dense index per variant. Exhaustive, so a new variant fails to compile +/// here until it gets an index, and [`every_variant_has_a_sample`] then fails +/// until [`samples`] carries it. +fn variant_index(code: &ErrorCode) -> usize { + match code { + ErrorCode::DeadlineExceeded => 0, + ErrorCode::RejectedConstraint { .. } => 1, + ErrorCode::RejectedPrevalidation { .. } => 2, + ErrorCode::RetryableRefusal { .. } => 3, + ErrorCode::SyncRejected { .. } => 4, + ErrorCode::SyncNotApplied { .. } => 5, + ErrorCode::NotFound => 6, + ErrorCode::RejectedAuthz { .. } => 7, + ErrorCode::ConflictRetry => 8, + ErrorCode::CrdtFrontierMismatch { .. } => 9, + ErrorCode::FanOutExceeded => 10, + ErrorCode::ResourcesExhausted => 11, + ErrorCode::RejectedDanglingEdge { .. } => 12, + ErrorCode::DuplicateWrite => 13, + ErrorCode::AppendOnlyViolation { .. } => 14, + ErrorCode::BalanceViolation { .. } => 15, + ErrorCode::PeriodLocked { .. } => 16, + ErrorCode::PeriodLockMisconfigured { .. } => 17, + ErrorCode::RetentionViolation { .. } => 18, + ErrorCode::LegalHoldActive { .. } => 19, + ErrorCode::StateTransitionViolation { .. } => 20, + ErrorCode::TransitionCheckViolation { .. } => 21, + ErrorCode::TypeGuardViolation { .. } => 22, + ErrorCode::TypeMismatch { .. } => 23, + ErrorCode::CounterFault { .. } => 24, + ErrorCode::InsufficientBalance { .. } => 25, + ErrorCode::RateExceeded { .. } => 26, + ErrorCode::CollectionDraining { .. } => 27, + ErrorCode::RecursionDepthExceeded { .. } => 28, + ErrorCode::UndefinedColumn { .. } => 29, + ErrorCode::Internal { .. } => 30, + ErrorCode::Unsupported { .. } => 31, + ErrorCode::RollbackFailed { .. } => 32, + ErrorCode::OllpRetryRequired => 33, + ErrorCode::TxnOverlayMemoryExceeded { .. } => 34, + ErrorCode::DivisionByZero => 35, + ErrorCode::UndefinedFunction { .. } => 36, + ErrorCode::DataException { .. } => 37, + ErrorCode::DispatchCapacity { .. } => 38, + ErrorCode::ExpiredBeforeExecution => 39, + ErrorCode::BadRequest { .. } => 40, + ErrorCode::TransactionRollback { .. } => 41, + ErrorCode::ActiveSqlTransaction { .. } => 42, + ErrorCode::DependentObjectsExist { .. } => 43, + } +} + +fn provenance() -> SyncProvenance { + SyncProvenance { + producer_id: 1, + epoch: 1, + stream_id: 1, + seq: 1, + } +} + +/// One sample per variant, plus one per value that picks a different +/// SQLSTATE: each constraint kind and each counter fault. +fn samples() -> Vec { + let text = || "detail".to_owned(); + let collection = || "c".to_owned(); + let mut samples = vec![ + ErrorCode::DeadlineExceeded, + ErrorCode::RejectedPrevalidation { reason: text() }, + ErrorCode::RetryableRefusal { reason: text() }, + ErrorCode::SyncRejected { + violation: ViolationType::PermissionDenied, + applied_seq: 1, + provenance: provenance(), + }, + ErrorCode::SyncRejected { + violation: ViolationType::RateLimited, + applied_seq: 1, + provenance: provenance(), + }, + ErrorCode::SyncNotApplied { + hold: SyncHold::Gap { expected: 2 }, + applied_seq: 1, + }, + ErrorCode::NotFound, + ErrorCode::RejectedAuthz { resource: text() }, + ErrorCode::ConflictRetry, + ErrorCode::CrdtFrontierMismatch { + expected: [0; 32], + actual: [1; 32], + }, + ErrorCode::FanOutExceeded, + ErrorCode::ResourcesExhausted, + ErrorCode::RejectedDanglingEdge { + missing_node: text(), + }, + ErrorCode::DuplicateWrite, + ErrorCode::AppendOnlyViolation { + collection: collection(), + }, + ErrorCode::BalanceViolation { + collection: collection(), + detail: text(), + }, + ErrorCode::PeriodLocked { + collection: collection(), + }, + ErrorCode::PeriodLockMisconfigured { + collection: collection(), + ref_table: "periods".into(), + status_column: "status".into(), + row_identity: "p1".into(), + }, + ErrorCode::RetentionViolation { + collection: collection(), + }, + ErrorCode::LegalHoldActive { + collection: collection(), + }, + ErrorCode::StateTransitionViolation { + collection: collection(), + detail: text(), + }, + ErrorCode::TransitionCheckViolation { + collection: collection(), + detail: text(), + }, + ErrorCode::TypeGuardViolation { + collection: collection(), + detail: text(), + }, + ErrorCode::TypeMismatch { + collection: collection(), + detail: text(), + }, + ErrorCode::InsufficientBalance { + collection: collection(), + detail: text(), + }, + ErrorCode::RateExceeded { + gate: "g".into(), + retry_after_ms: 10, + }, + ErrorCode::CollectionDraining { + collection: collection(), + }, + ErrorCode::RecursionDepthExceeded { + cte_name: "walk".into(), + max_depth: 100, + }, + ErrorCode::UndefinedColumn { column: "x".into() }, + ErrorCode::Internal { detail: text() }, + ErrorCode::Unsupported { detail: text() }, + ErrorCode::RollbackFailed { + entry_index: 0, + detail: text(), + }, + ErrorCode::OllpRetryRequired, + ErrorCode::TxnOverlayMemoryExceeded { limit: 1 << 20 }, + ErrorCode::DivisionByZero, + ErrorCode::UndefinedFunction { name: "f".into() }, + ErrorCode::DataException { detail: text() }, + ErrorCode::DispatchCapacity { reason: text() }, + ErrorCode::ExpiredBeforeExecution, + ErrorCode::BadRequest { detail: text() }, + ErrorCode::TransactionRollback { detail: text() }, + ErrorCode::ActiveSqlTransaction { detail: text() }, + ErrorCode::DependentObjectsExist { + object: "role \"analyst\"".into(), + detail: text(), + }, + ]; + for constraint in [ + "not_null", + "unique", + "generated_always", + "fk_missing", + "rls_policy", + "permission_denied", + "check", + ] { + samples.push(ErrorCode::RejectedConstraint { + constraint: constraint.into(), + detail: text(), + }); + } + for fault in [ + CounterFault::NotAnInteger, + CounterFault::NotAFloat, + CounterFault::IntegerOverflow, + CounterFault::NonFinite, + ] { + samples.push(ErrorCode::CounterFault { + collection: collection(), + fault, + }); + } + samples +} + +fn class(state: &str) -> &str { + state.get(..2).unwrap_or(state) +} + +#[test] +fn every_variant_has_a_sample() { + let mut seen = [false; VARIANT_COUNT]; + for code in samples() { + seen[variant_index(&code)] = true; + } + let missing: Vec = (0..VARIANT_COUNT).filter(|i| !seen[*i]).collect(); + assert!(missing.is_empty(), "variants with no sample: {missing:?}"); +} + +/// The native frame carries pgwire's SQLSTATE, and its numeric code has the +/// same SQLSTATE class, on both native renderings: the typed `Err` and the +/// raw response frame. +#[test] +fn every_data_plane_code_has_one_class_on_native_and_pgwire() { + for code in samples() { + let err = crate::Error::DataPlane(code.clone()); + let (_, pg_state, _) = error_to_sqlstate(&err); + + let native = native_error_fields(&err); + assert_eq!(native.sqlstate, pg_state, "native SQLSTATE for {code:?}"); + let native_state = numeric_code_to_sqlstate(native.code); + assert_eq!( + class(native_state), + class(pg_state), + "{code:?}: pgwire sends {pg_state}, native code {} renders {native_state}", + native.code + ); + + let frame = error_code_to_native(1, Some(&code)); + let payload = frame.error.expect("error frames carry a payload"); + assert_eq!( + payload.code, pg_state, + "response-frame SQLSTATE for {code:?}" + ); + assert_eq!( + payload.ndb_code, native.code.0, + "response-frame code for {code:?}" + ); + } +} + +/// A classified Data-Plane verdict never reads as a server fault over HTTP. +#[test] +fn classified_data_plane_codes_are_not_http_500() { + for code in samples() { + let err = crate::Error::DataPlane(code.clone()); + let (_, pg_state, _) = error_to_sqlstate(&err); + let (status, _) = GatewayErrorMap::to_http(&err); + if pg_state == sqlstate::INTERNAL_ERROR { + assert_eq!(status, 500, "{code:?}"); + } else { + assert_ne!(status, 500, "{code:?} is {pg_state} on pgwire"); + } + } +} + +/// The SQLSTATE status table agrees with the gateway status table for every +/// Data-Plane code. A DDL error and a query error of one class answer one +/// HTTP status. +#[test] +fn sqlstate_status_agrees_with_the_gateway_status() { + for code in samples() { + let err = crate::Error::DataPlane(code.clone()); + let (_, pg_state, _) = error_to_sqlstate(&err); + let (status, _) = GatewayErrorMap::to_http(&err); + assert_eq!( + GatewayErrorMap::sqlstate_to_http(pg_state), + status, + "{code:?} is {pg_state} on pgwire" + ); + } +} + +/// `Unsupported` is feature-not-supported on every surface. +#[test] +fn unsupported_is_feature_not_supported_everywhere() { + let err = crate::Error::DataPlane(ErrorCode::Unsupported { + detail: "not on this engine".into(), + }); + let (_, pg_state, _) = error_to_sqlstate(&err); + assert_eq!(pg_state, sqlstate::FEATURE_NOT_SUPPORTED); + let native = native_error_fields(&err); + assert_eq!(native.code, nodedb_types::error::ErrorCode::SQL_NOT_ENABLED); + assert_eq!(GatewayErrorMap::to_http(&err).0, 501); +} + +/// Control-Plane errors a client acts on. Each has a class of its own. +fn control_plane_samples() -> Vec { + vec![ + crate::Error::RetryableSchemaChanged { + descriptor: "orders".into(), + }, + crate::Error::SessionTokenExpired, + ] +} + +/// A Control-Plane error has one class on native and pgwire, and its native +/// numeric code renders in that class. +#[test] +fn control_plane_errors_have_one_class_on_native_and_pgwire() { + for err in control_plane_samples() { + let (_, pg_state, _) = error_to_sqlstate(&err); + assert_ne!(pg_state, sqlstate::INTERNAL_ERROR, "{err:?} has no class"); + + let native = native_error_fields(&err); + assert_eq!(native.sqlstate, pg_state, "native SQLSTATE for {err:?}"); + let native_state = numeric_code_to_sqlstate(native.code); + assert_eq!( + class(native_state), + class(pg_state), + "{err:?}: pgwire sends {pg_state}, native code {} renders {native_state}", + native.code + ); + } +} + +/// A schema change the server could not absorb is the retryable +/// serialization class on every surface. +#[test] +fn schema_change_is_a_retryable_serialization_failure() { + let err = crate::Error::RetryableSchemaChanged { + descriptor: "orders".into(), + }; + assert_eq!(error_to_sqlstate(&err).1, sqlstate::SERIALIZATION_FAILURE); + let native = native_error_fields(&err); + assert_eq!(native.sqlstate, sqlstate::SERIALIZATION_FAILURE); + assert_eq!(native.code, nodedb_types::error::ErrorCode::WRITE_CONFLICT); + assert!(crate::error_classify::classify(&err).is_retriable()); + let status = GatewayErrorMap::to_http(&err).0; + assert_eq!(status, 409); + assert_eq!( + GatewayErrorMap::sqlstate_to_http(sqlstate::SERIALIZATION_FAILURE), + status + ); +} + +/// An expired session token is invalid authorization on every surface. +#[test] +fn expired_session_token_is_invalid_authorization_everywhere() { + let err = crate::Error::SessionTokenExpired; + assert_eq!(error_to_sqlstate(&err).1, sqlstate::AUTH_TOKEN_EXPIRED.0); + let native = native_error_fields(&err); + assert_eq!(native.sqlstate, sqlstate::AUTH_TOKEN_EXPIRED.0); + assert_eq!(native.code, nodedb_types::error::ErrorCode::AUTH_EXPIRED); + assert_eq!( + numeric_code_to_sqlstate(native.code), + sqlstate::AUTH_TOKEN_EXPIRED.0 + ); + let status = GatewayErrorMap::to_http(&err).0; + assert_eq!(status, 401); + assert_eq!( + GatewayErrorMap::sqlstate_to_http(sqlstate::AUTH_TOKEN_EXPIRED.0), + status + ); +} + +/// The number of `crate::Error` variants [`error_variant_index`] numbers. +const ERROR_VARIANT_COUNT: usize = 110; + +/// A dense index per `crate::Error` variant. Exhaustive, so a new variant +/// fails to compile here until it gets an index, and +/// [`every_error_variant_has_a_sample`] then fails until +/// [`error_samples`] carries it. +pub(crate) fn error_variant_index(err: &crate::Error) -> usize { + use crate::Error as E; + match err { + E::RejectedConstraint { .. } => 0, + E::TxnOverlayMemoryExceeded { .. } => 1, + E::RejectedAuthz { .. } => 2, + E::OffsetRegression { .. } => 3, + E::DeadlineExceeded { .. } => 4, + E::ConflictRetry { .. } => 5, + E::CalvinSerializationConflict => 6, + E::CalvinParticipantError => 7, + E::RejectedPrevalidation { .. } => 8, + E::RetryableRefusal { .. } => 9, + E::AppendOnlyViolation { .. } => 10, + E::BalanceViolation { .. } => 11, + E::MaterializedSumTargetNotFound { .. } => 12, + E::MaterializedSumResolutionMissing { .. } => 13, + E::PeriodLocked { .. } => 14, + E::PeriodLockMisconfigured { .. } => 15, + E::RetentionViolation { .. } => 16, + E::LegalHoldActive { .. } => 17, + E::StateTransitionViolation { .. } => 18, + E::TransitionCheckViolation { .. } => 19, + E::TypeGuardViolation { .. } => 20, + E::TypeMismatch { .. } => 21, + E::InsufficientBalance { .. } => 22, + E::RateExceeded { .. } => 23, + E::CollectionNotFound { .. } => 24, + E::DocumentNotFound { .. } => 25, + E::CollectionDeactivated { .. } => 26, + E::VShardAdmissionCapacityExceeded { .. } => 27, + E::CrdtAdmissionRetriesExhausted { .. } => 28, + E::CrdtAdmissionInvalidPlan { .. } => 29, + E::CrdtAdmissionCallerFence => 30, + E::CrdtApplyRequiresAdmission => 31, + E::CrdtApplyForbiddenInTransaction => 32, + E::NotInTransactionBlock { .. } => 33, + E::CrdtAdmissionTimeout { .. } => 34, + E::NoLeader { .. } => 35, + E::NotLeader { .. } => 36, + E::FanOutExceeded { .. } => 37, + E::CrossCollectionNotColocated { .. } => 38, + E::SourceFrozen { .. } => 39, + E::CloneWriteRequiresMaterialize { .. } => 40, + E::BadRequest { .. } => 41, + E::BackupTenantMismatch { .. } => 42, + E::BackupKeyMismatch => 43, + E::QuotaOvercommit { .. } => 44, + E::PlanError { .. } => 45, + E::FeatureNotSupported { .. } => 46, + E::UndefinedFunction { .. } => 47, + E::UndefinedObject { .. } => 48, + E::ObjectNotInPrerequisiteState { .. } => 49, + E::UndefinedColumn { .. } => 50, + E::AmbiguousColumn { .. } => 51, + E::UnknownStrictField { .. } => 52, + E::DivisionByZero => 53, + E::DataException { .. } => 54, + E::InvalidLimitValue { .. } => 55, + E::RetryableSchemaChanged { .. } => 56, + E::RetryableLeaderChange { .. } => 57, + E::GroupQuorumUnavailable { .. } => 58, + E::GroupMarksUnavailable { .. } => 59, + E::MetadataLeaderUnavailable => 60, + E::AuthorizationStateBehind { .. } => 61, + E::ExecutionLimitExceeded { .. } => 62, + E::LimitExceeded { .. } => 63, + E::Wal(_) => 64, + E::Dispatch { .. } => 65, + E::DispatchCapacity { .. } => 66, + E::Storage { .. } => 67, + E::ColdStorage { .. } => 68, + E::Serialization { .. } => 69, + E::Codec { .. } => 70, + E::SegmentCorrupted { .. } => 71, + E::MemoryExhausted { .. } => 72, + E::Backpressure { .. } => 73, + E::Crdt(_) => 74, + E::Io(_) => 75, + E::Config { .. } => 76, + E::Encryption { .. } => 77, + E::Bridge { .. } => 78, + E::VersionCompat { .. } => 79, + E::Internal { .. } => 80, + E::Shaping(_) => 81, + E::RemoteTyped { .. } => 82, + E::DescriptorVersionAnomaly { .. } => 83, + E::CollectionPurgeRowMissing { .. } => 84, + E::CatalogIntegrityViolation { .. } => 85, + E::DataPlane(_) => 86, + E::Promql(_) => 87, + E::DependentObjectsExist { .. } => 88, + E::CascadeCycle { .. } => 89, + E::CrossShardInExplicitTransaction => 90, + E::SequencerUnavailable => 91, + E::SessionCapExceeded { .. } => 92, + E::SessionIdleTimeout => 93, + E::SessionTokenExpired => 94, + E::SessionKilledByAdmin => 95, + E::SessionUserDropped => 96, + E::OidcProviderTenantUnbound => 97, + E::OidcProviderTenantUnavailable { .. } => 98, + E::ExternalRoleUndefined { .. } => 99, + E::OidcNoDefaultDatabase { .. } => 100, + E::TenantVectorDimExceeded { .. } => 101, + E::TenantGraphDepthExceeded { .. } => 102, + E::RoleInheritanceCycle { .. } => 103, + E::RoleInheritanceDepthExceeded { .. } => 104, + E::OllpExhausted { .. } => 105, + E::MirrorReadOnly { .. } => 106, + E::StaleReadNotLeader { .. } => 107, + E::RoleInUse { .. } => 108, + E::Ddl(_) => 109, + } +} + +/// One sample per `crate::Error` variant. +pub(crate) fn error_samples() -> Vec { + use crate::Error as E; + use crate::types::{DatabaseId, RequestId, TenantId, VShardId}; + + let text = || "detail".to_owned(); + let collection = || "c".to_owned(); + vec![ + E::RejectedConstraint { + collection: collection(), + constraint: "unique".into(), + detail: text(), + }, + E::TxnOverlayMemoryExceeded { limit: 1 << 20 }, + E::RejectedAuthz { + tenant_id: TenantId::new(1), + resource: text(), + }, + E::OffsetRegression { + stream: "s".into(), + group: "g".into(), + partition_id: 0, + current_lsn: 2, + current_sequence: 2, + attempted_lsn: 1, + attempted_sequence: 1, + }, + E::DeadlineExceeded { + request_id: RequestId::new(1), + }, + E::ConflictRetry { + collection: collection(), + document_id: "d".into(), + }, + E::CalvinSerializationConflict, + E::CalvinParticipantError, + E::RejectedPrevalidation { + constraint: "check".into(), + reason: text(), + }, + E::RetryableRefusal { reason: text() }, + E::AppendOnlyViolation { + collection: collection(), + detail: text(), + }, + E::BalanceViolation { + collection: collection(), + detail: text(), + }, + E::MaterializedSumTargetNotFound { + target_collection: "t".into(), + join_column: "k".into(), + join_value: "1".into(), + }, + E::MaterializedSumResolutionMissing { + target_collection: "t".into(), + join_column: "k".into(), + join_value: "1".into(), + }, + E::PeriodLocked { + collection: collection(), + detail: text(), + }, + E::PeriodLockMisconfigured { + collection: collection(), + ref_table: "periods".into(), + status_column: "status".into(), + row_identity: "p1".into(), + }, + E::RetentionViolation { + collection: collection(), + detail: text(), + }, + E::LegalHoldActive { + collection: collection(), + detail: text(), + }, + E::StateTransitionViolation { + collection: collection(), + detail: text(), + }, + E::TransitionCheckViolation { + collection: collection(), + detail: text(), + }, + E::TypeGuardViolation { + collection: collection(), + detail: text(), + }, + E::TypeMismatch { + collection: collection(), + key: "k".into(), + detail: text(), + }, + E::InsufficientBalance { + collection: collection(), + key: "k".into(), + detail: text(), + }, + E::RateExceeded { + gate: "g".into(), + detail: text(), + retry_after_ms: 10, + }, + E::CollectionNotFound { + tenant_id: TenantId::new(1), + collection: collection(), + }, + E::DocumentNotFound { + collection: collection(), + document_id: "d".into(), + }, + E::CollectionDeactivated { + tenant_id: TenantId::new(1), + collection: collection(), + retention_expires_at_ns: 1, + }, + E::VShardAdmissionCapacityExceeded { + vshard_id: VShardId::new(1), + capacity: 4, + }, + E::CrdtAdmissionRetriesExhausted { + vshard_id: VShardId::new(1), + attempts: 3, + }, + E::CrdtAdmissionInvalidPlan { reason: "empty" }, + E::CrdtAdmissionCallerFence, + E::CrdtApplyRequiresAdmission, + E::CrdtApplyForbiddenInTransaction, + E::NotInTransactionBlock { + statement: "VACUUM".into(), + }, + E::CrdtAdmissionTimeout { + vshard_id: VShardId::new(1), + timeout_ms: 10, + }, + E::NoLeader { + vshard_id: VShardId::new(1), + }, + E::NotLeader { + vshard_id: VShardId::new(1), + leader_node: 2, + leader_addr: "10.0.0.1:9000".into(), + }, + E::FanOutExceeded { + shards_touched: 9, + limit: 8, + }, + E::CrossCollectionNotColocated { + op: "insert-select", + source_collection: "a".into(), + target_collection: "b".into(), + }, + E::SourceFrozen { + database_id: DatabaseId::new(7), + }, + E::CloneWriteRequiresMaterialize { + collection: collection(), + engine: "kv".into(), + database: "db".into(), + reason: "shadowed", + }, + E::BadRequest { detail: text() }, + E::BackupTenantMismatch { + expected: 1, + actual: 2, + }, + E::BackupKeyMismatch, + E::QuotaOvercommit { + field: "max_storage".into(), + detail: text(), + }, + E::PlanError { detail: text() }, + E::FeatureNotSupported { detail: text() }, + E::UndefinedFunction { name: "f".into() }, + E::UndefinedObject { + kind: "sequence", + name: "s".into(), + }, + E::ObjectNotInPrerequisiteState { + object: "s".into(), + detail: text(), + }, + E::UndefinedColumn { column: "x".into() }, + E::AmbiguousColumn { + column: "id".into(), + }, + E::UnknownStrictField { + collection: collection(), + column: "x".into(), + }, + E::DivisionByZero, + E::DataException { detail: text() }, + E::InvalidLimitValue { + clause: "LIMIT", + value: "-1".into(), + }, + E::RetryableSchemaChanged { + descriptor: "orders".into(), + }, + E::RetryableLeaderChange { + group_id: 1, + log_index: 2, + }, + E::GroupQuorumUnavailable { + group_id: 1, + voters: vec![1, 2, 3], + unreachable: vec![2, 3], + }, + E::GroupMarksUnavailable { + group_id: 1, + refused_by: vec![2], + }, + E::MetadataLeaderUnavailable, + E::AuthorizationStateBehind { detail: text() }, + E::ExecutionLimitExceeded { detail: text() }, + E::LimitExceeded { + limit_name: "max_rows", + value: 10, + max: 5, + }, + E::Wal(nodedb_wal::WalError::Sealed), + E::Dispatch { detail: text() }, + E::DispatchCapacity { + scope: crate::DispatchCapacityScope::QueueFull { + core_id: 0, + capacity: 4, + }, + }, + E::Storage { + engine: "kv".into(), + detail: text(), + }, + E::ColdStorage { detail: text() }, + E::Serialization { + format: "msgpack".into(), + detail: text(), + }, + E::Codec { detail: text() }, + E::SegmentCorrupted { detail: text() }, + E::MemoryExhausted { + engine: "kv".into(), + }, + E::Backpressure { + engine: nodedb_mem::EngineId::Vector, + }, + E::Crdt(nodedb_crdt::CrdtError::ConstraintViolation { + constraint: "unique".into(), + collection: collection(), + detail: text(), + }), + E::Io(std::io::Error::other("disk")), + E::Config { detail: text() }, + E::Encryption { detail: text() }, + E::Bridge { detail: text() }, + E::VersionCompat { detail: text() }, + E::Internal { detail: text() }, + E::Shaping(Box::new(nodedb_types::NodeDbError::bad_request(text()))), + E::RemoteTyped { + code: nodedb_types::error::ErrorCode::WRITE_CONFLICT, + message: text(), + }, + E::DescriptorVersionAnomaly { + descriptor: "orders".into(), + carried: 5, + prior: 2, + }, + E::CollectionPurgeRowMissing { + database_id: 1, + tenant_id: 1, + name: collection(), + }, + E::CatalogIntegrityViolation { + entry_kind: "PutCollection".into(), + detail: text(), + }, + E::DataPlane(ErrorCode::NotFound), + E::Promql(crate::control::promql::PromqlError::UnexpectedEof), + E::DependentObjectsExist { + tenant_id: 1, + root_kind: "collection", + root_name: collection(), + dependent_count: 1, + dependents: vec![("view".into(), "v".into())], + }, + E::CascadeCycle { + tenant_id: 1, + root: collection(), + depth: 64, + }, + E::CrossShardInExplicitTransaction, + E::SequencerUnavailable, + E::SessionCapExceeded { cap: 8 }, + E::SessionIdleTimeout, + E::SessionTokenExpired, + E::SessionKilledByAdmin, + E::SessionUserDropped, + E::OidcProviderTenantUnbound, + E::OidcProviderTenantUnavailable { tenant_id: 1 }, + E::ExternalRoleUndefined { + subject: "alice".into(), + role: "auditor".into(), + tenant_id: 1, + }, + E::OidcNoDefaultDatabase { + sub: "alice".into(), + }, + E::TenantVectorDimExceeded { + dim: 4096, + limit: 1024, + }, + E::TenantGraphDepthExceeded { + depth: 20, + limit: 10, + }, + E::RoleInheritanceCycle { + child: "a".into(), + parent: "b".into(), + }, + E::RoleInheritanceDepthExceeded { depth: 9, limit: 8 }, + E::OllpExhausted { + retries: 3, + cause: crate::OllpExhaustedCause::PredicateDrift, + }, + E::MirrorReadOnly { + database: "db".into(), + }, + E::StaleReadNotLeader { + database: "db".into(), + source_cluster: "src".into(), + detail: text(), + }, + E::RoleInUse { + role: "analyst".into(), + dependents: crate::control::security::role_assignment::RoleDependents::Users(vec![ + "bob".into(), + ]), + }, + E::Ddl(Box::new( + crate::control::server::shared::ddl::DdlError::new( + sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, + text(), + ), + )), + ] +} + +#[test] +fn every_error_variant_has_a_sample() { + let mut seen = [false; ERROR_VARIANT_COUNT]; + for err in error_samples() { + seen[error_variant_index(&err)] = true; + } + let missing: Vec = (0..ERROR_VARIANT_COUNT).filter(|i| !seen[*i]).collect(); + assert!( + missing.is_empty(), + "error variants with no sample: {missing:?}" + ); +} + +/// Every `crate::Error` variant answers the HTTP status its pgwire SQLSTATE +/// class has. Only an internal or system error reads as a 500. +#[test] +fn every_error_variant_has_the_http_status_of_its_sqlstate() { + for err in error_samples() { + let (_, pg_state, _) = error_to_sqlstate(&err); + let (status, _) = GatewayErrorMap::to_http(&err); + assert_eq!( + status, + GatewayErrorMap::sqlstate_to_http(pg_state), + "{err:?} is {pg_state} on pgwire" + ); + let server_fault = matches!(class(pg_state), "XX" | "58"); + assert_eq!( + status == 500, + server_fault, + "{err:?} is {pg_state} on pgwire but HTTP {status}" + ); + } +} + +/// Every `crate::Error` variant renders the SQLSTATE class it renders locally +/// after it crosses a node hop, through both wire encoders and the decoder. +#[test] +fn every_error_variant_keeps_its_class_across_a_node_hop() { + use nodedb_cluster::rpc_codec::TypedClusterError; + + use crate::control::cluster::data_plane_error_wire::execution_error_to_typed; + + let encoders: [HopEncoder; 2] = [ + ("execution_error_to_typed", execution_error_to_typed), + ("From", TypedClusterError::from), + ]; + for (name, encode) in encoders { + for (err, twin) in error_samples().into_iter().zip(error_samples()) { + let (_, local, _) = error_to_sqlstate(&err); + let rebuilt = crate::Error::from(encode(twin)); + let (_, remote, _) = error_to_sqlstate(&rebuilt); + assert_eq!( + class(remote), + class(local), + "{name}: {err:?} is {local} locally but {remote} after the hop as {rebuilt:?}" + ); + } + } +} + +/// The SQLSTATE each Control-Plane variant renders where it has a class of +/// its own, pinned by variant index. +fn classified_sqlstates() -> Vec<(usize, &'static str)> { + vec![ + (3, sqlstate::INVALID_PARAMETER_VALUE), + (27, sqlstate::TOO_MANY_CONNECTIONS), + (28, sqlstate::SERIALIZATION_FAILURE), + (29, sqlstate::SYNTAX_ERROR), + (30, sqlstate::SYNTAX_ERROR), + (31, sqlstate::SYNTAX_ERROR), + (32, sqlstate::ACTIVE_SQL_TRANSACTION), + (34, sqlstate::QUERY_CANCELED.0), + (44, sqlstate::QUOTA_OVERCOMMIT), + (62, sqlstate::SYNTAX_ERROR), + (63, sqlstate::SYNTAX_ERROR), + (87, sqlstate::SYNTAX_ERROR), + (88, sqlstate::DEPENDENT_OBJECTS_STILL_EXIST), + (90, sqlstate::ACTIVE_SQL_TRANSACTION), + (91, sqlstate::SYNTAX_ERROR), + (92, sqlstate::SYNTAX_ERROR), + (93, sqlstate::SYNTAX_ERROR), + (95, sqlstate::SYNTAX_ERROR), + (96, sqlstate::SYNTAX_ERROR), + (97, sqlstate::SYNTAX_ERROR), + (98, sqlstate::SYNTAX_ERROR), + (99, sqlstate::SYNTAX_ERROR), + (100, sqlstate::SYNTAX_ERROR), + (101, sqlstate::QUOTA_EXCEEDED), + (102, sqlstate::QUOTA_EXCEEDED), + (103, sqlstate::SYNTAX_ERROR), + (104, sqlstate::SYNTAX_ERROR), + (106, sqlstate::READ_ONLY_SQL_TRANSACTION), + (107, sqlstate::STALE_READ_NOT_LEADER), + (108, sqlstate::DEPENDENT_OBJECTS_STILL_EXIST), + (109, sqlstate::DEPENDENT_OBJECTS_STILL_EXIST), + ] +} + +/// A client-facing Control-Plane variant renders its own SQLSTATE, never the +/// internal-error default. +#[test] +fn client_facing_variants_render_their_own_sqlstate() { + let expected = classified_sqlstates(); + let mut seen = 0; + for err in error_samples() { + let index = error_variant_index(&err); + if let Some((_, state)) = expected.iter().find(|(i, _)| *i == index) { + assert_eq!(error_to_sqlstate(&err).1, *state, "{err:?}"); + seen += 1; + } + } + assert_eq!(seen, expected.len(), "a pinned variant has no sample"); +} + +/// A variant with a dedicated public code renders that code's class on the +/// numeric table too, so native and remote renderings agree with pgwire. +#[test] +fn dedicated_codes_render_the_class_of_their_variant() { + use nodedb_types::error::ErrorCode as Ec; + + assert_eq!( + numeric_code_to_sqlstate(Ec::QUOTA_OVERCOMMIT), + sqlstate::QUOTA_OVERCOMMIT + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::TENANT_VECTOR_DIM_EXCEEDED), + sqlstate::QUOTA_EXCEEDED + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::TENANT_GRAPH_DEPTH_EXCEEDED), + sqlstate::QUOTA_EXCEEDED + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::MIRROR_READ_ONLY), + sqlstate::READ_ONLY_SQL_TRANSACTION + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::STALE_READ_NOT_LEADER), + sqlstate::STALE_READ_NOT_LEADER + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::TRANSACTION_ROLLBACK), + sqlstate::TRANSACTION_ROLLBACK + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::ACTIVE_SQL_TRANSACTION), + sqlstate::ACTIVE_SQL_TRANSACTION + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::DEPENDENT_OBJECTS_EXIST), + sqlstate::DEPENDENT_OBJECTS_STILL_EXIST + ); +} + +/// The transaction-state and dependency variants render their exact +/// SQLSTATE after a node hop through both encoders, and carry the public +/// code of that class. +#[test] +fn transaction_and_dependency_variants_keep_their_sqlstate_across_a_hop() { + use nodedb_cluster::rpc_codec::TypedClusterError; + use nodedb_types::error::ErrorCode as Ec; + + use crate::control::cluster::data_plane_error_wire::execution_error_to_typed; + + let cases: [ClassCase; 6] = [ + ( + || crate::Error::CalvinParticipantError, + sqlstate::TRANSACTION_ROLLBACK, + Ec::TRANSACTION_ROLLBACK, + ), + ( + || crate::Error::NotInTransactionBlock { + statement: "VACUUM".into(), + }, + sqlstate::ACTIVE_SQL_TRANSACTION, + Ec::ACTIVE_SQL_TRANSACTION, + ), + ( + || crate::Error::CrdtApplyForbiddenInTransaction, + sqlstate::ACTIVE_SQL_TRANSACTION, + Ec::ACTIVE_SQL_TRANSACTION, + ), + ( + || crate::Error::CrossShardInExplicitTransaction, + sqlstate::ACTIVE_SQL_TRANSACTION, + Ec::ACTIVE_SQL_TRANSACTION, + ), + ( + || crate::Error::DependentObjectsExist { + tenant_id: 1, + root_kind: "collection", + root_name: "c".into(), + dependent_count: 1, + dependents: vec![("view".into(), "v".into())], + }, + sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, + Ec::DEPENDENT_OBJECTS_EXIST, + ), + ( + || crate::Error::RoleInUse { + role: "analyst".into(), + dependents: crate::control::security::role_assignment::RoleDependents::ChildRoles( + vec!["junior".into()], + ), + }, + sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, + Ec::DEPENDENT_OBJECTS_EXIST, + ), + ]; + let encoders: [HopEncoder; 2] = [ + ("execution_error_to_typed", execution_error_to_typed), + ("From", TypedClusterError::from), + ]; + for (make, state, code) in cases { + let err = make(); + assert_eq!(error_to_sqlstate(&err).1, state, "{err:?} locally"); + assert_eq!(native_error_fields(&err).code, code, "{err:?} native code"); + for (name, encode) in encoders { + let rebuilt = crate::Error::from(encode(make())); + assert_eq!( + error_to_sqlstate(&rebuilt).1, + state, + "{name}: {err:?} after the hop as {rebuilt:?}" + ); + } + + // Across the SPSC bridge: the Data-Plane code the variant becomes. + let bridged = ErrorCode::from(make()); + let on_bridge = crate::Error::DataPlane(bridged.clone()); + assert_eq!( + error_to_sqlstate(&on_bridge).1, + state, + "{err:?} across the bridge as {bridged:?}" + ); + assert_eq!( + native_error_fields(&on_bridge).code, + code, + "{err:?} native code across the bridge" + ); + + // Across the cluster Data-Plane wire: the code survives verbatim. + let wire = nodedb_cluster::rpc_codec::DataPlaneErrorCode::from(bridged.clone()); + let back = ErrorCode::from(wire); + assert_eq!(back, bridged, "{err:?} across the cluster wire"); + assert_eq!( + error_to_sqlstate(&crate::Error::DataPlane(back)).1, + state, + "{err:?} after the cluster wire" + ); + } +} + +/// Each transaction-state and dependency Data-Plane code crosses the cluster +/// wire verbatim, and renders one SQLSTATE on both sides. +#[test] +fn transaction_and_dependency_codes_roundtrip_the_cluster_wire() { + use nodedb_cluster::rpc_codec::DataPlaneErrorCode; + + let cases = [ + ( + ErrorCode::TransactionRollback { + detail: "participant aborted".into(), + }, + sqlstate::TRANSACTION_ROLLBACK, + ), + ( + ErrorCode::ActiveSqlTransaction { + detail: "VACUUM cannot run inside a transaction block".into(), + }, + sqlstate::ACTIVE_SQL_TRANSACTION, + ), + ( + ErrorCode::DependentObjectsExist { + object: "collection 'c'".into(), + detail: "cannot drop collection 'c': 1 dependent(s) exist (view:v)".into(), + }, + sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, + ), + ]; + for (code, state) in cases { + let local = crate::Error::DataPlane(code.clone()); + assert_eq!(error_to_sqlstate(&local).1, state, "{code:?} locally"); + let back = ErrorCode::from(DataPlaneErrorCode::from(code.clone())); + assert_eq!(back, code, "{code:?} across the cluster wire"); + let remote = crate::Error::DataPlane(back); + assert_eq!(error_to_sqlstate(&remote).1, state, "{code:?} remotely"); + assert_eq!( + native_error_fields(&remote).code, + native_error_fields(&local).code, + "{code:?} native code" + ); + } +} + +/// A DDL error keeps its exact SQLSTATE and code, including a SQLSTATE no +/// named constant covers. +#[test] +fn a_ddl_error_keeps_its_exact_sqlstate_and_code() { + use crate::control::server::shared::ddl::DdlError; + + for state in ["42710", "42P07", sqlstate::INSUFFICIENT_PRIVILEGE, "57014"] { + let ddl = if state == "57014" { + DdlError::from_error(&crate::Error::DeadlineExceeded { + request_id: crate::types::RequestId::new(1), + }) + } else { + DdlError::new(state, "refused") + }; + let expected_code = ddl.code; + let err = crate::Error::from(ddl); + assert_eq!(error_to_sqlstate(&err).1, state, "{err:?}"); + let native = native_error_fields(&err); + assert_eq!(native.sqlstate, state, "{err:?} native SQLSTATE"); + assert_eq!(native.code, expected_code, "{err:?} native code"); + } +} diff --git a/nodedb/src/control/gateway/error_map/http.rs b/nodedb/src/control/gateway/error_map/http.rs index 281e31435..0f9e4316f 100644 --- a/nodedb/src/control/gateway/error_map/http.rs +++ b/nodedb/src/control/gateway/error_map/http.rs @@ -3,52 +3,33 @@ //! HTTP error shape: `(status_code, message)`. use super::gateway_map::GatewayErrorMap; -use super::remote_code::remote_code_to_http_status; +use super::sqlstate_status::sqlstate_to_http_status; use crate::Error; impl GatewayErrorMap { + /// Map a SQLSTATE into an HTTP status, for an error that reaches HTTP as + /// a SQLSTATE, such as a DDL error. [`Self::to_http`] reads the same + /// table, so a DDL error and a gateway error of one class answer one + /// status. + pub fn sqlstate_to_http(sqlstate: &str) -> u16 { + sqlstate_to_http_status(sqlstate) + } + /// Map a gateway error into `(http_status_code, message)` for HTTP. /// - /// Uses standard HTTP status semantics: - /// - 400 Bad Request for client-side errors (bad SQL, not found) - /// - 403 Forbidden for authz errors - /// - 409 Conflict for write-conflict / constraint violations - /// - 503 Service Unavailable for routing/leader errors - /// - 504 Gateway Timeout for deadline exceeded - /// - 500 Internal Server Error as the default fallback + /// The status follows the SQLSTATE pgwire renders for the error, through + /// the one SQLSTATE status table. One error answers one class on pgwire, + /// native and HTTP. Only an `XX000` or `58` class error reads as a 500. + /// A Data-Plane verdict answers with its public message. pub fn to_http(err: &Error) -> (u16, String) { - match err { - Error::NotLeader { leader_addr, .. } => ( - 503, - format!("cluster in leader election; leader hint: {leader_addr}"), - ), - Error::DeadlineExceeded { .. } => (504, err.to_string()), - Error::RetryableSchemaChanged { .. } => (503, err.to_string()), - Error::CollectionNotFound { collection, .. } => { - (404, format!("collection \"{collection}\" does not exist")) - } - Error::RejectedAuthz { .. } => (403, err.to_string()), - Error::BadRequest { detail } => (400, detail.clone()), - Error::PlanError { detail } => (400, detail.clone()), - Error::RejectedConstraint { detail, .. } => (409, detail.clone()), - Error::NoLeader { .. } => (503, err.to_string()), - Error::Serialization { .. } | Error::Codec { .. } => (500, err.to_string()), - Error::Internal { .. } => (500, err.to_string()), - // 501 Not Implemented: a valid op refused because cross-core - // source-shipping is not yet supported (fail-closed safety floor). - Error::CrossCollectionNotColocated { .. } => (501, err.to_string()), - Error::RemoteTyped { code, message } => { - (remote_code_to_http_status(*code), message.clone()) - } - Error::DataPlane(_) => { - let public = crate::error_classify::classify(err); - ( - remote_code_to_http_status(public.code()), - public.message().to_owned(), - ) - } - _ => (500, err.to_string()), - } + let (_severity, state, message) = + crate::control::server::pgwire::types::error_to_sqlstate(err); + let message = if let Error::DataPlane(_) = err { + crate::error_classify::classify(err).message().to_owned() + } else { + message + }; + (sqlstate_to_http_status(state), message) } } @@ -87,6 +68,40 @@ mod tests { assert_eq!(status, 500); } + #[test] + fn http_data_plane_not_found() { + let err = Error::DataPlane(crate::bridge::envelope::ErrorCode::NotFound); + assert_eq!(GatewayErrorMap::to_http(&err).0, 404); + } + + #[test] + fn http_conflict_retry() { + let err = Error::ConflictRetry { + collection: "orders".into(), + document_id: "o1".into(), + }; + assert_eq!(GatewayErrorMap::to_http(&err).0, 409); + } + + #[test] + fn http_backup_key_mismatch_is_invalid_authorization() { + assert_eq!(GatewayErrorMap::to_http(&Error::BackupKeyMismatch).0, 401); + } + + /// A remote rendering of an error keeps the status of the local one. + #[test] + fn http_backup_key_mismatch_keeps_its_status_across_nodes() { + use nodedb_types::error::ErrorCode; + let remote = Error::RemoteTyped { + code: ErrorCode::BACKUP_KEY_MISMATCH, + message: "wrong backup KEK".into(), + }; + assert_eq!( + GatewayErrorMap::to_http(&remote).0, + GatewayErrorMap::to_http(&Error::BackupKeyMismatch).0 + ); + } + #[test] fn to_http_remote_typed_is_wired_to_helper() { use nodedb_types::error::ErrorCode; diff --git a/nodedb/src/control/gateway/error_map/mod.rs b/nodedb/src/control/gateway/error_map/mod.rs index b3f51476c..a704ab5c8 100644 --- a/nodedb/src/control/gateway/error_map/mod.rs +++ b/nodedb/src/control/gateway/error_map/mod.rs @@ -6,12 +6,17 @@ //! One module per protocol surface owns that surface's mapping, so a change //! to its SQLSTATE / HTTP / RESP / native codes is a one-file edit. +#[cfg(test)] +pub(crate) mod class_parity; mod gateway_map; mod http; mod native; mod pgwire; mod remote_code; mod resp; +mod sqlstate_status; +#[cfg(test)] +mod system_dispatch_refusal; #[cfg(test)] mod test_fixtures; diff --git a/nodedb/src/control/gateway/error_map/native.rs b/nodedb/src/control/gateway/error_map/native.rs index f9686cb90..b3eff1b2b 100644 --- a/nodedb/src/control/gateway/error_map/native.rs +++ b/nodedb/src/control/gateway/error_map/native.rs @@ -2,99 +2,71 @@ //! Native-protocol error shape: `(numeric code, message)`. +use nodedb_types::error::ErrorCode; + use super::gateway_map::GatewayErrorMap; use crate::Error; -/// Error code constants (subset matching `nodedb_types` numeric codes). -const CODE_NOT_LEADER: u32 = 10; -const CODE_DEADLINE: u32 = 20; -const CODE_SCHEMA_CHANGED: u32 = 30; -const CODE_NOT_FOUND: u32 = 40; -const CODE_AUTHZ: u32 = 50; -const CODE_BAD_REQUEST: u32 = 60; -const CODE_CONSTRAINT: u32 = 70; -const CODE_INTERNAL: u32 = 99; - impl GatewayErrorMap { /// Map a gateway error into `(code, message)` for the native protocol. /// - /// Error codes are aligned with `nodedb_types::error::ErrorCode` numeric - /// values so native clients can switch on the code without string matching. - pub fn to_native(err: &Error) -> (u32, String) { - match err { - Error::NotLeader { leader_addr, .. } => { - (CODE_NOT_LEADER, format!("not leader; hint: {leader_addr}")) - } - Error::DeadlineExceeded { .. } => (CODE_DEADLINE, err.to_string()), - Error::RetryableSchemaChanged { .. } => (CODE_SCHEMA_CHANGED, err.to_string()), - Error::CollectionNotFound { collection, .. } => ( - CODE_NOT_FOUND, - format!("collection \"{collection}\" not found"), - ), - Error::RejectedAuthz { .. } => (CODE_AUTHZ, err.to_string()), - Error::BadRequest { detail } | Error::PlanError { detail } => { - (CODE_BAD_REQUEST, detail.clone()) - } - Error::RejectedConstraint { detail, .. } => (CODE_CONSTRAINT, detail.clone()), - Error::CrossCollectionNotColocated { .. } => (CODE_BAD_REQUEST, err.to_string()), - Error::RemoteTyped { code, message } => { - use nodedb_types::error::ErrorCode as Ec; - let native_code = match *code { - Ec::DEADLINE_EXCEEDED => CODE_DEADLINE, - Ec::COLLECTION_NOT_FOUND => CODE_NOT_FOUND, - Ec::AUTHORIZATION_DENIED => CODE_AUTHZ, - Ec::BAD_REQUEST | Ec::PLAN_ERROR => CODE_BAD_REQUEST, - Ec::CONSTRAINT_VIOLATION => CODE_CONSTRAINT, - _ => CODE_INTERNAL, - }; - (native_code, message.clone()) - } - _ => (CODE_INTERNAL, err.to_string()), - } + /// The code and message are the ones the native error frame carries, + /// from the one native mapping the listener uses. The code is the stable + /// `nodedb_types::error::ErrorCode`, so a native client switches on it + /// without string matching. + pub fn to_native(err: &Error) -> (ErrorCode, String) { + let fields = crate::control::server::native::dispatch::native_error_fields(err); + (fields.code, fields.message) } } #[cfg(test)] mod tests { - use super::super::test_fixtures::{ - authz, deadline, internal, not_found, not_leader, schema_changed, - }; + use super::super::test_fixtures::{authz, deadline, internal, not_found, not_leader}; use super::*; #[test] fn native_not_leader() { - let (code, msg) = GatewayErrorMap::to_native(¬_leader()); - assert_eq!(code, 10); - assert!(msg.contains("hint:")); + let (code, _) = GatewayErrorMap::to_native(¬_leader()); + assert_eq!(code, ErrorCode::NOT_LEADER); } #[test] fn native_deadline() { let (code, _) = GatewayErrorMap::to_native(&deadline()); - assert_eq!(code, 20); - } - - #[test] - fn native_schema_changed() { - let (code, _) = GatewayErrorMap::to_native(&schema_changed()); - assert_eq!(code, 30); + assert_eq!(code, ErrorCode::DEADLINE_EXCEEDED); } #[test] fn native_not_found() { - let (code, _) = GatewayErrorMap::to_native(¬_found()); - assert_eq!(code, 40); + let (code, msg) = GatewayErrorMap::to_native(¬_found()); + assert_eq!(code, ErrorCode::COLLECTION_NOT_FOUND); + assert!(msg.contains("missing_col")); } #[test] fn native_authz() { let (code, _) = GatewayErrorMap::to_native(&authz()); - assert_eq!(code, 50); + assert_eq!(code, ErrorCode::AUTHORIZATION_DENIED); } #[test] fn native_internal() { let (code, _) = GatewayErrorMap::to_native(&internal()); - assert_eq!(code, 99); + assert_eq!(code, ErrorCode::INTERNAL); + } + + /// The gateway map and the native listener read one mapping, so the code + /// a gateway caller sees is the code the wire frame carries. + #[test] + fn gateway_map_matches_the_wire_frame() { + let err = Error::DataPlane(crate::bridge::envelope::ErrorCode::Unsupported { + detail: "not here".into(), + }); + let frame = crate::control::server::native::dispatch::error_to_native(1, &err); + let payload = frame.error.expect("error frames carry a payload"); + let (code, message) = GatewayErrorMap::to_native(&err); + assert_eq!(code.0, payload.ndb_code); + assert_eq!(message, payload.message); } } diff --git a/nodedb/src/control/gateway/error_map/pgwire.rs b/nodedb/src/control/gateway/error_map/pgwire.rs index 6c69cb6dc..7bfbf67ff 100644 --- a/nodedb/src/control/gateway/error_map/pgwire.rs +++ b/nodedb/src/control/gateway/error_map/pgwire.rs @@ -34,15 +34,15 @@ mod tests { #[test] fn pgwire_deadline() { let (code, _) = GatewayErrorMap::to_pgwire(&deadline()); - assert_eq!(code, sqlstate::QUERY_CANCELED); + assert_eq!(code, sqlstate::QUERY_CANCELED.0); } - /// Both paths answer `INTERNAL_ERROR` from the shared table, with the - /// variant's own message naming the descriptor. + /// Both paths answer `SERIALIZATION_FAILURE`, the SQLSTATE a client + /// retries on, with the variant's own message naming the descriptor. #[test] fn pgwire_schema_changed() { let (code, msg) = GatewayErrorMap::to_pgwire(&schema_changed()); - assert_eq!(code, sqlstate::INTERNAL_ERROR); + assert_eq!(code, sqlstate::SERIALIZATION_FAILURE); assert!(msg.contains("users")); } @@ -106,6 +106,9 @@ mod tests { column: "x".into(), }, Error::DivisionByZero, + Error::DataException { + detail: "vector_distance(): vector dimension mismatch: expected 3, got 2".into(), + }, Error::InvalidLimitValue { clause: "LIMIT", value: "-1".into(), @@ -159,6 +162,12 @@ mod tests { actual: 2, }, Error::BackupKeyMismatch, + Error::DispatchCapacity { + scope: crate::DispatchCapacityScope::QueueFull { + core_id: 0, + capacity: 4, + }, + }, ]; for err in samples { diff --git a/nodedb/src/control/gateway/error_map/remote_code.rs b/nodedb/src/control/gateway/error_map/remote_code.rs index 90e6a5987..e983aa33f 100644 --- a/nodedb/src/control/gateway/error_map/remote_code.rs +++ b/nodedb/src/control/gateway/error_map/remote_code.rs @@ -1,25 +1,10 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Numeric-`ErrorCode` fallbacks shared by the HTTP and RESP surfaces. +//! Numeric-`ErrorCode` fallback for the RESP surface. //! //! A remote peer on a newer build can mint a code this build does not know, -//! so both helpers degrade to the generic shape rather than misclassifying. - -/// Map a numeric `ErrorCode` from a `RemoteTyped` error to an HTTP status, -/// mirroring the local variant arms in `to_http` for the same condition -/// (e.g. `CONSTRAINT_VIOLATION` mirrors `RejectedConstraint`'s 409). -pub(super) fn remote_code_to_http_status(code: nodedb_types::error::ErrorCode) -> u16 { - use nodedb_types::error::ErrorCode as Ec; - match code { - Ec::NOT_LEADER | Ec::NO_LEADER => 503, - Ec::DEADLINE_EXCEEDED => 504, - Ec::COLLECTION_NOT_FOUND => 404, - Ec::AUTHORIZATION_DENIED => 403, - Ec::BAD_REQUEST | Ec::PLAN_ERROR => 400, - Ec::CONSTRAINT_VIOLATION | Ec::WRITE_CONFLICT => 409, - _ => 500, - } -} +//! so the helper degrades to the generic shape rather than misclassifying. +//! HTTP needs no helper: `to_http` renders a remote code through its SQLSTATE. /// Map a numeric `ErrorCode` from a `RemoteTyped` error to a RESP error /// prefix, mirroring the local variant arms in `to_resp`. @@ -31,6 +16,7 @@ pub(super) fn remote_code_to_resp_prefix(code: nodedb_types::error::ErrorCode) - Ec::AUTHORIZATION_DENIED => "NOPERM", Ec::CONSTRAINT_VIOLATION => "CONSTRAINT", Ec::TYPE_MISMATCH => "WRONGTYPE", + Ec::SERVER_OVERLOAD => "BUSY", _ => "ERR", } } @@ -39,29 +25,6 @@ pub(super) fn remote_code_to_resp_prefix(code: nodedb_types::error::ErrorCode) - mod tests { use super::*; - #[test] - fn remote_http_status_maps_known_code() { - use nodedb_types::error::ErrorCode; - // Mirrors `RejectedConstraint`'s 409 in `to_http`. - assert_eq!( - remote_code_to_http_status(ErrorCode::CONSTRAINT_VIOLATION), - 409 - ); - assert_eq!( - remote_code_to_http_status(ErrorCode::AUTHORIZATION_DENIED), - 403 - ); - } - - #[test] - fn remote_http_status_unmapped_code_falls_back_to_500() { - use nodedb_types::error::ErrorCode; - // A code with no explicit arm (e.g. one a newer remote node minted - // that this build doesn't recognize) must degrade to the generic - // 500 fallback, not silently misreport a specific status. - assert_eq!(remote_code_to_http_status(ErrorCode(65000)), 500); - } - #[test] fn remote_resp_prefix_maps_known_code() { use nodedb_types::error::ErrorCode; @@ -82,8 +45,7 @@ mod tests { #[test] fn remote_resp_prefix_unmapped_code_falls_back_to_err() { use nodedb_types::error::ErrorCode; - // Same degrade path as the HTTP fallback: an unrecognized remote code - // must still surface as the generic `ERR` prefix. + // An unrecognized remote code surfaces as the generic `ERR` prefix. assert_eq!(remote_code_to_resp_prefix(ErrorCode(65000)), "ERR"); } } diff --git a/nodedb/src/control/gateway/error_map/resp.rs b/nodedb/src/control/gateway/error_map/resp.rs index 0e30c1a0b..738c16b72 100644 --- a/nodedb/src/control/gateway/error_map/resp.rs +++ b/nodedb/src/control/gateway/error_map/resp.rs @@ -27,9 +27,17 @@ impl GatewayErrorMap { Error::RejectedConstraint { detail, .. } => format!("CONSTRAINT {detail}"), Error::TypeMismatch { detail, .. } => format!("WRONGTYPE {detail}"), Error::RetryableSchemaChanged { .. } => format!("ERR {err}"), + Error::DispatchCapacity { .. } => format!("BUSY {err}"), Error::RemoteTyped { code, message } => { format!("{} {message}", remote_code_to_resp_prefix(*code)) } + // A counter fault answers with the exact reply Redis gives for + // the same condition. + Error::DataPlane(crate::bridge::envelope::ErrorCode::CounterFault { + fault, .. + }) => { + format!("ERR {}", fault.message()) + } Error::DataPlane(_) => { let public = crate::error_classify::classify(err); format!( @@ -38,7 +46,109 @@ impl GatewayErrorMap { public.message() ) } - _ => format!("ERR {err}"), + // Every other variant takes the prefix of its public code, the + // same prefix a remote rendering of it gets. + Error::TxnOverlayMemoryExceeded { .. } + | Error::OffsetRegression { .. } + | Error::ConflictRetry { .. } + | Error::CalvinSerializationConflict + | Error::CalvinParticipantError + | Error::RejectedPrevalidation { .. } + | Error::RetryableRefusal { .. } + | Error::AppendOnlyViolation { .. } + | Error::BalanceViolation { .. } + | Error::MaterializedSumTargetNotFound { .. } + | Error::MaterializedSumResolutionMissing { .. } + | Error::PeriodLocked { .. } + | Error::PeriodLockMisconfigured { .. } + | Error::RetentionViolation { .. } + | Error::LegalHoldActive { .. } + | Error::StateTransitionViolation { .. } + | Error::TransitionCheckViolation { .. } + | Error::TypeGuardViolation { .. } + | Error::InsufficientBalance { .. } + | Error::RateExceeded { .. } + | Error::DocumentNotFound { .. } + | Error::CollectionDeactivated { .. } + | Error::VShardAdmissionCapacityExceeded { .. } + | Error::CrdtAdmissionRetriesExhausted { .. } + | Error::CrdtAdmissionInvalidPlan { .. } + | Error::CrdtAdmissionCallerFence + | Error::CrdtApplyRequiresAdmission + | Error::CrdtApplyForbiddenInTransaction + | Error::NotInTransactionBlock { .. } + | Error::CrdtAdmissionTimeout { .. } + | Error::NoLeader { .. } + | Error::FanOutExceeded { .. } + | Error::CrossCollectionNotColocated { .. } + | Error::SourceFrozen { .. } + | Error::CloneWriteRequiresMaterialize { .. } + | Error::BackupTenantMismatch { .. } + | Error::BackupKeyMismatch + | Error::QuotaOvercommit { .. } + | Error::FeatureNotSupported { .. } + | Error::UndefinedFunction { .. } + | Error::UndefinedObject { .. } + | Error::ObjectNotInPrerequisiteState { .. } + | Error::UndefinedColumn { .. } + | Error::AmbiguousColumn { .. } + | Error::UnknownStrictField { .. } + | Error::DivisionByZero + | Error::DataException { .. } + | Error::InvalidLimitValue { .. } + | Error::RetryableLeaderChange { .. } + | Error::GroupQuorumUnavailable { .. } + | Error::GroupMarksUnavailable { .. } + | Error::MetadataLeaderUnavailable + | Error::AuthorizationStateBehind { .. } + | Error::ExecutionLimitExceeded { .. } + | Error::LimitExceeded { .. } + | Error::Wal(_) + | Error::Dispatch { .. } + | Error::Storage { .. } + | Error::ColdStorage { .. } + | Error::Serialization { .. } + | Error::Codec { .. } + | Error::SegmentCorrupted { .. } + | Error::MemoryExhausted { .. } + | Error::Backpressure { .. } + | Error::Crdt(_) + | Error::Io(_) + | Error::Config { .. } + | Error::Encryption { .. } + | Error::Bridge { .. } + | Error::VersionCompat { .. } + | Error::Internal { .. } + | Error::Shaping(_) + | Error::Ddl(_) + | Error::DescriptorVersionAnomaly { .. } + | Error::CollectionPurgeRowMissing { .. } + | Error::CatalogIntegrityViolation { .. } + | Error::Promql(_) + | Error::DependentObjectsExist { .. } + | Error::RoleInUse { .. } + | Error::CascadeCycle { .. } + | Error::CrossShardInExplicitTransaction + | Error::SequencerUnavailable + | Error::SessionCapExceeded { .. } + | Error::SessionIdleTimeout + | Error::SessionTokenExpired + | Error::SessionKilledByAdmin + | Error::SessionUserDropped + | Error::OidcProviderTenantUnbound + | Error::OidcProviderTenantUnavailable { .. } + | Error::ExternalRoleUndefined { .. } + | Error::OidcNoDefaultDatabase { .. } + | Error::TenantVectorDimExceeded { .. } + | Error::TenantGraphDepthExceeded { .. } + | Error::RoleInheritanceCycle { .. } + | Error::RoleInheritanceDepthExceeded { .. } + | Error::OllpExhausted { .. } + | Error::MirrorReadOnly { .. } + | Error::StaleReadNotLeader { .. } => format!( + "{} {err}", + remote_code_to_resp_prefix(crate::error_classify::classify(err).code()) + ), } } } @@ -89,6 +199,33 @@ mod tests { assert!(msg.starts_with("WRONGTYPE "), "{msg}"); } + #[test] + fn resp_counter_faults_use_the_redis_error_text() { + use crate::bridge::envelope::{CounterFault, ErrorCode}; + let cases = [ + ( + CounterFault::NotAnInteger, + "ERR value is not an integer or out of range", + ), + (CounterFault::NotAFloat, "ERR value is not a valid float"), + ( + CounterFault::IntegerOverflow, + "ERR increment or decrement would overflow", + ), + ( + CounterFault::NonFinite, + "ERR increment would produce NaN or Infinity", + ), + ]; + for (fault, expected) in cases { + let err = Error::DataPlane(ErrorCode::CounterFault { + collection: "counters".into(), + fault, + }); + assert_eq!(GatewayErrorMap::to_resp(&err), expected); + } + } + #[test] fn to_resp_remote_typed_is_wired_to_helper() { use nodedb_types::error::ErrorCode; @@ -99,4 +236,14 @@ mod tests { let msg = GatewayErrorMap::to_resp(&err); assert_eq!(msg, "CONSTRAINT unique key clash"); } + + /// A variant with no arm of its own takes the prefix of its public code. + #[test] + fn resp_prefix_follows_the_public_code() { + let err = Error::CrdtAdmissionTimeout { + vshard_id: crate::types::VShardId::new(1), + timeout_ms: 10, + }; + assert!(GatewayErrorMap::to_resp(&err).starts_with("TIMEOUT ")); + } } diff --git a/nodedb/src/control/gateway/error_map/sqlstate_status.rs b/nodedb/src/control/gateway/error_map/sqlstate_status.rs new file mode 100644 index 000000000..258c1c188 --- /dev/null +++ b/nodedb/src/control/gateway/error_map/sqlstate_status.rs @@ -0,0 +1,115 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! SQLSTATE to HTTP status, for an error that reaches HTTP as a SQLSTATE. +//! +//! A DDL error carries a SQLSTATE and a numeric code. Several DDL SQLSTATEs +//! share one code, so the status follows the SQLSTATE. `to_http` reads this +//! table for every gateway error, so it is the one status table. + +use nodedb_types::error::sqlstate; + +/// Map a SQLSTATE to an HTTP status. +/// +/// - 5xx for an internal or system error, and for an unavailable server +/// - 4xx per class for a request the client must change +/// - 409 for a conflict or a constraint violation +/// - 429 for a rate limit +/// - 501 for an unsupported feature +pub(super) fn sqlstate_to_http_status(state: &str) -> u16 { + const QUERY_CANCELED: &str = sqlstate::QUERY_CANCELED.0; + match state { + sqlstate::INSUFFICIENT_PRIVILEGE => 403, + sqlstate::UNDEFINED_TABLE => 404, + // The object already exists: the request conflicts with the catalog. + "42710" | "42P07" | "42723" => 409, + sqlstate::TOO_MANY_CONNECTIONS => 429, + // A configured quota the request exceeds. A retry fails the same way. + sqlstate::CONFIGURATION_LIMIT_EXCEEDED => 400, + // No leader or no quorum answered. A retry succeeds later. + sqlstate::LOCK_NOT_AVAILABLE => 503, + QUERY_CANCELED => 504, + _ => class_status(state), + } +} + +/// The status of a SQLSTATE class. +fn class_status(state: &str) -> u16 { + match state.get(..2).unwrap_or(state) { + // No data, or an unknown database. + "02" | "3D" => 404, + "0A" => 501, + // Connection failure, insufficient resources, operator intervention. + "08" | "53" | "57" => 503, + // Data exception, invalid transaction state, syntax or access rule, + // program limit. + "22" | "25" | "42" | "54" => 400, + // Constraint violation, dependent objects, transaction rollback, + // object not in prerequisite state. + "23" | "2B" | "40" | "55" => 409, + "28" => 401, + // `XX` internal error, `58` system error, and any other class. + _ => 500, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn internal_errors_are_server_faults() { + assert_eq!(sqlstate_to_http_status(sqlstate::INTERNAL_ERROR), 500); + assert_eq!(sqlstate_to_http_status(sqlstate::IO_ERROR), 500); + } + + #[test] + fn feature_not_supported_is_not_implemented() { + assert_eq!( + sqlstate_to_http_status(sqlstate::FEATURE_NOT_SUPPORTED), + 501 + ); + } + + #[test] + fn conflicts_and_constraints_are_409() { + for state in [ + sqlstate::UNIQUE_VIOLATION, + sqlstate::NOT_NULL_VIOLATION, + sqlstate::CHECK_VIOLATION, + sqlstate::SERIALIZATION_FAILURE, + sqlstate::OBJECT_NOT_IN_PREREQUISITE_STATE, + "42P07", + "2BP01", + ] { + assert_eq!(sqlstate_to_http_status(state), 409, "{state}"); + } + } + + #[test] + fn rate_limits_are_429() { + assert_eq!(sqlstate_to_http_status(sqlstate::TOO_MANY_CONNECTIONS), 429); + } + + #[test] + fn client_errors_take_their_class_status() { + assert_eq!(sqlstate_to_http_status(sqlstate::SYNTAX_ERROR), 400); + assert_eq!(sqlstate_to_http_status(sqlstate::DATA_EXCEPTION), 400); + assert_eq!( + sqlstate_to_http_status(sqlstate::INSUFFICIENT_PRIVILEGE), + 403 + ); + assert_eq!( + sqlstate_to_http_status(sqlstate::INVALID_AUTHORIZATION), + 401 + ); + assert_eq!(sqlstate_to_http_status(sqlstate::UNDEFINED_TABLE), 404); + assert_eq!(sqlstate_to_http_status(sqlstate::INVALID_CATALOG_NAME), 404); + } + + #[test] + fn unavailability_is_503_and_a_deadline_is_504() { + assert_eq!(sqlstate_to_http_status(sqlstate::SERVER_OVERLOAD), 503); + assert_eq!(sqlstate_to_http_status(sqlstate::LOCK_NOT_AVAILABLE), 503); + assert_eq!(sqlstate_to_http_status(sqlstate::QUERY_CANCELED.0), 504); + } +} diff --git a/nodedb/src/control/gateway/error_map/system_dispatch_refusal.rs b/nodedb/src/control/gateway/error_map/system_dispatch_refusal.rs new file mode 100644 index 000000000..f63dcf8b0 --- /dev/null +++ b/nodedb/src/control/gateway/error_map/system_dispatch_refusal.rs @@ -0,0 +1,187 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A typed Data-Plane refusal keeps its SQLSTATE through the system door. +//! +//! `CREATE VECTOR INDEX` registers its parameters through `apply_in_engine`, +//! which dispatches `VectorOp::SetParams` through `dispatch_system`. A core +//! whose index already holds vectors refuses that plan with +//! `ErrorCode::Unsupported`. A fake core gives that exact refusal here, so the +//! test runs the real dispatch, response routing and DDL error mapping with no +//! real Data-Plane core. + +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use nodedb_physical::physical_plan::VectorOp; +use nodedb_types::error::sqlstate; + +use crate::bridge::dispatch::{BridgeResponse, CoreChannelDataSide, Dispatcher}; +use crate::bridge::envelope::{ErrorCode, Payload, PhysicalPlan, Response, Status}; +use crate::control::server::shared::ddl::engine_apply::apply_in_engine; +use crate::control::server::shared::ddl::sync_dispatch::{ + SystemReason, SystemTask, dispatch_system, +}; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, Lsn, TenantId}; +use crate::wal::WalManager; + +const COLLECTION: &str = "vectors"; + +/// The refusal a core gives `SetParams` on an index that holds vectors. +fn materialized_refusal() -> ErrorCode { + ErrorCode::Unsupported { + detail: "changing vector index params after the index holds vectors is not \ + supported; drop and recreate the collection" + .into(), + } +} + +fn set_params_plan() -> PhysicalPlan { + PhysicalPlan::Vector(VectorOp::SetParams { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, COLLECTION), + field_name: "emb".into(), + dim: 3, + m: 16, + ef_construction: 200, + metric: "cosine".into(), + index_type: "hnsw".into(), + pq_m: 0, + ivf_cells: 0, + ivf_nprobe: 0, + }) +} + +/// State plus the fake core's data side. The caller keeps the `TempDir` +/// alive for as long as `state` is in use. +fn fixture() -> (Arc, CoreChannelDataSide, tempfile::TempDir) { + let dir = tempfile::tempdir().expect("create test directory"); + let wal = Arc::new( + WalManager::open_for_testing(&dir.path().join("test.wal")).expect("open test WAL"), + ); + let (dispatcher, mut sides) = Dispatcher::new(1, 64); + let side = sides.pop().expect("one data side"); + let state = SharedState::new(dispatcher, wal).expect("construct shared state"); + (state, side, dir) +} + +/// Fake core: pops the one dispatched request and answers it with an error +/// status carrying `code`. +async fn refuse_once( + mut side: CoreChannelDataSide, + state: Arc, + code: Option, +) { + let deadline = Instant::now() + Duration::from_secs(5); + let mut handled = false; + while !handled && Instant::now() < deadline { + if let Ok(request) = side.request_rx.try_pop() { + side.response_tx + .try_push(BridgeResponse { + inner: Response { + request_id: request.inner.request_id, + status: Status::Error, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: code.clone().map(Box::new), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + }, + }) + .expect("fake core response queue has capacity"); + handled = true; + } + state.poll_and_route_responses(); + tokio::task::yield_now().await; + } + assert!(handled, "fake core received the dispatched request"); + state.poll_and_route_responses(); +} + +/// The system door returns the refusal's own code, never `Internal`. +#[tokio::test] +async fn dispatch_system_keeps_the_refusal_code() { + let (state, side, _dir) = fixture(); + let responder = tokio::spawn(refuse_once( + side, + Arc::clone(&state), + Some(materialized_refusal()), + )); + let result = dispatch_system( + &state, + SystemTask::new( + SystemReason::DdlApply, + TenantId::new(1), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, COLLECTION), + set_params_plan(), + ), + Duration::from_secs(5), + ) + .await; + responder.await.expect("responder completes"); + + match result { + Err(crate::Error::DataPlane(code)) => assert_eq!(code, materialized_refusal()), + other => panic!("expected the typed Data-Plane refusal, got {other:?}"), + } +} + +/// The DDL statement answers the refusal's SQLSTATE (`0A000`) and class, with +/// the statement context before the message. +#[tokio::test] +async fn create_vector_index_refusal_keeps_its_sqlstate() { + let (state, side, _dir) = fixture(); + let responder = tokio::spawn(refuse_once( + side, + Arc::clone(&state), + Some(materialized_refusal()), + )); + let result = apply_in_engine( + &state, + TenantId::new(1), + DatabaseId::DEFAULT, + COLLECTION, + set_params_plan(), + "CREATE VECTOR INDEX", + ) + .await; + responder.await.expect("responder completes"); + + let err = result.expect_err("the refused SetParams must fail the statement"); + assert_eq!(err.sqlstate, sqlstate::FEATURE_NOT_SUPPORTED, "{err:?}"); + assert_ne!(err.sqlstate, sqlstate::INTERNAL_ERROR); + assert_eq!(err.code, nodedb_types::error::ErrorCode::SQL_NOT_ENABLED); + assert!( + err.message.starts_with("CREATE VECTOR INDEX: "), + "context leads the message: {}", + err.message + ); + assert!( + err.message.contains("holds vectors"), + "the refusal detail survives: {}", + err.message + ); +} + +/// A refusal with no code has no class of its own, so it is the one case +/// that stays internal. +#[tokio::test] +async fn refusal_without_a_code_is_internal() { + let (state, side, _dir) = fixture(); + let responder = tokio::spawn(refuse_once(side, Arc::clone(&state), None)); + let result = apply_in_engine( + &state, + TenantId::new(1), + DatabaseId::DEFAULT, + COLLECTION, + set_params_plan(), + "CREATE VECTOR INDEX", + ) + .await; + responder.await.expect("responder completes"); + + let err = result.expect_err("an uncoded refusal must fail the statement"); + assert_eq!(err.sqlstate, sqlstate::INTERNAL_ERROR, "{err:?}"); +} diff --git a/nodedb/src/control/gateway/mod.rs b/nodedb/src/control/gateway/mod.rs index 50058b703..da868d892 100644 --- a/nodedb/src/control/gateway/mod.rs +++ b/nodedb/src/control/gateway/mod.rs @@ -10,6 +10,7 @@ pub mod fuser; pub mod invalidation; pub mod key_extractor; pub mod lowered_plan; +pub mod outcome; pub mod plan_cache; pub mod retry; pub mod route; diff --git a/nodedb/src/control/gateway/outcome.rs b/nodedb/src/control/gateway/outcome.rs new file mode 100644 index 000000000..767b88ef2 --- /dev/null +++ b/nodedb/src/control/gateway/outcome.rs @@ -0,0 +1,154 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Gateway execution results, and their Data-Plane response shape. +//! +//! A transport that renders a Data-Plane `Response` gets the same shape from +//! the gateway that local SPSC dispatch gives it. The shape keeps the error +//! status and code of a `NotFound` verdict, the read watermark, and the +//! read-version LSN. + +use crate::Error; +use crate::bridge::envelope::{ErrorCode, Payload, Response, Status}; +use crate::control::server::shared::clone_write::CloneCheckedTask; +use crate::types::{Lsn, RequestId, VShardId}; + +use super::core::{Gateway, QueryContext, authorized_plan_for_context}; + +/// Everything one gateway execution observed across its routes. +pub struct GatewayOutcome { + /// One payload, fused when several routes answered. + pub payloads: Vec>, + /// One `(vshard, watermark_lsn)` per participating shard. + pub shard_watermarks: Vec<(VShardId, Lsn)>, + /// Max-folded per-collection read-version LSN. + pub read_version_lsn: Lsn, + /// A single-route plan's owning core refused it with `ErrorCode::NotFound`. + /// + /// Always `false` for a fan-out: there it means a shard holds no slice. + pub not_found: bool, +} + +impl GatewayOutcome { + /// Payloads, per-shard watermarks, and read-version LSN. + pub fn into_parts(self) -> (Vec>, Vec<(VShardId, Lsn)>, Lsn) { + (self.payloads, self.shard_watermarks, self.read_version_lsn) + } + + /// The response local SPSC dispatch returns for the same task. + pub fn into_response(self) -> Response { + let watermark_lsn = self + .shard_watermarks + .iter() + .map(|(_, lsn)| *lsn) + .max() + .unwrap_or(Lsn::ZERO); + if self.not_found { + return not_found_response(watermark_lsn, self.read_version_lsn); + } + let payload = self + .payloads + .into_iter() + .next() + .map(Payload::from_vec) + .unwrap_or_else(Payload::empty); + response( + Status::Ok, + None, + payload, + watermark_lsn, + self.read_version_lsn, + ) + } +} + +impl Gateway { + /// Execute one authorized task and return its Data-Plane response shape. + /// + /// A `NotFound` verdict returns as an error-status response, the same as + /// local SPSC dispatch returns it. The caller records a phantom read from + /// it and renders the verdict. Every other error returns as its typed + /// `Err`. + pub async fn execute_response( + &self, + ctx: &QueryContext, + checked: CloneCheckedTask, + ) -> Result { + let plan = authorized_plan_for_context(ctx, checked)?; + match self.execute_plan_outcome(ctx, plan).await { + Ok(outcome) => Ok(outcome.into_response()), + // A remote leaseholder returns its `NotFound` verdict as a typed + // error. It gets the same shape as a local one. + Err(Error::DataPlane(ErrorCode::NotFound)) => { + Ok(not_found_response(Lsn::ZERO, Lsn::ZERO)) + } + Err(error) => Err(error), + } + } +} + +fn not_found_response(watermark_lsn: Lsn, read_version_lsn: Lsn) -> Response { + response( + Status::Error, + Some(ErrorCode::NotFound), + Payload::empty(), + watermark_lsn, + read_version_lsn, + ) +} + +fn response( + status: Status, + error_code: Option, + payload: Payload, + watermark_lsn: Lsn, + read_version_lsn: Lsn, +) -> Response { + Response { + request_id: RequestId::new(0), + status, + attempt: 0, + partial: false, + payload, + watermark_lsn, + error_code: error_code.map(Box::new), + read_set_valid: None, + read_version_lsn, + write_set: Vec::new(), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn outcome(not_found: bool) -> GatewayOutcome { + GatewayOutcome { + payloads: vec![vec![0x90]], + shard_watermarks: vec![ + (VShardId::new(3), Lsn::new(7)), + (VShardId::new(4), Lsn::new(9)), + ], + read_version_lsn: Lsn::new(5), + not_found, + } + } + + #[test] + fn a_not_found_verdict_keeps_its_error_status_and_code() { + let resp = outcome(true).into_response(); + assert_eq!(resp.status, Status::Error); + assert_eq!(resp.error_code.as_deref(), Some(&ErrorCode::NotFound)); + assert_eq!(resp.watermark_lsn, Lsn::new(9)); + assert_eq!(resp.read_version_lsn, Lsn::new(5)); + } + + #[test] + fn a_success_keeps_its_payload_and_lsns() { + let resp = outcome(false).into_response(); + assert_eq!(resp.status, Status::Ok); + assert!(resp.error_code.is_none()); + assert_eq!(resp.payload.to_vec(), vec![0x90u8]); + assert_eq!(resp.watermark_lsn, Lsn::new(9)); + assert_eq!(resp.read_version_lsn, Lsn::new(5)); + } +} diff --git a/nodedb/src/control/gateway/router.rs b/nodedb/src/control/gateway/router.rs index 59050b599..d1bf99e84 100644 --- a/nodedb/src/control/gateway/router.rs +++ b/nodedb/src/control/gateway/router.rs @@ -9,7 +9,8 @@ //! //! 1. Consult the `strategy_fn` closure (backed by the catalog) for the plan's //! primary collection to determine its [`PartitionStrategy`]: -//! - `CollectionHomed` → one vShard derived from [`vshard_for_collection`]. +//! - `CollectionHomed` → one vShard derived from [`vshard_for_collection`] +//! over the collection's canonical key. //! - `KeyPartitioned` → one vShard per distinct key via [`VShardId::from_key`] //! (deduplicated; multiple keys mapping to the same vShard share one route). //! 2. Look up the Raft group leader for each vShard in the routing table. @@ -27,7 +28,7 @@ use nodedb_cluster::routing::{RoutingTable, vshard_for_collection}; use nodedb_types::PartitionStrategy; -use nodedb_types::id::{DatabaseId, VShardId}; +use nodedb_types::id::{CollectionKey, DatabaseId, VShardId}; use nodedb_physical::physical_plan::PhysicalPlan; @@ -63,29 +64,17 @@ pub fn route_plan( strategy_fn: impl Fn(&str) -> PartitionStrategy, extractor: &dyn KeyExtractor, ) -> Result> { - // Commit-time meta-ops (ResolveTxn / TransactionBatch) carry no collection - // name, so their vShard cannot be derived here — the primary_vshard - // fallback would silently send them to vShard 0 and durably apply the - // commit batch on the wrong core. They are dispatched with the - // task's pre-classified `vshard_id` (see `dispatch_single_shard`), never - // through the gateway. - { - use nodedb_physical::physical_plan::MetaOp; - if matches!( - &plan, - PhysicalPlan::Meta(MetaOp::ResolveTxn { .. } | MetaOp::TransactionBatch { .. }) - ) { - return Err(crate::Error::Internal { - detail: "commit meta-op cannot be routed by the gateway; \ - dispatch it with the task's explicit vshard_id" - .to_owned(), - }); - } + if is_task_vshard_scoped(&plan) { + return Err(crate::Error::Internal { + detail: "transaction meta-op cannot be routed by the gateway; \ + dispatch it with the task's explicit vshard_id" + .to_owned(), + }); } // In single-node mode every plan runs locally. let Some(routing) = routing else { - let vshard_id = primary_vshard(&plan, database_id); + let vshard_id = primary_vshard(&plan, database_id)?; return Ok(vec![TaskRoute { plan, decision: RouteDecision::Local, @@ -168,10 +157,12 @@ fn route_single_collection( match strategy { PartitionStrategy::CollectionHomed => { // Byte-identical to the original primary_vshard / resolve_decision path. - let vshard_id = primary_name - .as_deref() - .map(|name| vshard_for_collection(database_id, name)) - .unwrap_or(0); + let vshard_id = match primary_name.as_deref() { + Some(name) => { + vshard_for_collection(CollectionKey::from_qualified_str(database_id, name)?) + } + None => 0, + }; let decision = resolve_decision(vshard_id, local_node_id, Some(routing), None); Ok(vec![TaskRoute { plan, @@ -283,15 +274,44 @@ fn route_broadcast( routes } -/// Determine the primary vShard for a plan by hashing the first collection name. +/// Whether `plan` is a transaction meta-op that runs on the core of the +/// task's own `vshard_id`. /// -/// Falls back to vShard 0 for plans that have no named collection (Meta ops). -fn primary_vshard(plan: &PhysicalPlan, database_id: DatabaseId) -> u32 { - touched_collections(plan) - .into_iter() - .next() - .map(|name| vshard_for_collection(database_id, &name)) - .unwrap_or(0) +/// These ops name no collection, so the router cannot derive their vShard. The +/// `primary_vshard` fallback would send them to vShard 0: a staged write would +/// land in an overlay the commit never reads, and a commit would apply on the +/// wrong core. Callers dispatch them with the task's `vshard_id`, never +/// through the gateway. +pub fn is_task_vshard_scoped(plan: &PhysicalPlan) -> bool { + use nodedb_physical::physical_plan::MetaOp; + matches!( + plan, + PhysicalPlan::Meta( + MetaOp::StageWrite { .. } + | MetaOp::MarkSavepoint { .. } + | MetaOp::RollbackToSavepoint { .. } + | MetaOp::DropTxnOverlay { .. } + | MetaOp::ResolveTxn { .. } + | MetaOp::TransactionBatch { .. } + | MetaOp::ApplyTransactionRedo { .. } + ) + ) +} + +/// Determine the primary vShard for a plan from its first collection. +/// +/// The plan carries the database-qualified name. It is de-qualified into the +/// canonical key before hashing, so the route matches the vShard every other +/// path homes the collection to. Falls back to vShard 0 for plans that have +/// no named collection (Meta ops). +fn primary_vshard(plan: &PhysicalPlan, database_id: DatabaseId) -> Result { + match touched_collections(plan).into_iter().next() { + Some(name) => Ok(vshard_for_collection(CollectionKey::from_qualified_str( + database_id, + &name, + )?)), + None => Ok(0), + } } #[cfg(test)] @@ -342,6 +362,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let routes = route_plan( plan, @@ -449,7 +470,7 @@ mod tests { // would (the collection's owner). assert_eq!( routes[0].vshard_id, - vshard_for_collection(DatabaseId::DEFAULT, "events") + vshard_for_collection(CollectionKey::from_bare(DatabaseId::DEFAULT, "events")) ); } @@ -457,22 +478,39 @@ mod tests { fn find_collection_for_vshard(target: u32) -> String { for i in 0u64.. { let name = format!("col_{i}"); - if vshard_for_collection(DatabaseId::DEFAULT, &name) == target { + if vshard_for_collection(CollectionKey::from_bare(DatabaseId::DEFAULT, &name)) == target + { return name; } } unreachable!() } - /// Commit-time meta-ops carry no collection name, so the router cannot - /// derive their vShard — silently falling back to vShard 0 durably applies - /// the commit batch on the wrong core. They must be rejected here; - /// callers dispatch them with the task's pre-classified `vshard_id`. + /// Transaction meta-ops carry no collection name, so the router cannot + /// derive their vShard. The vShard 0 fallback stages a write in an overlay + /// the commit never reads, or applies a commit on the wrong core. #[test] - fn commit_meta_ops_are_rejected() { + fn transaction_meta_ops_are_rejected() { use nodedb_physical::physical_plan::MetaOp; + let txn_id = nodedb_types::id::TxnId::new(7); for plan in [ + PhysicalPlan::Meta(MetaOp::StageWrite { + plan: Box::new(PhysicalPlan::Kv(KvOp::Get { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "users"), + key: vec![], + rls_filters: vec![], + surrogate_ceiling: None, + })), + }), + PhysicalPlan::Meta(MetaOp::MarkSavepoint { txn_id }), + PhysicalPlan::Meta(MetaOp::RollbackToSavepoint { + txn_id, + value_marker: 0, + graph_marker: 0, + array_marker: 0, + }), + PhysicalPlan::Meta(MetaOp::DropTxnOverlay { txn_id }), PhysicalPlan::Meta(MetaOp::TransactionBatch { plans: vec![], txn_id: None, @@ -481,6 +519,12 @@ mod tests { txn_id: nodedb_types::id::TxnId::new(7), plans: vec![], }), + PhysicalPlan::Meta(MetaOp::ApplyTransactionRedo { + redo: vec![], + collections: vec![], + sum_targets: vec![], + origin: nodedb_physical::physical_plan::RedoOrigin::Commit, + }), ] { for table in [None, Some(single_node_table())] { let result = route_plan( @@ -491,9 +535,10 @@ mod tests { |_| PartitionStrategy::CollectionHomed, &crate::control::gateway::UnwiredKeyExtractor, ); + assert!(is_task_vshard_scoped(&plan), "{plan:?}"); assert!( result.is_err(), - "commit meta-op must not be routable via the gateway: {plan:?}" + "transaction meta-op must not be routable via the gateway: {plan:?}" ); } } diff --git a/nodedb/src/control/gateway/sql_execute.rs b/nodedb/src/control/gateway/sql_execute.rs index fc063cf2b..c3f06d3c9 100644 --- a/nodedb/src/control/gateway/sql_execute.rs +++ b/nodedb/src/control/gateway/sql_execute.rs @@ -54,11 +54,10 @@ impl Gateway { // Read before reverify runs, not derived per-name inside it: both // pseudo-entries share one live tenant snapshot the same way the // real collection entries share one catalog snapshot. - let permission_tree_version = shared - .permission_cache - .read() - .await - .tenant_version(tenant_id); + let permission_tree_version = + crate::control::security::auth_fence::permission_view(&shared, ctx.tenant_id) + .await? + .tenant_version(tenant_id); let rls_version = shared.rls.tenant_version(tenant_id); let ptree_key = permission_tree_version_key(tenant_id); let rls_key = rls_version_key(tenant_id); @@ -89,7 +88,7 @@ impl Gateway { return self .execute_with_version_set(ctx, plan, stored_vs) .await - .map(|(payloads, _watermarks, _read_version)| payloads); + .map(|outcome| outcome.payloads); } } } @@ -122,6 +121,6 @@ impl Gateway { let plan = authorized_plan_for_context(ctx, checked)?; self.execute_with_version_set(ctx, plan, actual_vs) .await - .map(|(payloads, _watermarks, _read_version)| payloads) + .map(|outcome| outcome.payloads) } } diff --git a/nodedb/src/control/gateway/version_set/plan_keys.rs b/nodedb/src/control/gateway/version_set/plan_keys.rs index 9bd354e41..1a61717db 100644 --- a/nodedb/src/control/gateway/version_set/plan_keys.rs +++ b/nodedb/src/control/gateway/version_set/plan_keys.rs @@ -37,7 +37,8 @@ fn kv_touched_collections(op: &nodedb_physical::physical_plan::KvOp, out: &mut V | RegisterSortedIndex { collection, .. } | PredicateUpdate { collection, .. } | PredicateDelete { collection, .. } - | MaterializeScan { collection, .. } => out.push(collection.as_str().to_owned()), + | MaterializeScan { collection, .. } + | SortedIndexTxnRead { collection, .. } => out.push(collection.as_str().to_owned()), // TransferItem touches two collections. TransferItem { diff --git a/nodedb/src/control/insert_select/copy_rows.rs b/nodedb/src/control/insert_select/copy_rows.rs index ddfcf1b71..8f61f34e7 100644 --- a/nodedb/src/control/insert_select/copy_rows.rs +++ b/nodedb/src/control/insert_select/copy_rows.rs @@ -119,9 +119,8 @@ pub(crate) fn assign_page_rows( }; let surrogate = assign_target_surrogate( state, - database_id, + nodedb_types::CollectionKey::from_qualified_str(database_id, target_collection)?, tenant_id, - target_collection, &spec.target_pk, &value, )?; diff --git a/nodedb/src/control/insert_select/expand_staged.rs b/nodedb/src/control/insert_select/expand_staged.rs index 583c42011..ff8ce7398 100644 --- a/nodedb/src/control/insert_select/expand_staged.rs +++ b/nodedb/src/control/insert_select/expand_staged.rs @@ -21,7 +21,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::insert_select::copy_rows::{assign_page_rows, resolve_copy_spec}; use crate::control::maintenance::clone_materializer::scan_source_page; use crate::control::state::SharedState; -use crate::types::{TxnId, VShardId}; +use crate::types::TxnId; use nodedb_physical::physical_plan::DocumentOp; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; @@ -70,8 +70,11 @@ pub(crate) async fn resolve_and_emit_insert_select_ops( // Recompute the target vShard (rather than reusing the staged task's) // to keep dispatch classification honest, as the MERGE expander does. - let vshard_id = - VShardId::from_collection_in_database(task.database_id, target_collection.as_str()); + let vshard_id = nodedb_types::CollectionKey::from_qualified_str( + task.database_id, + target_collection.as_str(), + )? + .vshard(); // Resolve materialized-sum targets: these ops stage directly, bypassing // statement-level resolution, so without this a bound target collection diff --git a/nodedb/src/control/lease/renewal.rs b/nodedb/src/control/lease/renewal.rs index 83e6ea12b..7274b6872 100644 --- a/nodedb/src/control/lease/renewal.rs +++ b/nodedb/src/control/lease/renewal.rs @@ -38,7 +38,7 @@ use nodedb_cluster::LoopMetrics; use nodedb_types::config::tuning::ClusterTransportTuning; use tokio::sync::watch; use tokio::task::JoinHandle; -use tracing::{debug, info, warn}; +use tracing::{debug, error, info}; use crate::control::state::SharedState; @@ -156,8 +156,9 @@ impl LeaseRenewalLoop { } /// One iteration: snapshot the near-expiry lease set under a - /// short read lock, then re-acquire each one. Errors are - /// logged at warn — the next tick retries automatically. + /// short read lock, then re-acquire each one. A failed step is + /// logged at error and recorded as a diagnostic capture. The + /// next tick retries it. /// /// **Why we use wall-clock nanoseconds, not `hlc_clock.peek()`**: /// `peek` returns the last HLC the clock observed, which may @@ -193,21 +194,38 @@ impl LeaseRenewalLoop { version, self.config.full_duration, ) { - warn!( + error!( descriptor = ?id, version, error = %e, - "descriptor lease renewal: re-acquire failed" + "descriptor lease renewal: re-acquire failed; the lease \ + expires unless a later tick refreshes it" + ); + crate::diag::descriptor_lease_not_renewed( + &e, + "renew", + &id, + version, + shared.node_id, ); self.loop_metrics.record_error("renew"); } } None => { if let Err(e) = super::release::release_leases(&shared, vec![id.clone()]) { - warn!( + error!( descriptor = ?id, + held_version, error = %e, - "descriptor lease renewal: release after drop failed" + "descriptor lease renewal: release after drop failed; the \ + lease blocks DDL drains on its descriptor until it expires" + ); + crate::diag::descriptor_lease_not_renewed( + &e, + "release", + &id, + held_version, + shared.node_id, ); self.loop_metrics.record_error("release"); } diff --git a/nodedb/src/control/server/dispatch_utils/collect.rs b/nodedb/src/control/local_dispatch/collect.rs similarity index 92% rename from nodedb/src/control/server/dispatch_utils/collect.rs rename to nodedb/src/control/local_dispatch/collect.rs index 561ec422b..67ff384eb 100644 --- a/nodedb/src/control/server/dispatch_utils/collect.rs +++ b/nodedb/src/control/local_dispatch/collect.rs @@ -21,7 +21,7 @@ pub(crate) enum DispatchCollectError { /// concatenated payload) or an error if the channel closed without a /// final chunk or if the accumulated payload would exceed the ceiling. pub(crate) async fn collect_bounded_response( - rx: &mut tokio::sync::mpsc::Receiver, + rx: &mut crate::control::ResponseReceiver, max_result_bytes: usize, ) -> Result { // Each streamed chunk is its OWN msgpack array (`encode_raw_document_rows` @@ -108,7 +108,7 @@ pub(crate) struct DeadlineCollect<'a> { /// the symptom and hand the client a generic internal error for its own /// timeout. pub(crate) async fn collect_under_deadline( - rx: &mut tokio::sync::mpsc::Receiver, + rx: &mut crate::control::ResponseReceiver, params: DeadlineCollect<'_>, ) -> crate::Result { let DeadlineCollect { @@ -223,7 +223,8 @@ mod collect_budget_tests { #[tokio::test] async fn non_streaming_single_response_passes_through() { - let (tx, mut rx) = mpsc::channel(4); + let (tx, rx) = mpsc::channel(4); + let mut rx = crate::control::ResponseReceiver::from_channel(rx); tx.send(final_bytes(100)).await.unwrap(); drop(tx); // Single terminal frame returns unmodified — no merge, exact bytes. @@ -235,7 +236,8 @@ mod collect_budget_tests { async fn streaming_merges_all_chunk_arrays() { // Three standalone array chunks must merge into ONE array with every // element — the regression: raw concatenation kept only the first array. - let (tx, mut rx) = mpsc::channel(4); + let (tx, rx) = mpsc::channel(4); + let mut rx = crate::control::ResponseReceiver::from_channel(rx); tx.send(partial_rows(1000)).await.unwrap(); tx.send(partial_rows(1000)).await.unwrap(); tx.send(final_rows(500)).await.unwrap(); @@ -251,7 +253,8 @@ mod collect_budget_tests { #[tokio::test] async fn streaming_over_budget_on_partial_aborts() { - let (tx, mut rx) = mpsc::channel(4); + let (tx, rx) = mpsc::channel(4); + let mut rx = crate::control::ResponseReceiver::from_channel(rx); tx.send(partial_bytes(600)).await.unwrap(); tx.send(partial_bytes(600)).await.unwrap(); drop(tx); @@ -264,7 +267,8 @@ mod collect_budget_tests { #[tokio::test] async fn streaming_over_budget_on_final_chunk_aborts() { - let (tx, mut rx) = mpsc::channel(4); + let (tx, rx) = mpsc::channel(4); + let mut rx = crate::control::ResponseReceiver::from_channel(rx); tx.send(partial_bytes(500)).await.unwrap(); tx.send(final_bytes(600)).await.unwrap(); drop(tx); @@ -274,7 +278,8 @@ mod collect_budget_tests { #[tokio::test] async fn a_collect_past_the_deadline_reports_the_deadline() { - let (_tx, mut rx) = mpsc::channel(4); + let (_tx, rx) = mpsc::channel(4); + let mut rx = crate::control::ResponseReceiver::from_channel(rx); let result = collect_under_deadline( &mut rx, DeadlineCollect { @@ -298,7 +303,8 @@ mod collect_budget_tests { // The channel closes rather than answering, and the deadline has // already passed: the closure follows from the statement running out // of time, so reporting it would report the symptom. - let (tx, mut rx) = mpsc::channel(4); + let (tx, rx) = mpsc::channel(4); + let mut rx = crate::control::ResponseReceiver::from_channel(rx); tx.send(partial_bytes(10)).await.unwrap(); drop(tx); let result = collect_under_deadline( @@ -319,7 +325,8 @@ mod collect_budget_tests { #[tokio::test] async fn a_producer_that_stopped_inside_the_budget_reports_the_closure() { - let (tx, mut rx) = mpsc::channel(4); + let (tx, rx) = mpsc::channel(4); + let mut rx = crate::control::ResponseReceiver::from_channel(rx); tx.send(partial_bytes(10)).await.unwrap(); drop(tx); let result = collect_under_deadline( @@ -340,7 +347,8 @@ mod collect_budget_tests { #[tokio::test] async fn channel_closed_without_final_is_explicit_error() { - let (tx, mut rx) = mpsc::channel(4); + let (tx, rx) = mpsc::channel(4); + let mut rx = crate::control::ResponseReceiver::from_channel(rx); tx.send(partial_bytes(10)).await.unwrap(); drop(tx); let err = collect_bounded_response(&mut rx, 1024).await.unwrap_err(); diff --git a/nodedb/src/control/server/dispatch_utils/error_status.rs b/nodedb/src/control/local_dispatch/error_status.rs similarity index 78% rename from nodedb/src/control/server/dispatch_utils/error_status.rs rename to nodedb/src/control/local_dispatch/error_status.rs index 5f3727e18..b2fa8c0c6 100644 --- a/nodedb/src/control/server/dispatch_utils/error_status.rs +++ b/nodedb/src/control/local_dispatch/error_status.rs @@ -15,8 +15,9 @@ use crate::bridge::envelope::{ErrorCode, Response, Status}; /// other code crosses as `Error::DataPlane` so its SQLSTATE survives. An /// error status carrying no code fails closed rather than reading as success. /// -/// `DeadlineExceeded` is the exception, and it crosses as -/// [`crate::Error::DeadlineExceeded`]. A shard refusing an expired task is the +/// `DeadlineExceeded` and `ExpiredBeforeExecution` are the exception, and +/// they cross as [`crate::Error::DeadlineExceeded`]. A shard refusing an +/// expired task is the /// statement running out of time — the same condition the Control-Plane timer /// reports — so both produce one variant and one SQLSTATE. Leaving it wrapped /// would make the SQLSTATE a client sees depend on which half of that race @@ -27,9 +28,11 @@ pub(crate) fn reject_data_plane_error(resp: &Response) -> crate::Result<()> { } match resp.error_code.as_deref() { Some(ErrorCode::NotFound) => Ok(()), - Some(ErrorCode::DeadlineExceeded) => Err(crate::Error::DeadlineExceeded { - request_id: resp.request_id, - }), + Some(ErrorCode::DeadlineExceeded | ErrorCode::ExpiredBeforeExecution) => { + Err(crate::Error::DeadlineExceeded { + request_id: resp.request_id, + }) + } Some(code) => Err(crate::Error::DataPlane(code.clone())), None => Err(crate::Error::DataPlane(ErrorCode::Internal { detail: "data plane returned an error status with no error code".into(), @@ -70,6 +73,17 @@ mod tests { } } + /// A task that expired before it started reports the same deadline. + #[test] + fn a_task_that_never_started_reports_the_deadline() { + match reject_data_plane_error(&refusal(ErrorCode::ExpiredBeforeExecution)) { + Err(crate::Error::DeadlineExceeded { request_id }) => { + assert_eq!(request_id, RequestId::new(9)); + } + other => panic!("expected the deadline variant, got {other:?}"), + } + } + #[test] fn every_other_verdict_keeps_its_data_plane_code() { match reject_data_plane_error(&refusal(ErrorCode::DivisionByZero)) { diff --git a/nodedb/src/control/local_dispatch/local_read.rs b/nodedb/src/control/local_dispatch/local_read.rs new file mode 100644 index 000000000..b751f1d38 --- /dev/null +++ b/nodedb/src/control/local_dispatch/local_read.rs @@ -0,0 +1,111 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Read-only dispatch of an internal scan to this node's own Data Plane. +//! +//! A caller that must never re-enter the write funnel reads through here. +//! [`LocalRead`] can express only reads, so no write plan can reach a core +//! through this path, and no function here calls `submit_write`. Holding a +//! write's acknowledgement open can therefore never wait on a request that +//! passes through the write path again. +//! +//! The request carries `Admission::Exempt(Read)`: a read never takes the +//! write fence. It reads local state on the core that homes the vShard. + +use std::time::{Duration, Instant}; + +use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; +use nodedb_types::{QualifiedCollection, SystemTimeScope}; + +use crate::bridge::envelope::{Admission, ExemptReason, Priority, Request, Response}; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, ReadConsistency, TenantId, TraceId, VShardId}; + +use super::collect::{DeadlineCollect, collect_under_deadline}; + +/// A read an internal caller issues against its own node. +/// +/// Every variant is a read. A write is not representable. +pub(crate) enum LocalRead { + /// Every current row of one document collection. + DocumentScan { collection: QualifiedCollection }, +} + +impl LocalRead { + fn into_plan(self) -> PhysicalPlan { + match self { + LocalRead::DocumentScan { collection } => PhysicalPlan::Document(DocumentOp::Scan { + collection, + filters: Vec::new(), + limit: usize::MAX, + offset: 0, + sort_keys: Vec::new(), + distinct: false, + projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), + system_time: SystemTimeScope::Current, + valid_at_ms: None, + prefilter: None, + }), + } + } +} + +/// Dispatch `read` to the core that homes `vshard_id` and collect its +/// bounded response before the request deadline. +pub(crate) async fn dispatch_local_read( + shared: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + vshard_id: VShardId, + read: LocalRead, +) -> crate::Result { + let deadline = + Instant::now() + Duration::from_secs(shared.tuning.network.default_deadline_secs); + let request_id = shared.next_request_id(); + let request = Request { + request_id, + tenant_id, + database_id, + vshard_id, + plan: read.into_plan(), + deadline, + priority: Priority::Normal, + trace_id: TraceId::ZERO, + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: crate::event::EventSource::User, + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + admission: Admission::Exempt(ExemptReason::Read), + }; + + let mut rx = shared.tracker.register(request_id); + let dispatched = match shared.dispatcher.lock() { + Ok(mut dispatcher) => dispatcher.dispatch(request), + Err(poisoned) => poisoned.into_inner().dispatch(request), + }; + if let Err(error) = dispatched { + // No response will ever arrive for a refused request. + shared.tracker.cancel(&request_id); + return Err(error); + } + let collected = collect_under_deadline( + &mut rx, + DeadlineCollect { + request_id, + deadline, + max_result_bytes: shared.tuning.network.max_query_result_bytes as usize, + context: "internal local read", + }, + ) + .await; + if collected.is_err() { + shared.tracker.cancel(&request_id); + } + collected +} diff --git a/nodedb/src/control/local_dispatch/mod.rs b/nodedb/src/control/local_dispatch/mod.rs new file mode 100644 index 000000000..24bd2ac16 --- /dev/null +++ b/nodedb/src/control/local_dispatch/mod.rs @@ -0,0 +1,15 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Dispatch to this node's own Data Plane: collect a request's bounded +//! response, turn an error response into a typed `Err`, and issue +//! read-only internal scans. `security` and `server` both build on it. + +mod collect; +mod error_status; +mod local_read; + +pub(crate) use collect::{ + DeadlineCollect, DispatchCollectError, collect_bounded_response, collect_under_deadline, +}; +pub(crate) use error_status::reject_data_plane_error; +pub(crate) use local_read::{LocalRead, dispatch_local_read}; diff --git a/nodedb/src/control/maintenance/clone_materializer/columnar.rs b/nodedb/src/control/maintenance/clone_materializer/columnar.rs index f8d547564..da1303f48 100644 --- a/nodedb/src/control/maintenance/clone_materializer/columnar.rs +++ b/nodedb/src/control/maintenance/clone_materializer/columnar.rs @@ -126,9 +126,8 @@ pub(super) async fn materialize_columnar_collection( let target_surrogate = state .surrogate_assigner .assign( - db_id, + nodedb_types::CollectionKey::from_bare(db_id, &coll.name), tenant_id, - &target_qualified, &source_surrogate_u32.to_be_bytes(), ) .map_err(|e| crate::Error::Storage { diff --git a/nodedb/src/control/maintenance/clone_materializer/dispatch.rs b/nodedb/src/control/maintenance/clone_materializer/dispatch.rs index 85163ae11..1d46d7211 100644 --- a/nodedb/src/control/maintenance/clone_materializer/dispatch.rs +++ b/nodedb/src/control/maintenance/clone_materializer/dispatch.rs @@ -34,7 +34,9 @@ pub(crate) async fn dispatch_local( plan: PhysicalPlan, txn_id: Option, ) -> crate::Result { - let vshard_id = VShardId::from_collection_in_database(database_id, collection_qualified); + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(database_id, collection_qualified)? + .vshard(); dispatch_local_on_vshard(state, tenant_id, database_id, vshard_id, plan, txn_id).await } diff --git a/nodedb/src/control/maintenance/clone_materializer/document.rs b/nodedb/src/control/maintenance/clone_materializer/document.rs index 18b94db46..5ff243b4c 100644 --- a/nodedb/src/control/maintenance/clone_materializer/document.rs +++ b/nodedb/src/control/maintenance/clone_materializer/document.rs @@ -106,9 +106,11 @@ pub(super) async fn materialize_document_collection( // would allocate for this (collection, pk) pair. let pk_bytes = catalog .get_pk_for_surrogate( - origin.source_database, + nodedb_types::CollectionKey::from_bare( + origin.source_database, + &origin.source_collection, + ), tenant_id, - &origin.source_collection, source_surrogate, ) .map_err(|e| crate::Error::Storage { @@ -130,7 +132,11 @@ pub(super) async fn materialize_document_collection( // the normal INSERT path would use. let target_surrogate = state .surrogate_assigner - .assign(db_id, tenant_id, &target_qualified, &pk_bytes) + .assign( + nodedb_types::CollectionKey::from_bare(db_id, &coll.name), + tenant_id, + &pk_bytes, + ) .map_err(|e| crate::Error::Storage { engine: "clone_materializer".into(), detail: format!( diff --git a/nodedb/src/control/maintenance/clone_materializer/kv.rs b/nodedb/src/control/maintenance/clone_materializer/kv.rs index abe438050..7c73cdb18 100644 --- a/nodedb/src/control/maintenance/clone_materializer/kv.rs +++ b/nodedb/src/control/maintenance/clone_materializer/kv.rs @@ -93,10 +93,11 @@ pub(super) async fn materialize_kv_collection( continue; } - let surrogate = - state - .surrogate_assigner - .assign(db_id, tenant_id, &target_qualified, &key)?; + let surrogate = state.surrogate_assigner.assign( + nodedb_types::CollectionKey::from_bare(db_id, &coll.name), + tenant_id, + &key, + )?; let plan = PhysicalPlan::Kv(KvOp::Put { collection: nodedb_types::QualifiedCollection::new(db_id, &coll.name), key: key.clone(), @@ -105,6 +106,7 @@ pub(super) async fn materialize_kv_collection( surrogate, returning: None, rls_filters: Vec::new(), + provenance: None, }); let resp = dispatch_local(state, tenant_id, db_id, &target_qualified, plan, None).await?; diff --git a/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs b/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs index bee1e3ab0..c83443852 100644 --- a/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs +++ b/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs @@ -18,7 +18,7 @@ use nodedb_types::TenantId; -use crate::bridge::envelope::{PhysicalPlan, Status}; +use crate::bridge::envelope::PhysicalPlan; use crate::control::maintenance::clone_materializer::{dispatch_local, read_all_source_rows}; use crate::control::state::SharedState; use crate::types::VShardId; @@ -91,8 +91,11 @@ pub(crate) async fn resolve_and_emit_merge_ops( })?; let target_pk = resolve_target_pk(&target, "MERGE")?; - let vshard_id = - VShardId::from_collection_in_database(task.database_id, target_collection.as_str()); + let vshard_id = nodedb_types::CollectionKey::from_qualified_str( + task.database_id, + target_collection.as_str(), + )? + .vshard(); let mut out: Vec = Vec::new(); emit_arms( state, @@ -178,15 +181,10 @@ async fn resolve_merge_arms( task.txn_id, ) .await?; - if resolve_resp.status != Status::Ok { - return Err(crate::Error::Dispatch { - detail: format!( - "in-transaction MERGE resolve failed: {:?}", - resolve_resp.error_code - ), - }); - } - decode_resolve(&resolve_resp.payload) + // A refused resolve keeps its Data-Plane code. + let payload = + crate::control::server::shared::response_payload::payload_or_typed_error(resolve_resp)?; + decode_resolve(&payload) } /// Rewrite the three resolved arms into concrete point-write tasks appended @@ -205,9 +203,8 @@ fn emit_arms( for (_join_key, body) in arms.inserts { let surrogate = assign_target_surrogate( state, - task.database_id, + nodedb_types::CollectionKey::from_qualified_str(task.database_id, target_collection)?, task.tenant_id, - target_collection, target_pk, &body, )?; diff --git a/nodedb/src/control/merge_orchestrator/orchestrator.rs b/nodedb/src/control/merge_orchestrator/orchestrator.rs index f8e024c6b..1d8cb442f 100644 --- a/nodedb/src/control/merge_orchestrator/orchestrator.rs +++ b/nodedb/src/control/merge_orchestrator/orchestrator.rs @@ -230,9 +230,11 @@ pub(crate) async fn run_merge(state: &SharedState, args: MergeArgs<'_>) -> crate for (join_key, body) in &insert_rows { let surrogate = assign_target_surrogate( state, - args.database_id, + nodedb_types::CollectionKey::from_qualified_str( + args.database_id, + args.target_collection, + )?, args.tenant_id, - args.target_collection, &target_pk, body, )?; diff --git a/nodedb/src/control/metadata_proposer.rs b/nodedb/src/control/metadata_proposer.rs deleted file mode 100644 index f5e8eee83..000000000 --- a/nodedb/src/control/metadata_proposer.rs +++ /dev/null @@ -1,710 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Synchronous `propose-and-wait-for-local-apply` helper for -//! replicated catalog DDL. -//! -//! The sole entry point pgwire DDL handlers use to write a -//! [`CatalogEntry`] through the metadata raft group (group 0). It is -//! deliberately sync — pgwire DDL handlers are not async, and -//! `tokio::task::block_in_place`-style wrapping keeps the blocking -//! wait from starving the tokio runtime. -//! -//! Semantics: -//! -//! 1. If no cluster is configured (`shared.metadata_raft` not -//! installed), returns `ProposeOutcome::LocalOnly`. The caller's -//! single-node direct-write path stays authoritative. -//! 2. If this node is the metadata-group leader, proposes the -//! entry, blocks until its local applied watermark reaches the -//! assigned log index (5s default timeout), and returns the -//! log index on success. -//! 3. If this node is NOT the leader, returns -//! `Error::Config { detail: "metadata propose: not leader ..." }`. -//! Gateway-side redirection will make this transparent. - -use std::sync::atomic::Ordering; -use std::sync::{Arc, Weak}; -use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH}; - -use tokio::runtime::RuntimeFlavor; - -use nodedb_cluster::{METADATA_GROUP_ID, MetadataEntry, WaitOutcome, encode_entry}; - -#[cfg(test)] -use nodedb_cluster::AppliedIndexWatcher; - -use crate::control::catalog_entry::{self, CatalogEntry}; -use crate::control::propose_outcome::ProposeOutcome; -use crate::control::state::SharedState; -use crate::error::Error; - -/// Default upper bound on how long a single -/// `propose_catalog_entry` call will block before returning an -/// error. -pub const DEFAULT_PROPOSE_TIMEOUT: Duration = Duration::from_secs(5); - -/// Default upper bound on how long a DDL drain will wait for -/// prior-version leases to release before giving up. Must be at -/// least `ClusterTransportTuning::descriptor_lease_duration_secs` -/// so an existing lease gets at least one full lifetime to -/// expire naturally. 35 seconds matches the 300s lease duration -/// plus a 30-second grace minus the typical 5-minute default -/// cut down for test budget — in production -/// `propose_catalog_entry_with_drain_timeout` can pass a longer -/// value if an operator is willing to wait. -pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(35); -const DDL_PREPARE_LEASE: Duration = Duration::from_secs(60); -const DDL_PREPARE_WAIT: Duration = Duration::from_secs(70); - -/// Type-erased handle for proposing to the metadata raft group. -/// -/// The apply watermark for the metadata group lives on -/// [`SharedState::applied_index_watcher`] (keyed by -/// [`nodedb_cluster::METADATA_GROUP_ID`]); callers of [`Self::propose`] -/// look it up there rather than receiving it through this handle. -pub trait MetadataRaftHandle: Send + Sync { - /// Propose a raw encoded `MetadataEntry` to the metadata group. - /// Returns its assigned log index on success. - fn propose(&self, bytes: Vec) -> Result; -} - -/// Concrete impl wrapping `nodedb_cluster::RaftLoop`. -/// -/// Holds the loop weakly: this handle lives on `SharedState`, which is -/// itself kept alive transitively by the `RaftLoop`, so a strong -/// reference here would close a cycle that pins both forever and blocks -/// clean shutdown. The loop is kept alive by its own spawned tasks; -/// `upgrade` therefore succeeds throughout normal operation and only -/// fails once the loop has been dropped on shutdown. -pub struct RaftLoopProposerHandle { - raft_loop: Weak< - nodedb_cluster::RaftLoop< - crate::control::cluster::SpscCommitApplier, - crate::control::LocalPlanExecutor, - >, - >, -} - -impl RaftLoopProposerHandle { - pub fn new( - raft_loop: Arc< - nodedb_cluster::RaftLoop< - crate::control::cluster::SpscCommitApplier, - crate::control::LocalPlanExecutor, - >, - >, - ) -> Self { - Self { - raft_loop: Arc::downgrade(&raft_loop), - } - } -} - -impl MetadataRaftHandle for RaftLoopProposerHandle { - fn propose(&self, bytes: Vec) -> Result { - // The cluster crate's `propose_to_metadata_group_via_leader` - // is async because it may need to forward to the metadata - // leader over QUIC. The trait method is sync because every - // caller (catalog DDL handlers, lease grant/release helpers) - // is itself sync but runs inside a tokio task. Wrap in - // `block_in_place` + the current runtime's `block_on` so the - // forwarding QUIC round-trip drives without starving the - // raft tick that produces the leader_hint. - // `upgrade` fails only once the raft loop has been dropped on - // shutdown; a request racing shutdown then fails cleanly with a - // typed error instead of panicking. - let raft_loop = self.raft_loop.upgrade().ok_or_else(|| Error::Config { - detail: "metadata propose: cluster not running".into(), - })?; - tokio::task::block_in_place(|| { - tokio::runtime::Handle::current() - .block_on(raft_loop.propose_to_metadata_group_via_leader(bytes)) - }) - .map_err(|e| match e { - // An election in progress is transient, not a failure of this - // proposal. Keep it typed rather than flattening it into a generic - // config error, so callers can wait the election out instead of - // failing the statement — a node that has just restarted answers - // every metadata proposal this way for a moment. - nodedb_cluster::ClusterError::Raft(nodedb_raft::RaftError::NotLeader { - leader_hint: None, - }) => Error::MetadataLeaderUnavailable, - other => Error::Config { - detail: format!("metadata propose: {other}"), - }, - }) - } -} - -fn wall_now_ns() -> u64 { - SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap_or_default() - .as_nanos() - .min(u64::MAX as u128) as u64 -} - -fn propose_metadata_and_wait( - shared: &SharedState, - handle: &dyn MetadataRaftHandle, - entry: &MetadataEntry, - timeout: Duration, -) -> Result { - let raw = encode_entry(entry).map_err(|e| Error::Config { - detail: format!("metadata entry encode: {e}"), - })?; - let index = handle.propose(raw)?; - let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); - let outcome = tokio::task::block_in_place(|| watcher.wait_for(index, timeout)); - match outcome { - WaitOutcome::Reached => Ok(index), - WaitOutcome::TimedOut => Err(Error::Config { - detail: format!( - "metadata propose timed out after {timeout:?} waiting for log index {index} (current: {})", - watcher.current() - ), - }), - WaitOutcome::GroupGone => Err(Error::Config { - detail: "metadata group no longer hosted on this node".into(), - }), - } -} - -/// RAII ownership of the metadata-Raft-serialized descriptor preparation lease. -/// The matching release is itself replicated, so another node cannot stamp from -/// the same prior catalog version until this guard is dropped and that release -/// has applied. -pub(crate) struct DdlPrepareGuard<'a> { - shared: &'a SharedState, - handle: &'a dyn MetadataRaftHandle, - token: u64, -} - -impl DdlPrepareGuard<'_> { - pub(crate) fn token(&self) -> u64 { - self.token - } -} - -impl Drop for DdlPrepareGuard<'_> { - fn drop(&mut self) { - if let Err(error) = propose_metadata_and_wait( - self.shared, - self.handle, - &MetadataEntry::DdlPrepareRelease { token: self.token }, - DEFAULT_PROPOSE_TIMEOUT, - ) { - tracing::error!(token = self.token, %error, "metadata DDL lease release failed"); - } - } -} - -pub(crate) fn acquire_ddl_prepare_lease<'a>( - shared: &'a SharedState, - handle: &'a dyn MetadataRaftHandle, -) -> Result, Error> { - let sequence = shared - .metadata_ddl_token_seq - .fetch_add(1, Ordering::Relaxed); - let token = shared.node_id.wrapping_mul(0x9e37_79b9_7f4a_7c15) - ^ wall_now_ns().rotate_left(17) - ^ sequence; - let deadline = Instant::now() + DDL_PREPARE_WAIT; - - loop { - propose_metadata_and_wait( - shared, - handle, - &MetadataEntry::DdlPrepareAcquire { token }, - DEFAULT_PROPOSE_TIMEOUT, - )?; - - loop { - let owner = *shared - .metadata_ddl_owner - .lock() - .map_err(|_| Error::Config { - detail: "metadata DDL owner lock poisoned".into(), - })?; - match owner { - Some((current, _)) if current == token => { - return Ok(DdlPrepareGuard { - shared, - handle, - token, - }); - } - Some((current, acquired_at)) - if shared.is_metadata_leader() - && acquired_at.elapsed() >= DDL_PREPARE_LEASE => - { - // Cancel the dead owner's pending record before releasing its - // lease, so it never lingers visible-but-unresolved past the lease. - if shared.pending_ddl.contains(current) { - propose_metadata_and_wait( - shared, - handle, - &MetadataEntry::DdlPendingCancel { token: current }, - DEFAULT_PROPOSE_TIMEOUT, - )?; - } - propose_metadata_and_wait( - shared, - handle, - &MetadataEntry::DdlPrepareRelease { token: current }, - DEFAULT_PROPOSE_TIMEOUT, - )?; - break; - } - None => break, - Some(_) if Instant::now() < deadline => { - // Reached from async tasks (ILP batch flush -> - // `propose_catalog_entry`), so hand the worker back to - // tokio rather than parking it: the lease owner this - // polls for is released by a raft apply that needs a - // worker to make progress. - tokio::task::block_in_place(|| { - std::thread::sleep(Duration::from_millis(10)); - }); - } - Some(_) => { - return Err(Error::Config { - detail: "metadata DDL preparation lease timed out".into(), - }); - } - } - } - } -} - -/// Take the local DDL preparation lock, handing the wait back to tokio when -/// the caller is on a multi-thread worker. -/// -/// The holder keeps this lock across the distributed preparation lease, the -/// descriptor drain and the local apply wait — each already wrapped in -/// `block_in_place`, but that only tells tokio about the waits *inside* the -/// lock, never about the wait *for* it. A bare `lock()` on a worker therefore -/// removes that worker from the runtime silently, including from the raft -/// apply work the current holder needs in order to finish, which turns -/// contention into a self-sustaining stall. -/// -/// `block_in_place` is a passthrough outside a multi-thread worker (plain sync -/// callers, blocking-pool threads) and panics on the current-thread runtime, -/// so it is applied only where it is both legal and meaningful — mirroring -/// `lease::drain_propose::poll_leases_drained`. -fn lock_ddl_preparation(shared: &SharedState) -> Result, Error> { - let acquire = || { - shared.metadata_ddl_lock.lock().map_err(|_| Error::Config { - detail: "metadata DDL preparation lock poisoned".into(), - }) - }; - match tokio::runtime::Handle::try_current() { - Ok(handle) if handle.runtime_flavor() == RuntimeFlavor::MultiThread => { - tokio::task::block_in_place(acquire) - } - _ => acquire(), - } -} - -/// Propose a `CatalogEntry` and block until the local applied-index -/// watcher confirms the entry has been applied on this node. -/// -/// The returned [`ProposeOutcome`] tells the caller whether to write the -/// catalog itself, leave it to the applier, or do nothing because the entry -/// is held for COMMIT. -pub fn propose_catalog_entry( - shared: &SharedState, - entry: &CatalogEntry, -) -> Result { - propose_catalog_entry_with_timeout(shared, entry, DEFAULT_PROPOSE_TIMEOUT) -} - -/// Same as [`propose_catalog_entry`] but with an explicit timeout. -pub fn propose_catalog_entry_with_timeout( - shared: &SharedState, - entry: &CatalogEntry, - timeout: Duration, -) -> Result { - // Buffering is decided first, ahead of every replication-mode gate: an open - // transaction owns the entry regardless of whether this deployment - // replicates DDL, and COMMIT re-runs the mode choice for the whole batch. - // Entries also stay unstamped until then, so repeated mutations of one - // descriptor receive distinct versions in commit order. - if crate::control::server::shared::session::ddl_buffer::try_buffer(entry.clone()) { - return Ok(ProposeOutcome::Buffered); - } - - let Some(handle) = shared.metadata_raft.get() else { - return Ok(ProposeOutcome::LocalOnly); - }; - - // Rolling-upgrade gate: until every node in the cluster reports - // at least `DISTRIBUTED_CATALOG_VERSION`, fall back to the legacy - // direct-write path on the originating node. Mixing the - // replicated and direct paths during a partial upgrade would - // diverge catalog state across nodes — see - // `control/rolling_upgrade.rs`. - if !shared - .cluster_version_view() - .can_activate_feature(crate::control::rolling_upgrade::DISTRIBUTED_CATALOG_VERSION) - { - tracing::warn!( - min_version = shared.cluster_version_view().min_version, - required = crate::control::rolling_upgrade::DISTRIBUTED_CATALOG_VERSION, - "metadata propose: cluster in compat mode (mixed-version), \ - falling back to legacy direct-write path" - ); - return Ok(ProposeOutcome::LocalOnly); - } - - // Serialize preparation through local apply confirmation. Without this, - // concurrent proposers can both observe persisted version N and emit N+1. - let _local_ddl_guard = lock_ddl_preparation(shared)?; - let distributed_ddl_guard = acquire_ddl_prepare_lease(shared, handle.as_ref())?; - - // Drain for Put* variants that carry descriptor_version. - // Leases acquired at plan time are refcounted and held - // through execute; when the last in-flight query using a - // descriptor completes, its `QueryLeaseScope` drops and the - // refcount hits zero, releasing the lease. Drain is what - // makes this an actual barrier: the proposer waits for all - // prior-version leases to release before committing the new - // `Put*`, giving long-running in-flight queries a bounded - // window (DEFAULT_DRAIN_TIMEOUT) to finish. - if let Some((descriptor_id, prior_version)) = - crate::control::lease::descriptor_id_and_prior_version(entry, shared) - && prior_version > 0 - { - crate::control::lease::drain_for_ddl( - shared, - descriptor_id, - prior_version, - DEFAULT_DRAIN_TIMEOUT, - // No transactional lease scope of its own: this is a bare, - // unbuffered DDL statement, not a COMMIT finalizing buffered DDL - // alongside a buffered write to the same descriptor. - 0, - )?; - } - - // Freeze the descriptor_version / constraint_version / - // modification_hlc HERE, at propose time, so the value is computed - // exactly once from this node's local catalog (`prior + 1`) and - // then replicated verbatim inside the entry. Every node applies the - // frozen value without re-deriving it, which makes replay-from-log - // on restart and re-delivery during learner catch-up idempotent — - // the divergence that a per-node apply-time stamp produced is gone. - // - // Gated on the same rolling-upgrade flag the apply path used to - // gate on: only stamp once every node can activate descriptor - // versioning; otherwise leave the entry's sentinel version `0` - // (downstream resolvers treat `0` as `1`). Older nodes in a - // mixed-version cluster lack the stamp logic, so a stamped value - // would not be reproduced symmetrically there. - let stamped_owned; - let entry: &CatalogEntry = if shared - .cluster_version_view() - .can_activate_feature(crate::control::rolling_upgrade::DESCRIPTOR_VERSIONING_VERSION) - { - stamped_owned = catalog_entry::descriptor_stamp::stamp( - entry.clone(), - &shared.hlc_clock, - shared.credentials.catalog(), - ); - &stamped_owned - } else { - entry - }; - - let payload = catalog_entry::encode(entry)?; - - // Attach J.4 audit context when the pgwire statement boundary - // installed one. Internal callers (descriptor lease grant/release, - // drain proposer) run outside that scope and emit the plain - // `CatalogDdl` variant — they have no SQL text to log. - let catalog_entry = match crate::control::server::shared::session::audit_context::current() { - Some(ctx) => MetadataEntry::CatalogDdlAudited { - payload, - auth_user_id: ctx.auth_user_id, - auth_user_name: ctx.auth_user_name, - sql_text: ctx.sql_text, - }, - None => MetadataEntry::CatalogDdl { payload }, - }; - let metadata_entry = MetadataEntry::DdlPrepared { - token: distributed_ddl_guard.token(), - entry: Box::new(catalog_entry), - }; - let raw = encode_entry(&metadata_entry).map_err(|e| Error::Config { - detail: format!("metadata entry encode: {e}"), - })?; - - let log_index = handle.propose(raw)?; - - let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); - // `wait_for` blocks the calling thread on a Condvar. When the - // caller is already inside a tokio task (pgwire handlers always - // are), parking the worker without telling tokio starves every - // other task that lands on it — including the raft tick that - // would otherwise bump the watcher. Wrap the blocking section - // in `block_in_place` so tokio reassigns a fresh worker. - let outcome = tokio::task::block_in_place(|| watcher.wait_for(log_index, timeout)); - match outcome { - WaitOutcome::Reached - if shared.metadata_ddl_applied_token.load(Ordering::Acquire) - == distributed_ddl_guard.token() => - { - Ok(ProposeOutcome::Replicated { log_index }) - } - WaitOutcome::Reached => Err(Error::Config { - detail: "metadata DDL preparation ownership was superseded before apply".into(), - }), - WaitOutcome::TimedOut => Err(Error::Config { - detail: format!( - "metadata propose timed out after {:?} waiting for log index {} (current: {})", - timeout, - log_index, - watcher.current() - ), - }), - WaitOutcome::GroupGone => Err(Error::Config { - detail: "metadata group no longer hosted on this node".into(), - }), - } -} - -/// Propose a surrogate high-watermark advance to the metadata Raft group -/// and wait for it to be applied locally. -/// -/// In single-node / no-cluster mode (no `metadata_raft` installed), -/// returns `Ok(0)` immediately — the WAL-only path on `SharedState` is -/// still sufficient. In cluster mode this is called by the leader-side -/// flush path instead of (or in addition to) the local WAL record, so -/// every follower's `SurrogateRegistry` advances to the same hwm via the -/// Raft commit. -/// -/// `hwm` is the highest surrogate that has been issued so far on this -/// node. Followers apply the entry by calling -/// `SurrogateRegistry::restore_hwm(hwm)` (idempotent, monotonic). -pub fn propose_surrogate_hwm(shared: &SharedState, hwm: u32) -> Result { - let Some(handle) = shared.metadata_raft.get() else { - return Ok(0); - }; - - let entry = MetadataEntry::SurrogateAlloc { hwm }; - let raw = encode_entry(&entry).map_err(|e| Error::Config { - detail: format!("surrogate_alloc encode: {e}"), - })?; - - let log_index = handle.propose(raw)?; - - let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); - let outcome = - tokio::task::block_in_place(|| watcher.wait_for(log_index, DEFAULT_PROPOSE_TIMEOUT)); - if !outcome.is_reached() { - return Err(Error::Config { - detail: format!("surrogate_alloc propose timed out waiting for log index {log_index}"), - }); - } - - Ok(log_index) -} - -/// Propose a HiLo surrogate batch reservation to the metadata Raft group -/// and wait for the commit (returns the assigned log index). -/// -/// In single-node / no-cluster mode (no `metadata_raft` installed), -/// returns `Ok(0)` immediately — single-node uses the local `alloc_one` -/// path and never reaches here. Kept as a safety guard only. -/// -/// The carved `[start, end)` range is NOT decided here: it is computed -/// at apply time on every node by advancing the global watermark in -/// identical log order (see `MetadataEntry::SurrogateReserve`). The -/// caller therefore cannot learn the range from this commit-wait alone -/// — `wait_for` returns on COMMIT, before the apply handler runs. The -/// owning node's apply handler fires an explicit completion signal -/// (`SurrogateAssigner::complete_reservation`) that the caller awaits -/// separately to learn the range. -/// -/// `node_id` + `request_id` identify this node's specific in-flight -/// reservation so the apply handler routes the batch + signal back to it. -pub fn propose_surrogate_reserve( - shared: &SharedState, - node_id: u64, - request_id: u64, - batch_size: u32, -) -> Result { - let Some(handle) = shared.metadata_raft.get() else { - return Ok(0); - }; - - let entry = MetadataEntry::SurrogateReserve { - node_id, - request_id, - batch_size, - }; - let raw = encode_entry(&entry).map_err(|e| Error::Config { - detail: format!("surrogate_reserve encode: {e}"), - })?; - - let log_index = handle.propose(raw)?; - - let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); - let outcome = - tokio::task::block_in_place(|| watcher.wait_for(log_index, DEFAULT_PROPOSE_TIMEOUT)); - if !outcome.is_reached() { - return Err(Error::Config { - detail: format!( - "surrogate_reserve propose timed out waiting for log index {log_index}" - ), - }); - } - - Ok(log_index) -} - -/// Propose a Lite client registration through the metadata Raft group and -/// wait for it to be applied locally. -/// -/// In single-node / no-cluster mode (no `metadata_raft` installed), -/// returns `Ok(0)` immediately — the local registry write already persisted -/// the state. In cluster mode every follower applies the entry via -/// `SyncProducerRegistry::apply_register` so the `(producer_id, epoch)` pair -/// agrees on all nodes and survives leader failover. -pub fn propose_sync_producer_register( - shared: &SharedState, - lite_id: &str, - producer_id: u64, - tenant_id: u64, - user_id: u64, - epoch: u64, - created_ms: i64, -) -> Result { - let Some(handle) = shared.metadata_raft.get() else { - return Ok(0); - }; - - let entry = MetadataEntry::SyncProducerRegister { - lite_id: lite_id.to_owned(), - producer_id, - tenant_id, - user_id, - epoch, - created_ms, - }; - let raw = encode_entry(&entry).map_err(|e| Error::Config { - detail: format!("sync_producer_register encode: {e}"), - })?; - - let log_index = handle.propose(raw)?; - - let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); - let outcome = - tokio::task::block_in_place(|| watcher.wait_for(log_index, DEFAULT_PROPOSE_TIMEOUT)); - if !outcome.is_reached() { - return Err(Error::Config { - detail: format!( - "sync_producer_register propose timed out waiting for log index {log_index}" - ), - }); - } - - Ok(log_index) -} - -/// Propose a Lite client epoch fence through the metadata Raft group and -/// wait for it to be applied locally. -/// -/// In single-node / no-cluster mode (no `metadata_raft` installed), -/// returns `Ok(0)` immediately — the local registry write already persisted -/// the state. In cluster mode every follower applies the entry via -/// `SyncProducerRegistry::apply_fence` (max-wins) so the epoch advance -/// survives leader failover. -pub fn propose_sync_producer_fence( - shared: &SharedState, - lite_id: &str, - new_epoch: u64, -) -> Result { - let Some(handle) = shared.metadata_raft.get() else { - return Ok(0); - }; - - let entry = MetadataEntry::SyncProducerFence { - lite_id: lite_id.to_owned(), - new_epoch, - }; - let raw = encode_entry(&entry).map_err(|e| Error::Config { - detail: format!("sync_producer_fence encode: {e}"), - })?; - - let log_index = handle.propose(raw)?; - - let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); - let outcome = - tokio::task::block_in_place(|| watcher.wait_for(log_index, DEFAULT_PROPOSE_TIMEOUT)); - if !outcome.is_reached() { - return Err(Error::Config { - detail: format!( - "sync_producer_fence propose timed out waiting for log index {log_index}" - ), - }); - } - - Ok(log_index) -} - -/// Propose ownership of one Loro peer id through the metadata Raft group and -/// wait for it to be applied locally. -/// -/// In single-node / no-cluster mode (no `metadata_raft` installed), returns -/// `Ok(0)` immediately — the local registry write already persisted the -/// ownership. In cluster mode the caller must re-read the owner after this -/// returns: the apply is lowest-producer-id-wins, so a node that lost a race it -/// did not know it was in learns the real owner only once the entry lands. -pub fn propose_sync_peer_bind( - shared: &SharedState, - binding: &crate::control::security::catalog::sync_producer::PeerBindingKey, - producer_id: u64, - bound_ms: i64, -) -> Result { - let Some(handle) = shared.metadata_raft.get() else { - return Ok(0); - }; - - let entry = MetadataEntry::SyncPeerBind { - database_id: binding.database_id, - tenant_id: binding.tenant_id, - collection: binding.collection.clone(), - peer_id: binding.peer_id, - producer_id, - bound_ms, - }; - let raw = encode_entry(&entry).map_err(|e| Error::Config { - detail: format!("sync_peer_bind encode: {e}"), - })?; - - let log_index = handle.propose(raw)?; - - let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); - let outcome = - tokio::task::block_in_place(|| watcher.wait_for(log_index, DEFAULT_PROPOSE_TIMEOUT)); - if !outcome.is_reached() { - return Err(Error::Config { - detail: format!("sync_peer_bind propose timed out waiting for log index {log_index}"), - }); - } - - Ok(log_index) -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn watcher_helper_returns_reached_on_past_target() { - let w = AppliedIndexWatcher::new(); - w.bump(10); - assert!(w.wait_for(5, Duration::from_millis(1)).is_reached()); - } -} diff --git a/nodedb/src/control/metadata_proposer/catalog.rs b/nodedb/src/control/metadata_proposer/catalog.rs new file mode 100644 index 000000000..8f2d7d10d --- /dev/null +++ b/nodedb/src/control/metadata_proposer/catalog.rs @@ -0,0 +1,208 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Propose a catalog entry and wait until this node applied it. + +use std::sync::atomic::Ordering; +use std::time::Duration; + +use nodedb_cluster::{METADATA_GROUP_ID, MetadataEntry, WaitOutcome, encode_entry}; + +use crate::control::catalog_entry::{self, CatalogEntry}; +use crate::control::propose_outcome::ProposeOutcome; +use crate::control::state::SharedState; +use crate::error::Error; + +use super::ddl_prepare::{acquire_ddl_prepare_lease, lock_ddl_preparation}; +use super::timeouts::{DEFAULT_DRAIN_TIMEOUT, DEFAULT_PROPOSE_TIMEOUT}; + +/// Propose a `CatalogEntry` and block until the local applied-index +/// watcher confirms the entry has been applied on this node. +/// +/// The returned [`ProposeOutcome`] tells the caller whether to write the +/// catalog itself, leave it to the applier, or do nothing because the entry +/// is held for COMMIT. +pub fn propose_catalog_entry( + shared: &SharedState, + entry: &CatalogEntry, +) -> Result { + propose_catalog_entry_with_timeout(shared, entry, DEFAULT_PROPOSE_TIMEOUT) +} + +/// Same as [`propose_catalog_entry`] but with an explicit timeout. +/// +/// An entry that changes authorization state returns only once it binds +/// every node: the authorization barrier runs after the local apply, with the +/// DDL preparation lock already released. +pub fn propose_catalog_entry_with_timeout( + shared: &SharedState, + entry: &CatalogEntry, + timeout: Duration, +) -> Result { + let outcome = propose_and_apply_locally(shared, entry, timeout)?; + if let ProposeOutcome::Replicated { log_index } = outcome + && entry.bears_authorization() + { + crate::control::security::auth_lease::block_on_barrier( + shared, + vec![nodedb_cluster::GroupCoverage { + group_id: METADATA_GROUP_ID, + through: log_index, + }], + )?; + } + Ok(outcome) +} + +/// Propose `entry` and wait until this node applied it. +fn propose_and_apply_locally( + shared: &SharedState, + entry: &CatalogEntry, + timeout: Duration, +) -> Result { + // Buffering is decided first, ahead of every replication-mode gate: an open + // transaction owns the entry regardless of whether this deployment + // replicates DDL, and COMMIT re-runs the mode choice for the whole batch. + // Entries also stay unstamped until then, so repeated mutations of one + // descriptor receive distinct versions in commit order. + if crate::control::server::shared::session::ddl_buffer::try_buffer(entry.clone()) { + return Ok(ProposeOutcome::Buffered); + } + + let Some(handle) = shared.metadata_raft.get() else { + return Ok(ProposeOutcome::LocalOnly); + }; + + // Rolling-upgrade gate: until every node in the cluster reports + // at least `DISTRIBUTED_CATALOG_VERSION`, fall back to the legacy + // direct-write path on the originating node. Mixing the + // replicated and direct paths during a partial upgrade would + // diverge catalog state across nodes — see + // `control/rolling_upgrade.rs`. + if !shared + .cluster_version_view() + .can_activate_feature(crate::control::rolling_upgrade::DISTRIBUTED_CATALOG_VERSION) + { + tracing::warn!( + min_version = shared.cluster_version_view().min_version, + required = crate::control::rolling_upgrade::DISTRIBUTED_CATALOG_VERSION, + "metadata propose: cluster in compat mode (mixed-version), \ + falling back to legacy direct-write path" + ); + return Ok(ProposeOutcome::LocalOnly); + } + + // Serialize preparation through local apply confirmation. Without this, + // concurrent proposers can both observe persisted version N and emit N+1. + let _local_ddl_guard = lock_ddl_preparation(shared)?; + let distributed_ddl_guard = acquire_ddl_prepare_lease(shared, handle.as_ref())?; + + // Drain for Put* variants that carry descriptor_version. + // Leases acquired at plan time are refcounted and held + // through execute; when the last in-flight query using a + // descriptor completes, its `QueryLeaseScope` drops and the + // refcount hits zero, releasing the lease. Drain is what + // makes this an actual barrier: the proposer waits for all + // prior-version leases to release before committing the new + // `Put*`, giving long-running in-flight queries a bounded + // window (DEFAULT_DRAIN_TIMEOUT) to finish. + if let Some((descriptor_id, prior_version)) = + crate::control::lease::descriptor_id_and_prior_version(entry, shared) + && prior_version > 0 + { + crate::control::lease::drain_for_ddl( + shared, + descriptor_id, + prior_version, + DEFAULT_DRAIN_TIMEOUT, + // No transactional lease scope of its own: this is a bare, + // unbuffered DDL statement, not a COMMIT finalizing buffered DDL + // alongside a buffered write to the same descriptor. + 0, + )?; + } + + // Freeze the descriptor_version / constraint_version / + // modification_hlc HERE, at propose time, so the value is computed + // exactly once from this node's local catalog (`prior + 1`) and + // then replicated verbatim inside the entry. Every node applies the + // frozen value without re-deriving it, which makes replay-from-log + // on restart and re-delivery during learner catch-up idempotent — + // the divergence that a per-node apply-time stamp produced is gone. + // + // Gated on the same rolling-upgrade flag the apply path used to + // gate on: only stamp once every node can activate descriptor + // versioning; otherwise leave the entry's sentinel version `0` + // (downstream resolvers treat `0` as `1`). Older nodes in a + // mixed-version cluster lack the stamp logic, so a stamped value + // would not be reproduced symmetrically there. + let stamped_owned; + let entry: &CatalogEntry = if shared + .cluster_version_view() + .can_activate_feature(crate::control::rolling_upgrade::DESCRIPTOR_VERSIONING_VERSION) + { + stamped_owned = catalog_entry::descriptor_stamp::stamp( + entry.clone(), + &shared.hlc_clock, + shared.credentials.catalog(), + ); + &stamped_owned + } else { + entry + }; + + let payload = catalog_entry::encode(entry)?; + + // Attach J.4 audit context when the pgwire statement boundary + // installed one. Internal callers (descriptor lease grant/release, + // drain proposer) run outside that scope and emit the plain + // `CatalogDdl` variant — they have no SQL text to log. + let catalog_entry = match crate::control::server::shared::session::audit_context::current() { + Some(ctx) => MetadataEntry::CatalogDdlAudited { + payload, + auth_user_id: ctx.auth_user_id, + auth_user_name: ctx.auth_user_name, + sql_text: ctx.sql_text, + }, + None => MetadataEntry::CatalogDdl { payload }, + }; + let metadata_entry = MetadataEntry::DdlPrepared { + token: distributed_ddl_guard.token(), + entry: Box::new(catalog_entry), + }; + let raw = encode_entry(&metadata_entry).map_err(|e| Error::Config { + detail: format!("metadata entry encode: {e}"), + })?; + + let log_index = handle.propose(raw)?; + + let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); + // `wait_for` blocks the calling thread on a Condvar. When the + // caller is already inside a tokio task (pgwire handlers always + // are), parking the worker without telling tokio starves every + // other task that lands on it — including the raft tick that + // would otherwise bump the watcher. Wrap the blocking section + // in `block_in_place` so tokio reassigns a fresh worker. + let outcome = tokio::task::block_in_place(|| watcher.wait_for(log_index, timeout)); + match outcome { + WaitOutcome::Reached + if shared.metadata_ddl_applied_token.load(Ordering::Acquire) + == distributed_ddl_guard.token() => + { + Ok(ProposeOutcome::Replicated { log_index }) + } + WaitOutcome::Reached => Err(Error::Config { + detail: "metadata DDL preparation ownership was superseded before apply".into(), + }), + WaitOutcome::TimedOut => Err(Error::Config { + detail: format!( + "metadata propose timed out after {:?} waiting for log index {} (current: {})", + timeout, + log_index, + watcher.current() + ), + }), + WaitOutcome::GroupGone => Err(Error::Config { + detail: "metadata group no longer hosted on this node".into(), + }), + } +} diff --git a/nodedb/src/control/metadata_proposer/ddl_prepare.rs b/nodedb/src/control/metadata_proposer/ddl_prepare.rs new file mode 100644 index 000000000..7f5310325 --- /dev/null +++ b/nodedb/src/control/metadata_proposer/ddl_prepare.rs @@ -0,0 +1,192 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The DDL preparation lease: the local lock and the replicated lease that +//! serialize descriptor preparation across the cluster. + +use std::sync::atomic::Ordering; +use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH}; + +use tokio::runtime::RuntimeFlavor; + +use nodedb_cluster::{METADATA_GROUP_ID, MetadataEntry, WaitOutcome, encode_entry}; + +use crate::control::state::SharedState; +use crate::error::Error; + +use super::handle::MetadataRaftHandle; +use super::timeouts::DEFAULT_PROPOSE_TIMEOUT; + +const DDL_PREPARE_LEASE: Duration = Duration::from_secs(60); +const DDL_PREPARE_WAIT: Duration = Duration::from_secs(70); + +fn wall_now_ns() -> u64 { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap_or_default() + .as_nanos() + .min(u64::MAX as u128) as u64 +} + +fn propose_metadata_and_wait( + shared: &SharedState, + handle: &dyn MetadataRaftHandle, + entry: &MetadataEntry, + timeout: Duration, +) -> Result { + let raw = encode_entry(entry).map_err(|e| Error::Config { + detail: format!("metadata entry encode: {e}"), + })?; + let index = handle.propose(raw)?; + let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); + let outcome = tokio::task::block_in_place(|| watcher.wait_for(index, timeout)); + match outcome { + WaitOutcome::Reached => Ok(index), + WaitOutcome::TimedOut => Err(Error::Config { + detail: format!( + "metadata propose timed out after {timeout:?} waiting for log index {index} (current: {})", + watcher.current() + ), + }), + WaitOutcome::GroupGone => Err(Error::Config { + detail: "metadata group no longer hosted on this node".into(), + }), + } +} + +/// RAII ownership of the metadata-Raft-serialized descriptor preparation lease. +/// The matching release is itself replicated, so another node cannot stamp from +/// the same prior catalog version until this guard is dropped and that release +/// has applied. +pub(crate) struct DdlPrepareGuard<'a> { + shared: &'a SharedState, + handle: &'a dyn MetadataRaftHandle, + token: u64, +} + +impl DdlPrepareGuard<'_> { + pub(crate) fn token(&self) -> u64 { + self.token + } +} + +impl Drop for DdlPrepareGuard<'_> { + fn drop(&mut self) { + if let Err(error) = propose_metadata_and_wait( + self.shared, + self.handle, + &MetadataEntry::DdlPrepareRelease { token: self.token }, + DEFAULT_PROPOSE_TIMEOUT, + ) { + tracing::error!(token = self.token, %error, "metadata DDL lease release failed"); + } + } +} + +pub(crate) fn acquire_ddl_prepare_lease<'a>( + shared: &'a SharedState, + handle: &'a dyn MetadataRaftHandle, +) -> Result, Error> { + let sequence = shared + .metadata_ddl_token_seq + .fetch_add(1, Ordering::Relaxed); + let token = shared.node_id.wrapping_mul(0x9e37_79b9_7f4a_7c15) + ^ wall_now_ns().rotate_left(17) + ^ sequence; + let deadline = Instant::now() + DDL_PREPARE_WAIT; + + loop { + propose_metadata_and_wait( + shared, + handle, + &MetadataEntry::DdlPrepareAcquire { token }, + DEFAULT_PROPOSE_TIMEOUT, + )?; + + loop { + let owner = *shared + .metadata_ddl_owner + .lock() + .map_err(|_| Error::Config { + detail: "metadata DDL owner lock poisoned".into(), + })?; + match owner { + Some((current, _)) if current == token => { + return Ok(DdlPrepareGuard { + shared, + handle, + token, + }); + } + Some((current, acquired_at)) + if shared.is_metadata_leader() + && acquired_at.elapsed() >= DDL_PREPARE_LEASE => + { + // Cancel the dead owner's pending record before releasing its + // lease, so it never lingers visible-but-unresolved past the lease. + if shared.pending_ddl.contains(current) { + propose_metadata_and_wait( + shared, + handle, + &MetadataEntry::DdlPendingCancel { token: current }, + DEFAULT_PROPOSE_TIMEOUT, + )?; + } + propose_metadata_and_wait( + shared, + handle, + &MetadataEntry::DdlPrepareRelease { token: current }, + DEFAULT_PROPOSE_TIMEOUT, + )?; + break; + } + None => break, + Some(_) if Instant::now() < deadline => { + // Reached from async tasks (ILP batch flush -> + // `propose_catalog_entry`), so hand the worker back to + // tokio rather than parking it: the lease owner this + // polls for is released by a raft apply that needs a + // worker to make progress. + tokio::task::block_in_place(|| { + std::thread::sleep(Duration::from_millis(10)); + }); + } + Some(_) => { + return Err(Error::Config { + detail: "metadata DDL preparation lease timed out".into(), + }); + } + } + } + } +} + +/// Take the local DDL preparation lock, handing the wait back to tokio when +/// the caller is on a multi-thread worker. +/// +/// The holder keeps this lock across the distributed preparation lease, the +/// descriptor drain and the local apply wait — each already wrapped in +/// `block_in_place`, but that only tells tokio about the waits *inside* the +/// lock, never about the wait *for* it. A bare `lock()` on a worker therefore +/// removes that worker from the runtime silently, including from the raft +/// apply work the current holder needs in order to finish, which turns +/// contention into a self-sustaining stall. +/// +/// `block_in_place` is a passthrough outside a multi-thread worker (plain sync +/// callers, blocking-pool threads) and panics on the current-thread runtime, +/// so it is applied only where it is both legal and meaningful — mirroring +/// `lease::drain_propose::poll_leases_drained`. +pub(super) fn lock_ddl_preparation( + shared: &SharedState, +) -> Result, Error> { + let acquire = || { + shared.metadata_ddl_lock.lock().map_err(|_| Error::Config { + detail: "metadata DDL preparation lock poisoned".into(), + }) + }; + match tokio::runtime::Handle::try_current() { + Ok(handle) if handle.runtime_flavor() == RuntimeFlavor::MultiThread => { + tokio::task::block_in_place(acquire) + } + _ => acquire(), + } +} diff --git a/nodedb/src/control/metadata_proposer/handle.rs b/nodedb/src/control/metadata_proposer/handle.rs new file mode 100644 index 000000000..48bf80222 --- /dev/null +++ b/nodedb/src/control/metadata_proposer/handle.rs @@ -0,0 +1,149 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The handle DDL proposes metadata entries through. + +use std::sync::{Arc, Weak}; + +use nodedb_cluster::ClusterError; +use nodedb_raft::RaftError; + +use crate::error::Error; + +/// Type-erased handle for proposing to the metadata raft group. +/// +/// The apply watermark for the metadata group lives on +/// [`crate::control::state::SharedState::applied_index_watcher`] (keyed by +/// [`nodedb_cluster::METADATA_GROUP_ID`]); callers of [`Self::propose`] +/// look it up there rather than receiving it through this handle. +pub trait MetadataRaftHandle: Send + Sync { + /// Propose a raw encoded `MetadataEntry` to the metadata group. + /// Returns its assigned log index on success. + fn propose(&self, bytes: Vec) -> Result; +} + +/// Concrete impl wrapping `nodedb_cluster::RaftLoop`. +/// +/// Holds the loop weakly: this handle lives on `SharedState`, which is +/// itself kept alive transitively by the `RaftLoop`, so a strong +/// reference here would close a cycle that pins both forever and blocks +/// clean shutdown. The loop is kept alive by its own spawned tasks; +/// `upgrade` therefore succeeds throughout normal operation and only +/// fails once the loop has been dropped on shutdown. +pub struct RaftLoopProposerHandle { + raft_loop: Weak< + nodedb_cluster::RaftLoop< + crate::control::cluster::SpscCommitApplier, + crate::control::LocalPlanExecutor, + >, + >, +} + +impl RaftLoopProposerHandle { + pub fn new( + raft_loop: Arc< + nodedb_cluster::RaftLoop< + crate::control::cluster::SpscCommitApplier, + crate::control::LocalPlanExecutor, + >, + >, + ) -> Self { + Self { + raft_loop: Arc::downgrade(&raft_loop), + } + } +} + +impl MetadataRaftHandle for RaftLoopProposerHandle { + fn propose(&self, bytes: Vec) -> Result { + // The cluster crate's `propose_to_metadata_group_via_leader` + // is async because it may need to forward to the metadata + // leader over QUIC. The trait method is sync because every + // caller (catalog DDL handlers, lease grant/release helpers) + // is itself sync but runs inside a tokio task. Wrap in + // `block_in_place` + the current runtime's `block_on` so the + // forwarding QUIC round-trip drives without starving the + // raft tick that produces the leader_hint. + // `upgrade` fails only once the raft loop has been dropped on + // shutdown; a request racing shutdown then fails cleanly with a + // typed error instead of panicking. + let raft_loop = self.raft_loop.upgrade().ok_or_else(|| Error::Config { + detail: "metadata propose: cluster not running".into(), + })?; + tokio::task::block_in_place(|| { + tokio::runtime::Handle::current() + .block_on(raft_loop.propose_to_metadata_group_via_leader(bytes)) + }) + .map_err(metadata_propose_error) + } +} + +/// The error a metadata proposal returns for a cluster error. +fn metadata_propose_error(error: ClusterError) -> Error { + match error { + // An election in progress is transient, not a failure of this + // proposal. Keep it typed rather than flattening it into a generic + // config error, so callers can wait the election out instead of + // failing the statement — a node that has just restarted answers + // every metadata proposal this way for a moment. + ClusterError::Raft(RaftError::NotLeader { leader_hint: None }) => { + Error::MetadataLeaderUnavailable + } + // A typed verdict keeps its class. + ClusterError::DataPlane { code } => Error::DataPlane(code.into()), + ClusterError::ShardExecution { error, .. } | ClusterError::StreamTerminal { error, .. } => { + Error::from(*error) + } + other @ (ClusterError::Raft( + RaftError::NotLeader { + leader_hint: Some(_), + } + | RaftError::LogCompacted { .. } + | RaftError::CompactionAheadOfApplied { .. } + | RaftError::ProposalRejected { .. } + | RaftError::InvalidTransferTarget { .. } + | RaftError::LeadershipTransferInProgress + | RaftError::GroupNotFound { .. } + | RaftError::Transport { .. } + | RaftError::Storage { .. } + | RaftError::Serialization { .. } + | RaftError::SnapshotFormat { .. } + | RaftError::Shutdown, + ) + | ClusterError::VShardNotMapped { .. } + | ClusterError::GroupNotFound { .. } + | ClusterError::LearnerNotCaughtUp { .. } + | ClusterError::MigrationInProgress { .. } + | ClusterError::MigrationPauseBudgetExceeded { .. } + | ClusterError::NodeUnreachable { .. } + | ClusterError::GhostNotFound { .. } + | ClusterError::Transport { .. } + | ClusterError::ShardTimeout { .. } + | ClusterError::Storage { .. } + | ClusterError::Codec { .. } + | ClusterError::UnsupportedWireVersion { .. } + | ClusterError::CircuitOpen { .. } + | ClusterError::JoinGroupDisappeared { .. } + | ClusterError::JoinCommitTimeout { .. } + | ClusterError::ReadIndexNotLeader { .. } + | ClusterError::ReadIndexTimeout { .. } + | ClusterError::Config { .. } + | ClusterError::MigrationCheckpoint(_) + | ClusterError::MigrationRecovery(_) + | ClusterError::WrongOwner { .. } + | ClusterError::Calvin(_) + | ClusterError::SnapshotCrcMismatch { .. } + | ClusterError::SnapshotOffsetRegression { .. } + | ClusterError::PartialSnapshotCorrupt { .. } + | ClusterError::PartialSnapshotCleanupFailed { .. } + | ClusterError::SnapshotApplyFailed { .. } + | ClusterError::Mirror(_) + | ClusterError::BspBarrier(_) + | ClusterError::VectorGather(_) + | ClusterError::SpatialGather(_) + | ClusterError::Bm25Gather(_) + | ClusterError::TsGather(_) + | ClusterError::RemoteUntyped { .. }) => Error::Config { + detail: format!("metadata propose: {other}"), + }, + } +} diff --git a/nodedb/src/control/metadata_proposer/mod.rs b/nodedb/src/control/metadata_proposer/mod.rs new file mode 100644 index 000000000..6a12b572a --- /dev/null +++ b/nodedb/src/control/metadata_proposer/mod.rs @@ -0,0 +1,38 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Synchronous `propose-and-wait-for-local-apply` helper for +//! replicated catalog DDL. +//! +//! The sole entry point pgwire DDL handlers use to write a +//! [`crate::control::catalog_entry::CatalogEntry`] through the metadata raft group (group 0). It is +//! deliberately sync — pgwire DDL handlers are not async, and +//! `tokio::task::block_in_place`-style wrapping keeps the blocking +//! wait from starving the tokio runtime. +//! +//! Semantics: +//! +//! 1. If no cluster is configured (`shared.metadata_raft` not +//! installed), returns `ProposeOutcome::LocalOnly`. The caller's +//! single-node direct-write path stays authoritative. +//! 2. If this node is the metadata-group leader, proposes the +//! entry, blocks until its local applied watermark reaches the +//! assigned log index (5s default timeout), and returns the +//! log index on success. +//! 3. If this node is NOT the leader, returns +//! `Error::Config { detail: "metadata propose: not leader ..." }`. +//! Gateway-side redirection will make this transparent. + +pub mod catalog; +pub mod ddl_prepare; +pub mod handle; +pub mod replicated_entries; +pub mod timeouts; + +pub use catalog::{propose_catalog_entry, propose_catalog_entry_with_timeout}; +pub(crate) use ddl_prepare::{DdlPrepareGuard, acquire_ddl_prepare_lease}; +pub use handle::{MetadataRaftHandle, RaftLoopProposerHandle}; +pub use replicated_entries::{ + propose_surrogate_hwm, propose_surrogate_reserve, propose_sync_peer_bind, + propose_sync_producer_fence, propose_sync_producer_register, +}; +pub use timeouts::{DEFAULT_DRAIN_TIMEOUT, DEFAULT_PROPOSE_TIMEOUT}; diff --git a/nodedb/src/control/metadata_proposer/replicated_entries.rs b/nodedb/src/control/metadata_proposer/replicated_entries.rs new file mode 100644 index 000000000..f45af8b4e --- /dev/null +++ b/nodedb/src/control/metadata_proposer/replicated_entries.rs @@ -0,0 +1,195 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Metadata entries that replicate node-local registries: surrogate +//! allocation and Lite sync producers. +//! +//! Each one proposes the entry and waits for its commit on this node. In +//! single-node mode (no `metadata_raft` installed) each returns `Ok(0)`: the +//! local write already persisted the state. + +use nodedb_cluster::{METADATA_GROUP_ID, MetadataEntry, encode_entry}; + +use crate::control::state::SharedState; +use crate::error::Error; + +use super::timeouts::DEFAULT_PROPOSE_TIMEOUT; + +/// Propose `entry` and wait until this node reaches its log index. `label` +/// names the entry in the errors. Returns `Ok(0)` when no cluster runs. +fn propose_and_wait( + shared: &SharedState, + entry: &MetadataEntry, + label: &str, +) -> Result { + let Some(handle) = shared.metadata_raft.get() else { + return Ok(0); + }; + let raw = encode_entry(entry).map_err(|e| Error::Config { + detail: format!("{label} encode: {e}"), + })?; + + let log_index = handle.propose(raw)?; + + let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); + let outcome = + tokio::task::block_in_place(|| watcher.wait_for(log_index, DEFAULT_PROPOSE_TIMEOUT)); + if !outcome.is_reached() { + return Err(Error::Config { + detail: format!("{label} propose timed out waiting for log index {log_index}"), + }); + } + + Ok(log_index) +} + +/// Propose a surrogate high-watermark advance to the metadata Raft group +/// and wait for it to be applied locally. +/// +/// In single-node / no-cluster mode (no `metadata_raft` installed), +/// returns `Ok(0)` immediately — the WAL-only path on `SharedState` is +/// still sufficient. In cluster mode this is called by the leader-side +/// flush path instead of (or in addition to) the local WAL record, so +/// every follower's `SurrogateRegistry` advances to the same hwm via the +/// Raft commit. +/// +/// `hwm` is the highest surrogate that has been issued so far on this +/// node. Followers apply the entry by calling +/// `SurrogateRegistry::restore_hwm(hwm)` (idempotent, monotonic). +pub fn propose_surrogate_hwm(shared: &SharedState, hwm: u32) -> Result { + propose_and_wait( + shared, + &MetadataEntry::SurrogateAlloc { hwm }, + "surrogate_alloc", + ) +} + +/// Propose a HiLo surrogate batch reservation to the metadata Raft group +/// and wait for the commit (returns the assigned log index). +/// +/// In single-node / no-cluster mode (no `metadata_raft` installed), +/// returns `Ok(0)` immediately — single-node uses the local `alloc_one` +/// path and never reaches here. Kept as a safety guard only. +/// +/// The carved `[start, end)` range is NOT decided here: it is computed +/// at apply time on every node by advancing the global watermark in +/// identical log order (see `MetadataEntry::SurrogateReserve`). The +/// caller therefore cannot learn the range from this commit-wait alone +/// — `wait_for` returns on COMMIT, before the apply handler runs. The +/// owning node's apply handler fires an explicit completion signal +/// (`SurrogateAssigner::complete_reservation`) that the caller awaits +/// separately to learn the range. +/// +/// `node_id` + `request_id` identify this node's specific in-flight +/// reservation so the apply handler routes the batch + signal back to it. +pub fn propose_surrogate_reserve( + shared: &SharedState, + node_id: u64, + request_id: u64, + batch_size: u32, +) -> Result { + propose_and_wait( + shared, + &MetadataEntry::SurrogateReserve { + node_id, + request_id, + batch_size, + }, + "surrogate_reserve", + ) +} + +/// Propose a Lite client registration through the metadata Raft group and +/// wait for it to be applied locally. +/// +/// In single-node / no-cluster mode (no `metadata_raft` installed), +/// returns `Ok(0)` immediately — the local registry write already persisted +/// the state. In cluster mode every follower applies the entry via +/// `SyncProducerRegistry::apply_register` so the `(producer_id, epoch)` pair +/// agrees on all nodes and survives leader failover. +pub fn propose_sync_producer_register( + shared: &SharedState, + lite_id: &str, + producer_id: u64, + tenant_id: u64, + user_id: u64, + epoch: u64, + created_ms: i64, +) -> Result { + propose_and_wait( + shared, + &MetadataEntry::SyncProducerRegister { + lite_id: lite_id.to_owned(), + producer_id, + tenant_id, + user_id, + epoch, + created_ms, + }, + "sync_producer_register", + ) +} + +/// Propose a Lite client epoch fence through the metadata Raft group and +/// wait for it to be applied locally. +/// +/// In single-node / no-cluster mode (no `metadata_raft` installed), +/// returns `Ok(0)` immediately — the local registry write already persisted +/// the state. In cluster mode every follower applies the entry via +/// `SyncProducerRegistry::apply_fence` (max-wins) so the epoch advance +/// survives leader failover. +pub fn propose_sync_producer_fence( + shared: &SharedState, + lite_id: &str, + new_epoch: u64, +) -> Result { + propose_and_wait( + shared, + &MetadataEntry::SyncProducerFence { + lite_id: lite_id.to_owned(), + new_epoch, + }, + "sync_producer_fence", + ) +} + +/// Propose ownership of one Loro peer id through the metadata Raft group and +/// wait for it to be applied locally. +/// +/// In single-node / no-cluster mode (no `metadata_raft` installed), returns +/// `Ok(0)` immediately — the local registry write already persisted the +/// ownership. In cluster mode the caller must re-read the owner after this +/// returns: the apply is lowest-producer-id-wins, so a node that lost a race it +/// did not know it was in learns the real owner only once the entry lands. +pub fn propose_sync_peer_bind( + shared: &SharedState, + binding: &crate::control::security::catalog::sync_producer::PeerBindingKey, + producer_id: u64, + bound_ms: i64, +) -> Result { + propose_and_wait( + shared, + &MetadataEntry::SyncPeerBind { + database_id: binding.database_id, + tenant_id: binding.tenant_id, + collection: binding.collection.clone(), + peer_id: binding.peer_id, + producer_id, + bound_ms, + }, + "sync_peer_bind", + ) +} + +#[cfg(test)] +mod tests { + use std::time::Duration; + + use nodedb_cluster::AppliedIndexWatcher; + + #[test] + fn watcher_helper_returns_reached_on_past_target() { + let w = AppliedIndexWatcher::new(); + w.bump(10); + assert!(w.wait_for(5, Duration::from_millis(1)).is_reached()); + } +} diff --git a/nodedb/src/control/metadata_proposer/timeouts.rs b/nodedb/src/control/metadata_proposer/timeouts.rs new file mode 100644 index 000000000..c1fe63ddb --- /dev/null +++ b/nodedb/src/control/metadata_proposer/timeouts.rs @@ -0,0 +1,21 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! How long a metadata proposal and a DDL drain wait. + +use std::time::Duration; + +/// Default upper bound on how long a single +/// `propose_catalog_entry` call will block before returning an +/// error. +pub const DEFAULT_PROPOSE_TIMEOUT: Duration = Duration::from_secs(5); + +/// Default upper bound on how long a DDL drain will wait for +/// prior-version leases to release before giving up. Must be at +/// least `ClusterTransportTuning::descriptor_lease_duration_secs` +/// so an existing lease gets at least one full lifetime to +/// expire naturally. 35 seconds matches the 300s lease duration +/// plus a 30-second grace minus the typical 5-minute default +/// cut down for test budget — in production +/// `propose_catalog_entry_with_drain_timeout` can pass a longer +/// value if an operator is willing to wait. +pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(35); diff --git a/nodedb/src/control/metrics/mod.rs b/nodedb/src/control/metrics/mod.rs index d09e47642..5bd4a42b4 100644 --- a/nodedb/src/control/metrics/mod.rs +++ b/nodedb/src/control/metrics/mod.rs @@ -12,5 +12,5 @@ pub use database::{DatabaseCounters, DatabaseMetricsRegistry, DatabaseQuotaMetri pub use histogram::AtomicHistogram; pub use per_vshard::{PerVShardMetrics, PerVShardMetricsRegistry, VShardStatsSnapshot}; pub use purge::PurgeMetrics; -pub use system::{CoreHeartbeats, SystemMetrics}; +pub use system::{CoreFailStopReport, CoreFailStops, CoreHeartbeats, SystemMetrics}; pub use tenant::TenantQuotaMetrics; diff --git a/nodedb/src/control/metrics/prometheus/engines.rs b/nodedb/src/control/metrics/prometheus/engines.rs index bed82ea04..b4b72aa55 100644 --- a/nodedb/src/control/metrics/prometheus/engines.rs +++ b/nodedb/src/control/metrics/prometheus/engines.rs @@ -31,6 +31,36 @@ impl SystemMetrics { "Vectors stored", self.vector_vectors_stored.load(Ordering::Relaxed), ); + counter( + out, + "nodedb_vector_builds_started_total", + "HNSW builds sent to a builder thread", + self.vector_builds_started.load(Ordering::Relaxed), + ); + counter( + out, + "nodedb_vector_builds_completed_total", + "HNSW builds installed", + self.vector_builds_completed.load(Ordering::Relaxed), + ); + counter( + out, + "nodedb_vector_builds_failed_total", + "HNSW builds that failed", + self.vector_builds_failed.load(Ordering::Relaxed), + ); + counter( + out, + "nodedb_vector_builds_deferred_total", + "Times a full builder queue kept an HNSW build waiting", + self.vector_builds_deferred.load(Ordering::Relaxed), + ); + gauge( + out, + "nodedb_vector_build_pending", + "HNSW builds waiting for or running on a builder", + self.vector_build_pending.load(Ordering::Relaxed), + ); self.vector_query_seconds.write_prometheus( out, "nodedb_vector_query_seconds", diff --git a/nodedb/src/control/metrics/system/core_fail_stop.rs b/nodedb/src/control/metrics/system/core_fail_stop.rs new file mode 100644 index 000000000..2c3d19039 --- /dev/null +++ b/nodedb/src/control/metrics/system/core_fail_stop.rs @@ -0,0 +1,136 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Node-wide record of Data Plane cores that fail-stopped. +//! +//! A core fail-stops when its state is unknown: a rollback failed part way, +//! or the work owed after a committed record's install failed. Such a core +//! refuses every request until restart. Its latch lives on the core. This +//! record is the node-wide view `/healthz`, the native `STATUS` and the +//! `nodedb_data_plane_core_fail_stopped` gauge read. +//! +//! The first report wins, like the Calvin halt and metadata-apply wedge +//! markers: a fail-stopped core stays stopped until restart, so a later +//! report cannot replace the cause an operator must act on. The gauge counts +//! every stopped core. + +use std::sync::OnceLock; +use std::sync::atomic::{AtomicU64, Ordering}; + +/// Why the first Data Plane core fail-stopped. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct CoreFailStopReport { + pub core_id: usize, + /// The cause label, as the core's ERROR line names it. + pub cause: &'static str, + pub detail: String, +} + +/// First-report-wins record of fail-stopped Data Plane cores. Never clears. +#[derive(Debug, Default)] +pub struct CoreFailStops { + first: OnceLock, + stopped: AtomicU64, +} + +impl CoreFailStops { + /// Record one core that fail-stopped. A core reports once. + pub fn record(&self, report: CoreFailStopReport) { + self.stopped.fetch_add(1, Ordering::Relaxed); + let _ = self.first.set(report); + } + + /// The first core that fail-stopped, if one did. + pub fn report(&self) -> Option<&CoreFailStopReport> { + self.first.get() + } + + pub fn is_stopped(&self) -> bool { + self.first.get().is_some() + } + + /// Number of cores that fail-stopped. + pub fn stopped_cores(&self) -> u64 { + self.stopped.load(Ordering::Relaxed) + } + + /// Append the `nodedb_data_plane_core_fail_stopped` gauge. + pub fn write_prometheus(&self, out: &mut String) { + use std::fmt::Write as _; + let _ = writeln!( + out, + "# HELP nodedb_data_plane_core_fail_stopped Data Plane cores that stopped \ + serving because their state is unknown\n\ + # TYPE nodedb_data_plane_core_fail_stopped gauge\n\ + nodedb_data_plane_core_fail_stopped {}", + self.stopped_cores() + ); + } +} + +/// Readiness-probe rendering for a node with a fail-stopped core: `503`, +/// degraded. The other cores keep serving. +pub fn to_http_response( + report: &CoreFailStopReport, + stopped_cores: u64, +) -> (axum::http::StatusCode, serde_json::Value) { + ( + axum::http::StatusCode::SERVICE_UNAVAILABLE, + serde_json::json!({ + "status": "degraded", + "reason": "data_plane_core_fail_stopped", + "core_id": report.core_id, + "cause": report.cause, + "error": report.detail, + "stopped_cores": stopped_cores, + }), + ) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn report(core_id: usize) -> CoreFailStopReport { + CoreFailStopReport { + core_id, + cause: "rollback_failed", + detail: "undo entry 3 failed".into(), + } + } + + #[test] + fn a_fresh_record_reports_no_stopped_core() { + let stops = CoreFailStops::default(); + assert!(!stops.is_stopped()); + assert_eq!(stops.stopped_cores(), 0); + } + + #[test] + fn the_first_report_wins_and_every_report_is_counted() { + let stops = CoreFailStops::default(); + stops.record(report(2)); + stops.record(report(5)); + assert_eq!(stops.report().map(|r| r.core_id), Some(2)); + assert_eq!(stops.stopped_cores(), 2); + } + + #[test] + fn the_gauge_counts_stopped_cores() { + let stops = CoreFailStops::default(); + stops.record(report(1)); + let mut out = String::new(); + stops.write_prometheus(&mut out); + assert!( + out.contains("nodedb_data_plane_core_fail_stopped 1"), + "{out}" + ); + } + + #[test] + fn the_readiness_body_names_the_core_and_cause() { + let (status, body) = to_http_response(&report(4), 1); + assert_eq!(status, axum::http::StatusCode::SERVICE_UNAVAILABLE); + assert_eq!(body["core_id"], serde_json::json!(4)); + assert_eq!(body["cause"], serde_json::json!("rollback_failed")); + } +} diff --git a/nodedb/src/control/metrics/system/fields.rs b/nodedb/src/control/metrics/system/fields.rs index 32f854f0d..a21da10da 100644 --- a/nodedb/src/control/metrics/system/fields.rs +++ b/nodedb/src/control/metrics/system/fields.rs @@ -8,6 +8,7 @@ use std::sync::{Arc, RwLock}; use super::super::histogram::{AtomicHistogram, WAL_FSYNC_BUCKETS_US}; use super::super::purge::PurgeMetrics; +use super::core_fail_stop::CoreFailStops; use super::heartbeat::CoreHeartbeats; use crate::data::executor::core_loop::pressure::ThrottleMetrics; use crate::data::io::IoMetrics; @@ -62,6 +63,16 @@ pub struct SystemMetrics { pub vector_collections: AtomicU64, pub vector_vectors_stored: AtomicU64, pub vector_query_seconds: AtomicHistogram, + /// HNSW builds sent to a builder thread. + pub vector_builds_started: AtomicU64, + /// HNSW builds installed on their core. + pub vector_builds_completed: AtomicU64, + /// HNSW builds that failed or could not be read. + pub vector_builds_failed: AtomicU64, + /// Times a core found its builder queue full and kept the job waiting. + pub vector_builds_deferred: AtomicU64, + /// HNSW builds waiting for or running on a builder, across all cores. + pub vector_build_pending: AtomicU64, pub graph_traversals: AtomicU64, pub graph_nodes: AtomicU64, @@ -219,6 +230,9 @@ pub struct SystemMetrics { /// that stops advancing is the only evidence a core has stopped /// completing iterations. pub core_heartbeats: CoreHeartbeats, + /// Cores that fail-stopped because their state is unknown. A core records + /// itself here once, as it stops. + pub core_fail_stops: CoreFailStops, } impl SystemMetrics { diff --git a/nodedb/src/control/metrics/system/mod.rs b/nodedb/src/control/metrics/system/mod.rs index e40533ba8..2fabc642f 100644 --- a/nodedb/src/control/metrics/system/mod.rs +++ b/nodedb/src/control/metrics/system/mod.rs @@ -1,9 +1,11 @@ // SPDX-License-Identifier: BUSL-1.1 +pub mod core_fail_stop; mod fields; mod heartbeat; mod record; mod render; +pub use core_fail_stop::{CoreFailStopReport, CoreFailStops}; pub use fields::SystemMetrics; pub use heartbeat::CoreHeartbeats; diff --git a/nodedb/src/control/metrics/system/record.rs b/nodedb/src/control/metrics/system/record.rs index e2eb7efe3..bb8d2d65e 100644 --- a/nodedb/src/control/metrics/system/record.rs +++ b/nodedb/src/control/metrics/system/record.rs @@ -172,6 +172,34 @@ impl SystemMetrics { self.vector_query_seconds.observe(latency_us); } + pub fn record_vector_build_started(&self) { + self.vector_builds_started.fetch_add(1, Ordering::Relaxed); + } + + pub fn record_vector_build_completed(&self) { + self.vector_builds_completed.fetch_add(1, Ordering::Relaxed); + } + + pub fn record_vector_build_failed(&self) { + self.vector_builds_failed.fetch_add(1, Ordering::Relaxed); + } + + pub fn record_vector_build_deferred(&self) { + self.vector_builds_deferred.fetch_add(1, Ordering::Relaxed); + } + + /// Move the cross-core pending-build gauge from one core's previous + /// count `before` to its current count `after`. + pub fn move_vector_build_pending(&self, before: u64, after: u64) { + if after > before { + self.vector_build_pending + .fetch_add(after - before, Ordering::Relaxed); + } else if before > after { + self.vector_build_pending + .fetch_sub(before - after, Ordering::Relaxed); + } + } + pub fn update_vector_stats(&self, collections: u64, vectors: u64) { self.vector_collections .store(collections, Ordering::Relaxed); diff --git a/nodedb/src/control/metrics/system/render.rs b/nodedb/src/control/metrics/system/render.rs index 0c233fd73..8a413e13f 100644 --- a/nodedb/src/control/metrics/system/render.rs +++ b/nodedb/src/control/metrics/system/render.rs @@ -18,6 +18,7 @@ impl SystemMetrics { self.purge.write_prometheus(&mut out); self.io_metrics.write_prometheus(&mut out); self.spsc_throttle.write_prometheus(&mut out); + self.core_fail_stops.write_prometheus(&mut out); out } diff --git a/nodedb/src/control/mod.rs b/nodedb/src/control/mod.rs index fb48758ae..697cdc12b 100644 --- a/nodedb/src/control/mod.rs +++ b/nodedb/src/control/mod.rs @@ -22,9 +22,12 @@ pub mod distributed_applier; pub mod event_action_error; pub mod event_trigger; pub mod exec_receiver; +#[cfg(feature = "failpoints")] +pub(crate) mod fail_gate; pub mod gateway; pub mod insert_select; pub mod lease; +pub mod local_dispatch; pub mod lock_utils; pub mod maintenance; pub mod merge_orchestrator; @@ -64,7 +67,7 @@ pub mod wal_replication; pub mod write_resolve; pub use exec_receiver::LocalPlanExecutor; -pub use request_tracker::RequestTracker; +pub use request_tracker::{RequestTracker, ResponseReceiver}; pub use rolling_upgrade::ClusterVersionView; pub use state::SharedState; pub use wal_replication::{DistributedApplier, ProposeTracker, create_distributed_applier}; diff --git a/nodedb/src/control/orchestrated_write.rs b/nodedb/src/control/orchestrated_write.rs index bd3987632..c618c76e7 100644 --- a/nodedb/src/control/orchestrated_write.rs +++ b/nodedb/src/control/orchestrated_write.rs @@ -20,7 +20,7 @@ use crate::control::state::SharedState; use crate::control::wal_replication::{ ReplicableWrite, propose_replicated_entry, to_replicated_entry, }; -use crate::types::{RequestId, VShardId}; +use crate::types::RequestId; /// Apply `plan`, a resolved write on `collection`, and return the Data-Plane /// response the statement renders. @@ -43,7 +43,11 @@ pub(crate) async fn apply_orchestrated_write( // WAL-only restart rebuilds the index from pre-write records. No-op // on a target with no write-set. crate::control::server::wal_dispatch::mint_dispatch_local_redo( - &state.wal, + state + .wal + .appender(crate::wal::manager::NO_APPLY_KEY) + // `dispatch_local` runs the write as a client write. + .with_event_source(crate::event::EventSource::User), tenant_id, database_id, collection, @@ -52,7 +56,9 @@ pub(crate) async fn apply_orchestrated_write( return Ok(resp); }; - let vshard_id = VShardId::from_collection_in_database(database_id, collection); + // `collection` is the plan's database-qualified name. + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?.vshard(); let replicable = ReplicableWrite::decide_for_replication(&plan)?; let entry = to_replicated_entry(tenant_id, database_id, vshard_id, &replicable)?.ok_or_else(|| { diff --git a/nodedb/src/control/otel/receiver.rs b/nodedb/src/control/otel/receiver.rs index bf16480bc..2cb8b75d3 100644 --- a/nodedb/src/control/otel/receiver.rs +++ b/nodedb/src/control/otel/receiver.rs @@ -428,7 +428,7 @@ pub(super) async fn authenticate_otel( crate::control::security::jwt_policy::enforce_stateful_jwt_policy( shared, verified.claims(), - identity.tenant_id, + &identity, ) .map_err(|_| "invalid bearer token".to_owned())?; return admit_transport(shared, identity, peer_addr); diff --git a/nodedb/src/control/planner/auto_tier.rs b/nodedb/src/control/planner/auto_tier.rs index eb23c9429..76c85bee2 100644 --- a/nodedb/src/control/planner/auto_tier.rs +++ b/nodedb/src/control/planner/auto_tier.rs @@ -14,7 +14,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::engine::timeseries::retention_policy::RetentionPolicyDef; -use crate::types::{DatabaseId, TenantId, VShardId}; +use crate::types::{DatabaseId, TenantId}; use nodedb_physical::physical_plan::TimeseriesOp; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; @@ -229,7 +229,7 @@ fn build_scan_task( } = scope; PhysicalTask { tenant_id, - vshard_id: VShardId::from_collection_in_database(database_id, collection), + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(), database_id, plan: PhysicalPlan::Timeseries(TimeseriesOp::Scan { collection: nodedb_types::QualifiedCollection::new(database_id, collection), diff --git a/nodedb/src/control/planner/calvin/dependent_recon.rs b/nodedb/src/control/planner/calvin/dependent_recon.rs index e29fa429d..4ff3ae86c 100644 --- a/nodedb/src/control/planner/calvin/dependent_recon.rs +++ b/nodedb/src/control/planner/calvin/dependent_recon.rs @@ -400,6 +400,20 @@ async fn dispatch_dependent_edge_recon_inner( } }; + // A write to a permission-tree source is acknowledged only once it binds + // every node. Tree sources live in the default database. + let sources = state.authorization_fence.sources(); + let binds_authorization = database_id == crate::types::DatabaseId::DEFAULT + && tasks.iter().any(|task| { + task.plan + .named_collections() + .iter() + .any(|collection| sources.is_source_collection(collection)) + }); + if binds_authorization { + crate::control::security::auth_lease::calvin_write_barrier(state).await?; + } + // Completion fired: the scheduler deposited the applied Response (with any // RETURNING rows) into the sidecar before proposing the ack that woke the // retry loop, so the entry is present now if this write carried RETURNING. @@ -407,12 +421,18 @@ async fn dispatch_dependent_edge_recon_inner( // `Conflict` (>1 RETURNING participant) fails loudly rather than returning a // partial cross-shard union. let drained = state - .calvin_apply_results + .calvin + .apply_results .lock() .unwrap_or_else(|p| p.into_inner()) .remove(&completed_txn); let apply_result = match drained { - Some(CalvinApplyResult::Single { response, .. }) => Some(response), + Some(CalvinApplyResult::Single { response, .. }) => { + // An installed txn whose reply failed to render deposits it as an + // error for the statement. + crate::control::local_dispatch::reject_data_plane_error(&response)?; + Some(response) + } Some(CalvinApplyResult::Conflict) => { return Err(Error::Internal { detail: "multi-participant cross-shard RETURNING not supported".to_owned(), diff --git a/nodedb/src/control/planner/calvin/dispatch.rs b/nodedb/src/control/planner/calvin/dispatch.rs index e29bc9145..f365ce63e 100644 --- a/nodedb/src/control/planner/calvin/dispatch.rs +++ b/nodedb/src/control/planner/calvin/dispatch.rs @@ -49,6 +49,7 @@ use crate::control::server::shared::session::read_set::ReadSetEntry; use crate::types::VShardId; use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; use nodedb_physical::physical_task::PhysicalTask; +use nodedb_types::CollectionKey; pub use crate::control::planner::calvin::predicate::predicate_class; pub use crate::control::planner::calvin::write_class::is_write_plan; @@ -77,14 +78,22 @@ pub fn is_dependent_predicate(plan: &PhysicalPlan) -> bool { /// /// Each [`ReadSetEntry`] homes to its collection's vShard using the SAME /// collection→vShard map `ReadWriteSet::participating_vshards` uses to derive the -/// `TxClass` read_set's participants. Each read retains its session database so -/// classification and the database-scoped transaction class agree. A read with -/// no extractable collection contributes nothing. -pub fn read_vshards_of(reads: &[ReadSetEntry]) -> BTreeSet { +/// `TxClass` read_set's participants. An entry carries the plan's +/// database-qualified name, so it is de-qualified into a [`CollectionKey`] +/// before hashing. Each read retains its session database so classification +/// and the database-scoped transaction class agree. A read with no extractable +/// collection contributes nothing. +pub fn read_vshards_of(reads: &[ReadSetEntry]) -> crate::Result> { reads .iter() .filter(|e| !e.collection.is_empty()) - .map(|e| VShardId::from_collection_in_database(e.database_id, &e.collection).as_u32()) + .map(|e| { + Ok( + CollectionKey::from_qualified_str(e.database_id, &e.collection)? + .vshard() + .as_u32(), + ) + }) .collect() } @@ -160,7 +169,7 @@ pub(crate) async fn dispatch_calvin_or_fast( // Interactive COMMIT threads its session read-set here; autocommit passes an // empty slice. The read vShards widen both the classification (below) and the // TxClass read_set participants (in `build_static_tx_class`) in lockstep. - let read_vshards = read_vshards_of(reads); + let read_vshards = read_vshards_of(reads)?; let class = classify_dispatch(tasks, &read_vshards); match &class { @@ -454,7 +463,9 @@ mod tests { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("dispatch_home_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); match first { Some((ref fname, fv)) if fv != vshard => return (fname.clone(), name), None => first = Some((name, vshard)), @@ -484,10 +495,12 @@ mod tests { // cross-node transaction. This test guarantees a future refactor of // `read_vshards_of` / `classify_dispatch` cannot reopen that hole. let (write_coll, read_coll) = two_distinct_vshard_collections(); - let write_vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, &write_coll).as_u32(); - let read_vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, &read_coll).as_u32(); + let write_vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &write_coll) + .vshard() + .as_u32(); + let read_vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &read_coll) + .vshard() + .as_u32(); let tasks = vec![doc_insert_task(write_vshard)]; @@ -504,7 +517,8 @@ mod tests { // The homing step under test: a foreign-collection read must home to a // vShard distinct from the write's, contributing a new participant. - let read_vshards = read_vshards_of(std::slice::from_ref(&read_entry)); + let read_vshards = + read_vshards_of(std::slice::from_ref(&read_entry)).expect("read vshards"); assert!( read_vshards.contains(&read_vshard) && !read_vshards.contains(&write_vshard), "read entry for `{read_coll}` must home to vShard {read_vshard}, not the write's {write_vshard}" diff --git a/nodedb/src/control/planner/calvin/dispatch_multi.rs b/nodedb/src/control/planner/calvin/dispatch_multi.rs index 624443634..c9ba806a3 100644 --- a/nodedb/src/control/planner/calvin/dispatch_multi.rs +++ b/nodedb/src/control/planner/calvin/dispatch_multi.rs @@ -162,7 +162,7 @@ pub(crate) async fn dispatch_tasks_to_calvin( reads: &[ReadSetEntry], lock_owner: Option, ) -> crate::Result> { - let read_vshards = read_vshards_of(reads); + let read_vshards = read_vshards_of(reads)?; match classify_dispatch(tasks, &read_vshards) { DispatchClass::MultiShard { .. } => { admit_legacy_multi_shard_dispatch(cross_shard_mode, position)?; @@ -186,12 +186,12 @@ mod tests { use nodedb_cluster::calvin::types::TxnIdWire; use nodedb_physical::physical_plan::{DocumentOp, GraphOp, PhysicalPlan}; use nodedb_physical::physical_task::PostSetOp; - use nodedb_types::Surrogate; + use nodedb_types::{CollectionKey, Surrogate}; fn task(collection: &str, surrogate: u32) -> PhysicalTask { PhysicalTask { tenant_id: TenantId::new(1), - vshard_id: VShardId::from_collection_in_database(DatabaseId::DEFAULT, collection), + vshard_id: CollectionKey::from_bare(DatabaseId::DEFAULT, collection).vshard(), database_id: DatabaseId::DEFAULT, plan: PhysicalPlan::Document(DocumentOp::PointInsert { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, collection), @@ -225,11 +225,11 @@ mod tests { } fn distinct_collection(from: &str) -> String { - let home = VShardId::from_collection_in_database(DatabaseId::DEFAULT, from); + let home = CollectionKey::from_bare(DatabaseId::DEFAULT, from).vshard(); (0..1024) .map(|index| format!("atomic_{index}")) .find(|candidate| { - VShardId::from_collection_in_database(DatabaseId::DEFAULT, candidate) != home + CollectionKey::from_bare(DatabaseId::DEFAULT, candidate).vshard() != home }) .expect("test routing domain must contain more than one vShard") } diff --git a/nodedb/src/control/planner/calvin/preexec.rs b/nodedb/src/control/planner/calvin/preexec.rs index 2922da5ef..d1a30a067 100644 --- a/nodedb/src/control/planner/calvin/preexec.rs +++ b/nodedb/src/control/planner/calvin/preexec.rs @@ -21,7 +21,7 @@ use nodedb_types::TenantId; use crate::control::server::dispatch_utils::dispatch_to_data_plane; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TraceId, VShardId}; +use crate::types::{DatabaseId, TraceId}; use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; /// One implicit graph edge surfaced from the pre-execution reconnaissance scan. @@ -84,7 +84,7 @@ pub async fn run_preexec_scan( collection: &str, filter_bytes: Vec, ) -> crate::Result { - let vshard_id = VShardId::from_collection_in_database(database_id, collection); + let vshard_id = nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(); let scan_plan = PhysicalPlan::Document(DocumentOp::Scan { collection: nodedb_types::QualifiedCollection::new(database_id, collection), @@ -128,13 +128,8 @@ pub async fn run_preexec_scan( database_id, txn_id: None, }; - let payloads = gateway - .execute_internal(&gw_ctx, scan_plan) - .await - .map_err(|e| crate::Error::Storage { - engine: "preexec-scan".into(), - detail: format!("pre-execution scan failed: {e}"), - })?; + // A shard verdict keeps its own typed error. + let payloads = gateway.execute_internal(&gw_ctx, scan_plan).await?; // A single-collection scan routes to one vshard → one payload. An // absent payload means zero matching rows. let payload = payloads.into_iter().next().unwrap_or_default(); @@ -151,16 +146,17 @@ pub async fn run_preexec_scan( ) .await?; - // A shard verdict keeps its own typed error, so a scan the statement's - // deadline cut short reports the deadline rather than a storage fault. - crate::control::server::dispatch_utils::reject_data_plane_error(&response)?; - if response.status != crate::bridge::envelope::Status::Ok { - return Err(crate::Error::Storage { - engine: "preexec-scan".into(), - detail: format!("pre-execution scan failed: {:?}", response.error_code), - }); - } + scan_from_response(&response) +} +/// Decode a local scan response. +/// +/// A shard verdict keeps its own typed error, so a scan the statement's +/// deadline cut short reports the deadline rather than a storage fault. +/// `reject_data_plane_error` passes only a `NotFound` refusal. Its payload is +/// empty, so it decodes as no matches, the answer the gateway path gives. +fn scan_from_response(response: &crate::bridge::envelope::Response) -> crate::Result { + crate::control::local_dispatch::reject_data_plane_error(response)?; Ok(decode_scan(&response.payload)) } @@ -276,6 +272,46 @@ fn decode_scan_json(json_str: &str) -> PreexecScan { #[cfg(test)] mod tests { use super::*; + use crate::bridge::envelope::{ErrorCode, Payload, Response, Status}; + use crate::types::{Lsn, RequestId}; + + fn refusal(code: ErrorCode) -> Response { + Response { + request_id: RequestId::new(1), + status: Status::Error, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: Some(Box::new(code)), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + /// A refused scan keeps its code, never a storage error. + #[test] + fn a_refused_scan_keeps_its_code() { + let code = ErrorCode::Unsupported { + detail: "not on this engine".into(), + }; + match scan_from_response(&refusal(code.clone())) { + Err(crate::Error::DataPlane(kept)) => assert_eq!(kept, code), + Err(other) => panic!("expected the typed refusal, got {other:?}"), + Ok(_) => panic!("a refused scan must fail"), + } + } + + /// A `NotFound` refusal means the shard holds no slice of the collection, + /// so the scan matched nothing. + #[test] + fn a_not_found_scan_matches_nothing() { + let scan = scan_from_response(&refusal(ErrorCode::NotFound)) + .expect("a NotFound refusal reads as an empty scan"); + assert!(scan.surrogates.is_empty()); + assert!(scan.edges.is_empty()); + } #[test] fn decode_empty_payload_returns_empty() { diff --git a/nodedb/src/control/planner/calvin/reservation.rs b/nodedb/src/control/planner/calvin/reservation.rs index 47aef4d2b..0a321083f 100644 --- a/nodedb/src/control/planner/calvin/reservation.rs +++ b/nodedb/src/control/planner/calvin/reservation.rs @@ -27,7 +27,7 @@ use nodedb_cluster::{ }; use crate::Error; -use crate::control::server::exchange::resolve::register_peers_from_topology; +use crate::control::cluster::warm_peers::register_peers_from_topology; use crate::control::state::SharedState; /// Submit a reserve-read to THIS node's reservation inbox and await the diff --git a/nodedb/src/control/planner/calvin/submit/assign.rs b/nodedb/src/control/planner/calvin/submit/assign.rs index ed2d34d76..49d414d9e 100644 --- a/nodedb/src/control/planner/calvin/submit/assign.rs +++ b/nodedb/src/control/planner/calvin/submit/assign.rs @@ -13,7 +13,7 @@ use nodedb_cluster::calvin::types::TxClass; use nodedb_cluster::{RaftRpc, SubmitCalvinInboxRequest, SubmitCalvinInboxResponse}; use crate::Error; -use crate::control::server::exchange::resolve::register_peers_from_topology; +use crate::control::cluster::warm_peers::register_peers_from_topology; use crate::control::state::SharedState; /// The sequencer ASSIGNMENT for a submitted dependent (OLLP) `TxClass`. diff --git a/nodedb/src/control/planner/calvin/submit/local.rs b/nodedb/src/control/planner/calvin/submit/local.rs index 57ea00b5a..30d51eac6 100644 --- a/nodedb/src/control/planner/calvin/submit/local.rs +++ b/nodedb/src/control/planner/calvin/submit/local.rs @@ -76,6 +76,23 @@ pub async fn submit_and_await_calvin_with_timeout( .get() .ok_or(Error::SequencerUnavailable)?; + // A write to a permission-tree source is acknowledged only once it binds + // every node. Tree sources live in the default database. + let binds_authorization = tx_class.database_id == crate::types::DatabaseId::DEFAULT + && tx_class + .write_set + .participating_vshards_in_database(tx_class.database_id) + .map_err(|e| Error::BadRequest { + detail: format!("Calvin transaction write set: {e}"), + })? + .iter() + .any(|vshard| { + state + .authorization_fence + .sources() + .is_source_vshard(vshard.as_u32()) + }); + let inbox_seq = inbox.submit(tx_class).map_err(|e| Error::BadRequest { detail: format!("Calvin sequencer rejected transaction: {e}"), })?; @@ -141,6 +158,9 @@ pub async fn submit_and_await_calvin_with_timeout( detail: "OLLP mismatch outcome on non-dependent Calvin path".to_owned(), }); } + if binds_authorization { + crate::control::security::auth_lease::calvin_write_barrier(state).await?; + } // Completion fired: the scheduler deposited the applied Response (with any // RETURNING rows) into the sidecar BEFORE proposing the ack that woke this @@ -150,12 +170,18 @@ pub async fn submit_and_await_calvin_with_timeout( // `Conflict` (>1 RETURNING participant) fails loudly rather than returning a // partial cross-shard union. let drained = state - .calvin_apply_results + .calvin + .apply_results .lock() .unwrap_or_else(|p| p.into_inner()) .remove(&TxnId::new(epoch, position)); match drained { - Some(CalvinApplyResult::Single { response, .. }) => Ok(Some(response)), + Some(CalvinApplyResult::Single { response, .. }) => { + // An installed txn whose reply failed to render deposits it as an + // error for the statement. + crate::control::local_dispatch::reject_data_plane_error(&response)?; + Ok(Some(response)) + } Some(CalvinApplyResult::Conflict) => Err(Error::Internal { detail: "multi-participant cross-shard RETURNING not supported".to_owned(), }), diff --git a/nodedb/src/control/planner/calvin/submit/routed.rs b/nodedb/src/control/planner/calvin/submit/routed.rs index f147a794c..ab3217423 100644 --- a/nodedb/src/control/planner/calvin/submit/routed.rs +++ b/nodedb/src/control/planner/calvin/submit/routed.rs @@ -15,7 +15,7 @@ use nodedb_cluster::{RaftRpc, SubmitCalvinTxnRequest, SubmitCalvinTxnResponse, T use crate::Error; use crate::bridge::envelope::Response; -use crate::control::server::exchange::resolve::register_peers_from_topology; +use crate::control::cluster::warm_peers::register_peers_from_topology; use crate::control::state::SharedState; use super::local::{submit_and_await_calvin, synthetic_returning_response}; diff --git a/nodedb/src/control/planner/calvin/tx_class/dependent_builder.rs b/nodedb/src/control/planner/calvin/tx_class/dependent_builder.rs index f906f561b..e001c3225 100644 --- a/nodedb/src/control/planner/calvin/tx_class/dependent_builder.rs +++ b/nodedb/src/control/planner/calvin/tx_class/dependent_builder.rs @@ -287,8 +287,9 @@ mod tests { // vshard. This is exactly the shape the contended single-shard // predicate-write routing path builds. let tasks = vec![bulk_delete_task("users")]; - let want_vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "users").as_u32(); + let want_vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users") + .vshard() + .as_u32(); // Strict builder rejects the single-vshard write set. let strict = build_dependent_tx_class(&tasks, TenantId::new(1), "users", &[7, 8], &[]); diff --git a/nodedb/src/control/planner/calvin/tx_class/shared.rs b/nodedb/src/control/planner/calvin/tx_class/shared.rs index dcce00baf..fa6373c3f 100644 --- a/nodedb/src/control/planner/calvin/tx_class/shared.rs +++ b/nodedb/src/control/planner/calvin/tx_class/shared.rs @@ -414,6 +414,7 @@ mod lockstep_tests { ttl_ms: 0, surrogate: Surrogate::new(3), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, })); } @@ -494,7 +495,7 @@ mod routing_agreement_tests { vec![ task( source_write(), - VShardId::from_collection_in_database(DB, SOURCE), + nodedb_types::CollectionKey::from_bare(DB, SOURCE).vshard(), ), task(balance_write(), crate::query::sum_target_vshard(DB, TARGET)), ] @@ -530,8 +531,8 @@ mod routing_agreement_tests { #[test] fn the_fixture_spans_two_vshards() { assert_ne!( - VShardId::from_collection_in_database(DB, SOURCE), - VShardId::from_collection_in_database(DB, TARGET), + nodedb_types::CollectionKey::from_bare(DB, SOURCE).vshard(), + nodedb_types::CollectionKey::from_bare(DB, TARGET).vshard(), "the balance-pairing case only exists when source and target hash apart" ); } @@ -559,8 +560,8 @@ mod routing_agreement_tests { let tx = build_static_tx_class(&tasks, TENANT, &[]).expect("build the transaction class"); let mut expected = vec![ - VShardId::from_collection_in_database(DB, SOURCE), - VShardId::from_collection_in_database(DB, TARGET), + nodedb_types::CollectionKey::from_bare(DB, SOURCE).vshard(), + nodedb_types::CollectionKey::from_bare(DB, TARGET).vshard(), ]; expected.sort_by_key(|v| v.as_u32()); assert_eq!( diff --git a/nodedb/src/control/planner/calvin/tx_class/static_builder.rs b/nodedb/src/control/planner/calvin/tx_class/static_builder.rs index dd7ad68b6..bcf7c7ab8 100644 --- a/nodedb/src/control/planner/calvin/tx_class/static_builder.rs +++ b/nodedb/src/control/planner/calvin/tx_class/static_builder.rs @@ -294,7 +294,9 @@ mod tests { let mut first: Option<(String, u32)> = None; for i in 0u32..1024 { let name = format!("coll_{i}"); - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let v = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); match &first { Some((fname, fv)) if *fv != v => return (fname.clone(), name), Some(_) => {} @@ -351,7 +353,8 @@ mod tests { #[test] fn single_vshard_builder_preserves_database_scope() { - let mut task = point_insert_task("db_scoped", 1); + // A plan in a non-default database names its collection qualified. + let mut task = point_insert_task("7/db_scoped", 1); task.database_id = DatabaseId::new(7); let tx = build_single_vshard_tx_class(&[task], TenantId::new(1), &[]) .expect("valid single-vshard TxClass"); @@ -444,14 +447,16 @@ mod tests { .map(|v| v.as_u32()) .collect(); for coll in [col_a.as_str(), col_b.as_str(), "read_col", "scan_col"] { - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, coll).as_u32(); + let v = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, coll) + .vshard() + .as_u32(); assert!( participants.contains(&v), "participant set must include the vShard of {coll}" ); } // Every write shard is still present (the read union never drops one). - for v in tx.write_set.participating_vshards() { + for v in tx.write_set.participating_vshards().expect("participants") { assert!( participants.contains(&v.as_u32()), "read union must not drop a write shard" @@ -470,7 +475,10 @@ mod tests { // Participants collapse to the write-derived set when there are no reads. assert_eq!( tx.participating_vshards(), - tx.write_set.participating_vshards().as_slice() + tx.write_set + .participating_vshards() + .expect("participants") + .as_slice() ); } @@ -557,8 +565,9 @@ mod tests { // One point-write task → one collection → one vshard. This is exactly the // shape the contended point-write routing path builds. let tasks = vec![point_insert_task("users", 7)]; - let want_vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "users").as_u32(); + let want_vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users") + .vshard() + .as_u32(); // Strict builder rejects the single-vshard write set. let strict = build_static_tx_class(&tasks, TenantId::new(1), &[]); diff --git a/nodedb/src/control/planner/calvin/write_class.rs b/nodedb/src/control/planner/calvin/write_class.rs index 5aa7c3f65..26fcb832b 100644 --- a/nodedb/src/control/planner/calvin/write_class.rs +++ b/nodedb/src/control/planner/calvin/write_class.rs @@ -139,13 +139,14 @@ fn kv_is_write(op: &KvOp) -> bool { | KvOp::BatchGet { .. } | KvOp::FieldGet { .. } | KvOp::MaterializeScan { .. } - // `SortedIndexRank`/`TopK`/`Range`/`Count`/`Score` are `Permission::Read` - // (query-only) despite the `SortedIndex*` naming. + // `SortedIndexRank`/`TopK`/`Range`/`Count`/`Score`/`TxnRead` are + // `Permission::Read` (query-only) despite the `SortedIndex*` naming. | KvOp::SortedIndexRank { .. } | KvOp::SortedIndexTopK { .. } | KvOp::SortedIndexRange { .. } | KvOp::SortedIndexCount { .. } | KvOp::SortedIndexScore { .. } + | KvOp::SortedIndexTxnRead { .. } // Read-only: reports what a governed write would apply, mutates // nothing, and is `NotAWrite` in `plan_vshard`. | KvOp::ResolveWrite(_) @@ -481,6 +482,7 @@ mod tests { ttl_ms: 0, surrogate: Surrogate::new(1), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }); assert!(is_write_plan(&plan), "KvOp::Incr must be a write"); } @@ -490,9 +492,10 @@ mod tests { let plan = PhysicalPlan::Kv(KvOp::IncrFloat { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "cache"), key: b"k".to_vec(), - delta: 1.5, + delta: "1.5".into(), surrogate: Surrogate::new(1), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }); assert!(is_write_plan(&plan), "KvOp::IncrFloat must be a write"); } diff --git a/nodedb/src/control/planner/catalog_adapter/adapter.rs b/nodedb/src/control/planner/catalog_adapter/adapter.rs index 73f120e4b..6a80c1489 100644 --- a/nodedb/src/control/planner/catalog_adapter/adapter.rs +++ b/nodedb/src/control/planner/catalog_adapter/adapter.rs @@ -122,6 +122,17 @@ impl OriginCatalog { } } + /// Bind the node-wide sequence counters, so a column DEFAULT that calls + /// `nextval` can allocate. For a planner built with [`OriginCatalog::new`] + /// that plans rows outside a pgwire session. + pub fn with_sequence_registry( + mut self, + registry: Arc, + ) -> Self { + self.sequence_registry = Some(registry); + self + } + /// Bind the calling session's `currval` map to this adapter. pub fn with_session_sequences( mut self, diff --git a/nodedb/src/control/planner/implicit_edges/insert.rs b/nodedb/src/control/planner/implicit_edges/insert.rs index 55d1c0083..c7d3e614f 100644 --- a/nodedb/src/control/planner/implicit_edges/insert.rs +++ b/nodedb/src/control/planner/implicit_edges/insert.rs @@ -82,26 +82,14 @@ pub async fn append_implicit_edge_tasks( let vsrc = VShardId::from_key(edge.src.as_bytes()); let vdst = VShardId::from_key(edge.dst.as_bytes()); - let src_surrogate = assign_surrogate_routed( - state, - vsrc, - database_id, - tenant_id, - &edge.collection, - edge.src.as_bytes(), - trace_id, - ) - .await?; - let dst_surrogate = assign_surrogate_routed( - state, - vdst, - database_id, - tenant_id, - &edge.collection, - edge.dst.as_bytes(), - trace_id, - ) - .await?; + // `edge.collection` is the plan's database-qualified name. + let key = nodedb_types::CollectionKey::from_qualified_str(database_id, &edge.collection)?; + let src_surrogate = + assign_surrogate_routed(state, vsrc, key, tenant_id, edge.src.as_bytes(), trace_id) + .await?; + let dst_surrogate = + assign_surrogate_routed(state, vdst, key, tenant_id, edge.dst.as_bytes(), trace_id) + .await?; let properties = match edge.weight { Some(w) => weight_properties(w), diff --git a/nodedb/src/control/planner/implicit_edges/routed.rs b/nodedb/src/control/planner/implicit_edges/routed.rs index 2cf02c6a3..b77fb13e4 100644 --- a/nodedb/src/control/planner/implicit_edges/routed.rs +++ b/nodedb/src/control/planner/implicit_edges/routed.rs @@ -54,26 +54,12 @@ pub(super) async fn push_edge_delete( let vsrc = VShardId::from_key(src.as_bytes()); let vdst = VShardId::from_key(dst.as_bytes()); - let src_surrogate = assign_surrogate_routed( - state, - vsrc, - database_id, - tenant_id, - collection, - src.as_bytes(), - trace_id, - ) - .await?; - let dst_surrogate = assign_surrogate_routed( - state, - vdst, - database_id, - tenant_id, - collection, - dst.as_bytes(), - trace_id, - ) - .await?; + // `collection` is the plan's database-qualified name. + let key = nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?; + let src_surrogate = + assign_surrogate_routed(state, vsrc, key, tenant_id, src.as_bytes(), trace_id).await?; + let dst_surrogate = + assign_surrogate_routed(state, vdst, key, tenant_id, dst.as_bytes(), trace_id).await?; out.push(PhysicalTask { tenant_id, @@ -131,26 +117,12 @@ pub(super) async fn push_edge_put( let vsrc = VShardId::from_key(src.as_bytes()); let vdst = VShardId::from_key(dst.as_bytes()); - let src_surrogate = assign_surrogate_routed( - state, - vsrc, - database_id, - tenant_id, - collection, - src.as_bytes(), - trace_id, - ) - .await?; - let dst_surrogate = assign_surrogate_routed( - state, - vdst, - database_id, - tenant_id, - collection, - dst.as_bytes(), - trace_id, - ) - .await?; + // `collection` is the plan's database-qualified name. + let key = nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?; + let src_surrogate = + assign_surrogate_routed(state, vsrc, key, tenant_id, src.as_bytes(), trace_id).await?; + let dst_surrogate = + assign_surrogate_routed(state, vdst, key, tenant_id, dst.as_bytes(), trace_id).await?; out.push(PhysicalTask { tenant_id, diff --git a/nodedb/src/control/planner/materialized_sum/cross_shard.rs b/nodedb/src/control/planner/materialized_sum/cross_shard.rs index 7e80fe91e..6b787cd4e 100644 --- a/nodedb/src/control/planner/materialized_sum/cross_shard.rs +++ b/nodedb/src/control/planner/materialized_sum/cross_shard.rs @@ -77,8 +77,9 @@ pub fn append_cross_shard_balance_tasks( }; let mut for_task = Vec::new(); + let source = nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?; for binding in bindings.iter() { - if sum_target_is_co_resident(database_id, collection, &binding.target_collection) { + if sum_target_is_co_resident(source, &binding.target_collection) { continue; } for (join_value, delta) in crate::query::binding_insert_deltas(binding, &docs)? { diff --git a/nodedb/src/control/planner/materialized_sum/recon.rs b/nodedb/src/control/planner/materialized_sum/recon.rs index 8ef97adbc..ceaabb30f 100644 --- a/nodedb/src/control/planner/materialized_sum/recon.rs +++ b/nodedb/src/control/planner/materialized_sum/recon.rs @@ -31,7 +31,7 @@ use nodedb_types::{Surrogate, TenantId}; use crate::control::server::dispatch_utils::dispatch_to_data_plane; use crate::control::state::SharedState; -use crate::types::{DatabaseId, Lsn, TraceId, VShardId}; +use crate::types::{DatabaseId, Lsn, TraceId}; use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; /// What a plan-time reconnaissance read observed, and the version it observed @@ -164,20 +164,18 @@ async fn execute_read( database_id, txn_id: None, }; + // A shard verdict keeps its own typed error. let (payloads, _watermarks, read_version_lsn) = gateway .execute_internal_with_watermarks(&gw_ctx, plan) - .await - .map_err(|e| crate::Error::Storage { - engine: "materialized-sum-recon".into(), - detail: format!("reconnaissance read failed: {e}"), - })?; + .await?; return Ok(ReconRead { rows: payloads, read_version_lsn, }); } - let vshard_id = VShardId::from_collection_in_database(database_id, collection); + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?.vshard(); let response = dispatch_to_data_plane( state, tenant_id, @@ -187,15 +185,19 @@ async fn execute_read( TraceId::ZERO, ) .await?; - // A shard verdict keeps its own typed error, so a read the statement's - // deadline cut short reports the deadline rather than a storage fault. - crate::control::server::dispatch_utils::reject_data_plane_error(&response)?; - if response.status != crate::bridge::envelope::Status::Ok { - return Err(crate::Error::Storage { - engine: "materialized-sum-recon".into(), - detail: format!("reconnaissance read failed: {:?}", response.error_code), - }); - } + read_from_response(&response) +} + +/// The rows of a local read response. +/// +/// A shard verdict keeps its own typed error, so a read the statement's +/// deadline cut short reports the deadline rather than a storage fault. +/// `reject_data_plane_error` passes only a `NotFound` refusal. Its payload is +/// empty, so it reads as no rows, the answer the gateway path gives. +fn read_from_response( + response: &crate::bridge::envelope::Response, +) -> crate::Result>>> { + crate::control::local_dispatch::reject_data_plane_error(response)?; Ok(ReconRead { read_version_lsn: response.read_version_lsn, rows: vec![response.payload.to_vec()], @@ -216,3 +218,47 @@ fn decode_rows(payload: &[u8]) -> Vec { .filter_map(|(_, body)| nodedb_types::json_from_msgpack(&body).ok()) .collect() } + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::{ErrorCode, Payload, Response, Status}; + use crate::types::RequestId; + + fn refusal(code: ErrorCode) -> Response { + Response { + request_id: RequestId::new(1), + status: Status::Error, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: Some(Box::new(code)), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + /// A refused read keeps its code, never a storage error. + #[test] + fn a_refused_read_keeps_its_code() { + let code = ErrorCode::Unsupported { + detail: "not on this engine".into(), + }; + match read_from_response(&refusal(code.clone())) { + Err(crate::Error::DataPlane(kept)) => assert_eq!(kept, code), + Err(other) => panic!("expected the typed refusal, got {other:?}"), + Ok(_) => panic!("a refused read must fail"), + } + } + + /// A `NotFound` refusal reads as one empty payload, which decodes to no + /// rows. + #[test] + fn a_not_found_read_has_no_rows() { + let read = read_from_response(&refusal(ErrorCode::NotFound)) + .expect("a NotFound refusal reads as no rows"); + assert!(read.rows.iter().all(|payload| payload.is_empty())); + } +} diff --git a/nodedb/src/control/planner/materialized_sum/resolve.rs b/nodedb/src/control/planner/materialized_sum/resolve.rs index c36dc0e61..24019d3bf 100644 --- a/nodedb/src/control/planner/materialized_sum/resolve.rs +++ b/nodedb/src/control/planner/materialized_sum/resolve.rs @@ -353,14 +353,12 @@ pub(super) async fn lookup_join_value( database_id: DatabaseId, trace_id: TraceId, ) -> crate::Result { - let target = db_qualified(database_id, &binding.target_collection); let vshard = VShardId::from_key(join_value.as_bytes()); lookup_surrogate_routed( state, vshard, - database_id, + nodedb_types::CollectionKey::from_bare(database_id, &binding.target_collection), tenant_id, - &target, join_value.as_bytes(), trace_id, ) @@ -408,13 +406,6 @@ async fn resolve_bodies( Ok(resolved) } -/// Qualify a catalog collection name for the plan / surrogate namespace. -fn db_qualified(database_id: DatabaseId, collection: &str) -> String { - nodedb_types::QualifiedCollection::new(database_id, collection) - .as_str() - .to_owned() -} - /// Strip the `"/"` prefix a planned collection name carries, yielding the /// catalog name the binding index is keyed on. fn strip_db_prefix(database_id: DatabaseId, qualified: &str) -> &str { @@ -544,7 +535,11 @@ mod tests { declare_binding(&state); let target_surrogate = state .surrogate_assigner - .assign(DB, TENANT, "accounts", b"acc-1") + .assign( + nodedb_types::CollectionKey::from_bare(DB, "accounts"), + TENANT, + b"acc-1", + ) .expect("bind target row"); let mut tasks = vec![insert_task("entries", body("acc-1"))]; @@ -577,11 +572,19 @@ mod tests { declare_second_binding(&state); let accounts_row = state .surrogate_assigner - .assign(DB, TENANT, "accounts", b"acc-1") + .assign( + nodedb_types::CollectionKey::from_bare(DB, "accounts"), + TENANT, + b"acc-1", + ) .expect("bind accounts row"); let audit_row = state .surrogate_assigner - .assign(DB, TENANT, "audit_totals", b"acc-1") + .assign( + nodedb_types::CollectionKey::from_bare(DB, "audit_totals"), + TENANT, + b"acc-1", + ) .expect("bind audit_totals row"); assert_ne!( accounts_row, audit_row, @@ -680,7 +683,11 @@ mod tests { declare_binding(&state); let target_surrogate = state .surrogate_assigner - .assign(DB, TENANT, "accounts", b"acc-1") + .assign( + nodedb_types::CollectionKey::from_bare(DB, "accounts"), + TENANT, + b"acc-1", + ) .expect("bind target row"); let mut tasks = vec![PhysicalTask { diff --git a/nodedb/src/control/planner/materialized_sum/settle.rs b/nodedb/src/control/planner/materialized_sum/settle.rs index 28d74dcf5..ab09823b0 100644 --- a/nodedb/src/control/planner/materialized_sum/settle.rs +++ b/nodedb/src/control/planner/materialized_sum/settle.rs @@ -65,8 +65,8 @@ use nodedb_physical::physical_plan::{ resolved_sum_surrogate, }; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; -use nodedb_types::Surrogate; use nodedb_types::id::TxnId; +use nodedb_types::{CollectionKey, Surrogate}; use crate::control::server::shared::session::read_set::{ EngineTag, ReadKey, ReadOrigin, ReadSetEntry, @@ -143,12 +143,9 @@ pub(super) fn settle_cross_shard_images( database_id: DatabaseId, ) -> crate::Result { let mut settlement = Settlement::empty(); + let source = CollectionKey::from_qualified_str(database_id, input.source_collection)?; for binding in bindings { - if sum_target_is_co_resident( - database_id, - input.source_collection, - &binding.target_collection, - ) { + if sum_target_is_co_resident(source, &binding.target_collection) { // One core owns both rows: the balance rides the source write's own // transaction and is atomic for free. continue; @@ -314,12 +311,9 @@ pub(super) fn co_resident_target_keys( database_id: DatabaseId, ) -> crate::Result> { let mut keep: Vec = Vec::new(); + let source = CollectionKey::from_qualified_str(database_id, input.source_collection)?; for binding in bindings { - if !sum_target_is_co_resident( - database_id, - input.source_collection, - &binding.target_collection, - ) { + if !sum_target_is_co_resident(source, &binding.target_collection) { continue; } for (old, new) in input.images { @@ -338,8 +332,6 @@ pub(super) fn co_resident_target_keys( mod tests { use super::*; - use crate::types::VShardId; - const TENANT: TenantId = TenantId::new(7); const DB: DatabaseId = DatabaseId::DEFAULT; @@ -347,10 +339,10 @@ mod tests { /// below exercise the path this module exists for. fn cross_shard_pair() -> (String, String) { let source = "settle_entries".to_string(); - let home = VShardId::from_collection_in_database(DB, &source); + let home = CollectionKey::from_bare(DB, &source).vshard(); let target = (0..2048) .map(|i| format!("settle_accounts_{i}")) - .find(|candidate| VShardId::from_collection_in_database(DB, candidate) != home) + .find(|candidate| CollectionKey::from_bare(DB, candidate).vshard() != home) .unwrap_or_else(|| panic!("the routing domain must hold more than one vShard")); (source, target) } diff --git a/nodedb/src/control/planner/period_lock/lookup.rs b/nodedb/src/control/planner/period_lock/lookup.rs index 958e716a1..297243e3a 100644 --- a/nodedb/src/control/planner/period_lock/lookup.rs +++ b/nodedb/src/control/planner/period_lock/lookup.rs @@ -35,9 +35,8 @@ pub(super) async fn lookup_period_surrogate( lookup_surrogate_routed( state, vshard, - database_id, + nodedb_types::CollectionKey::from_bare(database_id, ref_table), tenant_id, - ref_table, period_key.as_bytes(), trace_id, ) diff --git a/nodedb/src/control/planner/plan_error_map.rs b/nodedb/src/control/planner/plan_error_map.rs index 1e9dd8dbc..29f7dccfd 100644 --- a/nodedb/src/control/planner/plan_error_map.rs +++ b/nodedb/src/control/planner/plan_error_map.rs @@ -35,9 +35,11 @@ pub(crate) fn map_plan_error( nodedb_sql::SqlError::UndefinedFunction { name } => { crate::Error::UndefinedFunction { name } } - // A per-row sequence accessor is a refusal, not a syntax error, so it - // keeps SQLSTATE `0A000` rather than the `42601` the fallback gives. - nodedb_sql::SqlError::SequencePerRowUnsupported { .. } => { + // A per-row sequence accessor and a search function outside its + // search plan are refusals, not syntax errors, so they keep SQLSTATE + // `0A000` rather than the syntax class `42601`. + nodedb_sql::SqlError::SequencePerRowUnsupported { .. } + | nodedb_sql::SqlError::SearchFunctionOutsideSearch { .. } => { crate::Error::FeatureNotSupported { detail: error.to_string(), } @@ -51,6 +53,7 @@ pub(crate) fn map_plan_error( // A constant expression that divides by zero is the same condition the // row-scope evaluator raises, so it carries the same code. nodedb_sql::SqlError::DivisionByZero => crate::Error::DivisionByZero, + nodedb_sql::SqlError::DataException { detail } => crate::Error::DataException { detail }, nodedb_sql::SqlError::InvalidLimitValue { clause, value } => { crate::Error::InvalidLimitValue { clause, value } } @@ -63,8 +66,94 @@ pub(crate) fn map_plan_error( // A target/expression count mismatch is a syntax error in PostgreSQL, // so it renders 42601 through `BadRequest`. nodedb_sql::SqlError::Arity { detail } => crate::Error::BadRequest { detail }, - other => crate::Error::PlanError { + // Refusals of a constraint or clause NodeDB does not implement. The + // DDL router renders both as `0A000`, so the planner path does too. + nodedb_sql::SqlError::UnsupportedConstraint { .. } + | nodedb_sql::SqlError::ConflictingEngineClause { .. } => { + crate::Error::FeatureNotSupported { + detail: error.to_string(), + } + } + // A value out of range for its type is a data exception (class `22`), + // the class PostgreSQL and the DDL DEFAULT gate give it. + nodedb_sql::SqlError::ConstantOverflow { .. } + | nodedb_sql::SqlError::IntegerOutOfRange { .. } + | nodedb_sql::SqlError::FloatOutOfRange { .. } => crate::Error::DataException { + detail: error.to_string(), + }, + // The executor's recursion cap: the program-limit class (`54000`) the + // Data-Plane verdict for the same condition renders. + nodedb_sql::SqlError::RecursionDepthExceeded { + cte_name, + max_depth, + } => crate::Error::DataPlane(crate::bridge::envelope::ErrorCode::RecursionDepthExceeded { + cte_name, + max_depth, + }), + // Statement errors the client must fix: the syntax class `42601`. + other @ (nodedb_sql::SqlError::Parse { .. } + | nodedb_sql::SqlError::TypeMismatch { .. } + | nodedb_sql::SqlError::Unsupported { .. } + | nodedb_sql::SqlError::UnevaluableDefault { .. } + | nodedb_sql::SqlError::SetvalInColumnDefault { .. } + | nodedb_sql::SqlError::InvalidFunction { .. } + | nodedb_sql::SqlError::InvalidWindowFrame { .. } + | nodedb_sql::SqlError::MissingField { .. } + | nodedb_sql::SqlError::InsertColumnArityMismatch { .. } + | nodedb_sql::SqlError::PositionalKvInsertUnsupported { .. } + | nodedb_sql::SqlError::InvalidIdentifier { .. } + | nodedb_sql::SqlError::ReservedIdentifier { .. } + | nodedb_sql::SqlError::InvalidRecursiveSetOp { .. } + | nodedb_sql::SqlError::InvalidRecursiveSelfRef { .. } + | nodedb_sql::SqlError::RecursiveColumnMismatch { .. } + | nodedb_sql::SqlError::DuplicateRecursiveColumn { .. }) => crate::Error::PlanError { detail: other.to_string(), }, } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::types::TenantId; + + #[test] + fn a_value_out_of_range_is_a_data_exception() { + let error = nodedb_sql::SqlError::IntegerOutOfRange { + column: "qty".into(), + value: 1 << 40, + declared_type: "integer", + }; + match map_plan_error(error, TenantId::new(1)) { + crate::Error::DataException { detail } => assert!(detail.contains("out of range")), + other => panic!("expected a data exception, got {other:?}"), + } + } + + #[test] + fn an_unsupported_constraint_is_feature_not_supported() { + let error = nodedb_sql::SqlError::UnsupportedConstraint { + feature: "EXCLUDE".into(), + hint: "use a unique index".into(), + }; + assert!(matches!( + map_plan_error(error, TenantId::new(1)), + crate::Error::FeatureNotSupported { .. } + )); + } + + #[test] + fn a_recursion_cap_is_a_program_limit() { + let error = nodedb_sql::SqlError::RecursionDepthExceeded { + cte_name: "walk".into(), + max_depth: 100, + }; + assert!(matches!( + map_plan_error(error, TenantId::new(1)), + crate::Error::DataPlane(crate::bridge::envelope::ErrorCode::RecursionDepthExceeded { + max_depth: 100, + .. + }) + )); + } +} diff --git a/nodedb/src/control/planner/procedural/executor/core/dispatch.rs b/nodedb/src/control/planner/procedural/executor/core/dispatch.rs index 0606e485e..9712c88e3 100644 --- a/nodedb/src/control/planner/procedural/executor/core/dispatch.rs +++ b/nodedb/src/control/planner/procedural/executor/core/dispatch.rs @@ -8,6 +8,7 @@ use super::sql_literal_concat::fold_literal_string_concat; use crate::control::planner::procedural::ast::SqlExpr; use crate::control::planner::procedural::executor::bindings::RowBindings; use crate::control::planner::procedural::executor::eval; +use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; use crate::types::TraceId; impl<'a> StatementExecutor<'a> { @@ -172,13 +173,25 @@ impl<'a> StatementExecutor<'a> { } } - let outcome = crate::control::server::wal_dispatch::wal_append_if_write( - &self.state.wal, - task.tenant_id, - task.vshard_id, - task.database_id, - &task.plan, - )?; + // The window opens before the append and closes from the + // write's outcome inside the funnel. + let owner = RecordOwner { + tenant_id: task.tenant_id, + database_id: task.database_id, + vshard_id: task.vshard_id, + }; + let minted = MintedRecords::open(&self.state.outcome_floor); + let outcome = + match minted.append_plan(&self.state.wal, owner, &task.plan, self.event_source) + { + Ok(outcome) => outcome, + Err(error) => { + // Any record appended before the error never reaches + // a core. + minted.cancel(&self.state.wal, owner, 0).await?; + return Err(error); + } + }; crate::control::server::dispatch_utils::dispatch_trusted_internal_write_to_data_plane( self.state, @@ -192,6 +205,7 @@ impl<'a> StatementExecutor<'a> { txn_id: None, wal_lsn: outcome.lsn, resolved_now_ms: outcome.resolved_now_ms, + minted: Some(minted), }, ) .await?; @@ -294,96 +308,40 @@ impl<'a> StatementExecutor<'a> { } } - /// Flush the procedure transaction buffer: WAL append + dispatch as batch. + /// Commit the procedure transaction buffer as one system transaction. + /// + /// Every statement's tasks stage through the same path a client + /// transaction takes, and COMMIT resolves them into one redo record that + /// installs all of them or none. Restart replay installs that same + /// record. Each statement's descriptor leases stay on the tasks it + /// buffered until COMMIT has checked them. pub(super) async fn flush_transaction_buffer(&self) -> crate::Result<()> { - let (tasks, _lease_scopes) = if let Some(ref tx_ctx) = self.tx_ctx { + let statements = if let Some(ref tx_ctx) = self.tx_ctx { let mut guard = tx_ctx.lock().unwrap_or_else(|p| p.into_inner()); - guard.take_buffered() + guard.take_statements() } else { return Ok(()); }; - - // `_lease_scopes` owns every statement's descriptor admission through - // all WAL appends and the complete batch dispatch below. It is dropped - // only after this function returns, including on an execution error. - if tasks.is_empty() { + if statements.iter().all(|(tasks, _)| tasks.is_empty()) { return Ok(()); } - - // Each task's WAL record has its own LSN; the batch dispatch below - // carries the highest so the Data Plane's write-version floor advances - // past every write it applies. Same approximation for the resolved TTL - // instant: a single scalar can't represent one-per-task resolved - // instants for a heterogeneous multi-statement batch, so it is only - // threaded through when the buffer holds exactly one task (below); - // resolving that properly for N>1 would need `MetaOp::TransactionBatch` - // to carry a per-plan `Vec>`, a separate, wider change to - // the procedural batch-flush path, not this KV-write fix. - let mut max_wal_lsn: Option = None; - let mut single_task_resolved_now_ms: Option = None; - for task in &tasks { - let outcome = crate::control::server::wal_dispatch::wal_append_if_write( - &self.state.wal, - task.tenant_id, - task.vshard_id, - task.database_id, - &task.plan, - )?; - if let Some(lsn) = outcome.lsn { - max_wal_lsn = Some(max_wal_lsn.map_or(lsn, |cur| cur.max(lsn))); - } - single_task_resolved_now_ms = outcome.resolved_now_ms; - } - - if tasks.len() == 1 { - if let Some(task) = tasks.into_iter().next() { - crate::control::server::dispatch_utils::dispatch_trusted_internal_write_to_data_plane( - self.state, - crate::control::server::dispatch_utils::WriteDispatch { - tenant_id: task.tenant_id, - database_id: task.database_id, - vshard_id: task.vshard_id, - plan: task.plan, - trace_id: TraceId::ZERO, - event_source: self.event_source, - txn_id: None, - wal_lsn: max_wal_lsn, - resolved_now_ms: single_task_resolved_now_ms, - }, - ) - .await?; - } - } else { - let tenant_id = tasks[0].tenant_id; - let database_id = tasks[0].database_id; - let vshard_id = tasks[0].vshard_id; - let plans: Vec<_> = tasks.into_iter().map(|t| t.plan).collect(); - let batch_plan = crate::bridge::envelope::PhysicalPlan::Meta( - nodedb_physical::physical_plan::MetaOp::TransactionBatch { - plans, - txn_id: None, - }, - ); - crate::control::server::dispatch_utils::dispatch_trusted_internal_write_to_data_plane( - self.state, - crate::control::server::dispatch_utils::WriteDispatch { - tenant_id, - database_id, - vshard_id, - plan: batch_plan, - trace_id: TraceId::ZERO, - event_source: self.event_source, - txn_id: None, - wal_lsn: max_wal_lsn, - // N>1 batch: no single instant represents every task's - // resolved TTL — see the comment above the WAL-append loop. - resolved_now_ms: None, + let statements = statements + .into_iter() + .map( + |(tasks, lease_scope)| crate::control::system_txn::SystemTxnStatement { + tasks, + lease_scope: std::sync::Arc::new(lease_scope), }, ) - .await?; - } - - Ok(()) + .collect(); + crate::control::system_txn::run_statements_atomically( + self.state, + &self.identity_for_dispatch(), + statements, + self.event_source, + ) + .await + .map_err(crate::Error::from) } } @@ -573,7 +531,9 @@ mod cross_shard_origination_tests { fn remote_homed_name(prefix: &str) -> String { for i in 0..4096u32 { let name = format!("{prefix}_{i}"); - let vshard = nodedb_cluster::routing::vshard_for_collection(DatabaseId::DEFAULT, &name); + let vshard = nodedb_cluster::routing::vshard_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name), + ); if vshard % 2 == 1 { return name; } @@ -586,7 +546,9 @@ mod cross_shard_origination_tests { fn local_homed_name(prefix: &str) -> String { for i in 0..4096u32 { let name = format!("{prefix}_{i}"); - let vshard = nodedb_cluster::routing::vshard_for_collection(DatabaseId::DEFAULT, &name); + let vshard = nodedb_cluster::routing::vshard_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name), + ); if vshard.is_multiple_of(2) { return name; } @@ -652,7 +614,10 @@ mod cross_shard_origination_tests { assert_eq!(req.cascade_depth, 0); assert_eq!( req.target_vshard, - nodedb_cluster::routing::vshard_for_collection(DatabaseId::DEFAULT, &tgt) + nodedb_cluster::routing::vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + &tgt, + )) ); } diff --git a/nodedb/src/control/planner/procedural/executor/core/state.rs b/nodedb/src/control/planner/procedural/executor/core/state.rs index 6b5138c15..fa20841a8 100644 --- a/nodedb/src/control/planner/procedural/executor/core/state.rs +++ b/nodedb/src/control/planner/procedural/executor/core/state.rs @@ -35,7 +35,6 @@ pub struct CrossShardOrigin { /// Statement executor: steps through procedural SQL blocks with DML. pub struct StatementExecutor<'a> { pub(super) state: &'a SharedState, - #[allow(dead_code)] pub(super) identity: AuthenticatedIdentity, pub(super) tenant_id: TenantId, /// Database scope fixed for this executor's lifetime. diff --git a/nodedb/src/control/planner/procedural/executor/transaction.rs b/nodedb/src/control/planner/procedural/executor/transaction.rs index 0e67ed356..ae2ffad7f 100644 --- a/nodedb/src/control/planner/procedural/executor/transaction.rs +++ b/nodedb/src/control/planner/procedural/executor/transaction.rs @@ -23,7 +23,7 @@ struct BufferedStatementScope { /// Buffered transaction context for stored procedure execution. /// /// DML statements inside a procedure body are collected here until -/// an explicit COMMIT flushes them as a TransactionBatch, or ROLLBACK +/// an explicit COMMIT flushes them as one system transaction, or ROLLBACK /// discards them. An implicit COMMIT occurs at the end of the procedure. #[derive(Default)] pub struct ProcedureTransactionCtx { @@ -74,6 +74,26 @@ impl ProcedureTransactionCtx { (std::mem::take(&mut self.buffer), scopes) } + /// Take every buffered statement with its lease scope, in buffer order + /// (on COMMIT). Clears the savepoint stack. + pub fn take_statements(&mut self) -> Vec<(Vec, QueryLeaseScope)> { + self.savepoints.clear(); + let mut tasks = std::mem::take(&mut self.buffer); + let scopes = std::mem::take(&mut self.statement_scopes); + let mut statements = Vec::with_capacity(scopes.len() + 1); + // Each statement owns the tasks from its start to the next start. + for statement in scopes.into_iter().rev() { + let own = tasks.split_off(statement.task_start.min(tasks.len())); + statements.push((own, statement.scope)); + } + // Tasks buffered ahead of the first statement carry no lease. + if !tasks.is_empty() { + statements.push((tasks, QueryLeaseScope::empty())); + } + statements.reverse(); + statements + } + /// Take tasks only. Kept for existing task-oriented tests; it intentionally /// drops the associated scopes when the returned tasks are taken. pub fn take_buffered_tasks(&mut self) -> Vec { @@ -186,6 +206,20 @@ mod tests { assert!(ctx.take_buffered_tasks().is_empty()); } + #[test] + fn commit_takes_each_statement_with_its_own_tasks() { + let mut ctx = ProcedureTransactionCtx::new(); + ctx.buffer_statement( + vec![dummy_task("a"), dummy_task("b")], + QueryLeaseScope::empty(), + ); + ctx.buffer_statement(vec![dummy_task("c")], QueryLeaseScope::empty()); + let statements = ctx.take_statements(); + let sizes: Vec = statements.iter().map(|(tasks, _)| tasks.len()).collect(); + assert_eq!(sizes, vec![2, 1]); + assert!(ctx.take_statements().is_empty()); + } + #[test] fn rollback_and_commit_take_owned_statement_scopes() { let mut ctx = ProcedureTransactionCtx::new(); diff --git a/nodedb/src/control/planner/rls_injection/array.rs b/nodedb/src/control/planner/rls_injection/array.rs index 10417cd0d..61066c1e7 100644 --- a/nodedb/src/control/planner/rls_injection/array.rs +++ b/nodedb/src/control/planner/rls_injection/array.rs @@ -59,6 +59,8 @@ pub(super) fn inject_cluster_event(_ctx: &RlsCtx<'_>, op: &ClusterEventOp) -> cr // topic publish by topic name — neither names a collection this pass // could resolve a policy against. Access to a stream or topic is // authorized on the stream/topic object itself. - ClusterEventOp::ConsumeStream { .. } | ClusterEventOp::PublishTopic { .. } => Ok(()), + ClusterEventOp::ConsumeStream { .. } + | ClusterEventOp::PublishTopic { .. } + | ClusterEventOp::TenantWriteMarks { .. } => Ok(()), } } diff --git a/nodedb/src/control/planner/rls_injection/kv.rs b/nodedb/src/control/planner/rls_injection/kv.rs index fdc89f98e..43c642e35 100644 --- a/nodedb/src/control/planner/rls_injection/kv.rs +++ b/nodedb/src/control/planner/rls_injection/kv.rs @@ -63,6 +63,14 @@ pub(super) fn inject_kv(ctx: &RlsCtx<'_>, op: &mut KvOp) -> crate::Result<()> { and the plan names only the index", ), + // Refuse: the reply is ranked keys, a rank, or a count, with no row + // body to filter. The plan names the owning collection. + KvOp::SortedIndexTxnRead { collection, .. } => ctx.refuse_if_policy( + collection, + "a sorted-index read returns ranked keys, a rank, or a count taken from stored rows, \ + so the row filter cannot be evaluated", + ), + // Admit now: a single-scalar `value` write has no field to name, // so it fails the same evaluation rather than a carve-out. KvOp::Put { @@ -273,6 +281,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }) } @@ -286,6 +295,7 @@ mod tests { rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: None, rls_filters: Vec::new(), + provenance: None, }) } @@ -345,6 +355,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); assert!(matches!( inject(&mut plan, &store), @@ -458,6 +469,7 @@ mod tests { rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: None, rls_filters: Vec::new(), + provenance: None, }, KvOp::FieldSet { collection: collection(), @@ -518,6 +530,7 @@ mod tests { ttl_ms: 0, surrogate: nodedb_types::Surrogate::ZERO, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }); assert!(inject(&mut plan, &store).is_ok()); assert!(write_check(&plan).has_predicate()); diff --git a/nodedb/src/control/planner/rls_injection/meta.rs b/nodedb/src/control/planner/rls_injection/meta.rs index be87bc245..555323526 100644 --- a/nodedb/src/control/planner/rls_injection/meta.rs +++ b/nodedb/src/control/planner/rls_injection/meta.rs @@ -101,7 +101,8 @@ pub(super) fn inject_meta(ctx: &RlsCtx<'_>, op: &mut MetaOp) -> crate::Result<() | MetaOp::RollbackToSavepoint { .. } | MetaOp::CalvinFlush { .. } | MetaOp::CalvinDrop { .. } - | MetaOp::CalvinResolve { .. } => Ok(()), + | MetaOp::CalvinResolve { .. } + | MetaOp::ApplyTransactionRedo { .. } => Ok(()), } } @@ -134,7 +135,10 @@ mod tests { #[test] fn tenant_snapshot_is_refused_while_any_policy_applies() { let store = store_with_read_policy("users"); - let mut plan = PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { tenant_id: 1 }); + let mut plan = PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { + tenant_id: 1, + cut_watermark: None, + }); assert!(matches!( inject(&mut plan, &store), Err(crate::Error::PlanError { .. }) diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/array.rs b/nodedb/src/control/planner/rls_injection/permission_tree/array.rs index e72634961..dc2fba05d 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/array.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/array.rs @@ -57,6 +57,8 @@ pub(super) fn apply_cluster_event(_ctx: &PermCtx<'_>, op: &ClusterEventOp) -> cr // topic publish by topic name — neither names a collection this pass // could resolve a tree definition against. Access to a stream or topic // is authorized on the stream/topic object itself. - ClusterEventOp::ConsumeStream { .. } | ClusterEventOp::PublishTopic { .. } => Ok(()), + ClusterEventOp::ConsumeStream { .. } + | ClusterEventOp::PublishTopic { .. } + | ClusterEventOp::TenantWriteMarks { .. } => Ok(()), } } diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/kv.rs b/nodedb/src/control/planner/rls_injection/permission_tree/kv.rs index 7d60dc9be..a889f13a2 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/kv.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/kv.rs @@ -100,6 +100,14 @@ pub(super) fn apply_kv(ctx: &PermCtx<'_>, op: &mut KvOp) -> crate::Result<()> { and the plan names only the index", ), + // Refuse: the reply is ranked keys, a rank, or a count, with no row + // body to filter. The plan names the owning collection. + KvOp::SortedIndexTxnRead { collection, .. } => ctx.refuse_if_tree( + collection, + "a sorted-index read returns ranked keys, a rank, or a count taken from stored rows, \ + so the subtree filter cannot be evaluated", + ), + // Resolve against the wrapped op: it is the intercepted write // verbatim, so it authorizes at exactly the level that write does. KvOp::ResolveWrite(inner) => apply_kv(ctx, inner), diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/meta.rs b/nodedb/src/control/planner/rls_injection/permission_tree/meta.rs index bf3b5f2a3..26d6746aa 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/meta.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/meta.rs @@ -101,7 +101,8 @@ pub(super) fn apply_meta(ctx: &PermCtx<'_>, op: &mut MetaOp) -> crate::Result<() | MetaOp::RollbackToSavepoint { .. } | MetaOp::CalvinFlush { .. } | MetaOp::CalvinDrop { .. } - | MetaOp::CalvinResolve { .. } => Ok(()), + | MetaOp::CalvinResolve { .. } + | MetaOp::ApplyTransactionRedo { .. } => Ok(()), } } @@ -134,7 +135,10 @@ mod tests { #[test] fn tenant_snapshot_is_refused_while_any_tree_applies() { let cache = cache_with_tree("docs"); - let mut plan = PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { tenant_id: 1 }); + let mut plan = PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { + tenant_id: 1, + cut_watermark: None, + }); assert!(matches!( apply(&mut plan, &cache), Err(crate::Error::PlanError { .. }) diff --git a/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs b/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs index ae9ba6e9b..8206e806a 100644 --- a/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs +++ b/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs @@ -13,7 +13,7 @@ use nodedb_sql::types::{EngineType, Filter, SortKey, SqlExpr, SqlPlan}; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; @@ -93,7 +93,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_aggregate( join_type.as_str().to_string() }; - let vshard = VShardId::from_collection_in_database(ctx.database_id, &left_collection); + let vshard = + nodedb_types::CollectionKey::from_qualified_str(ctx.database_id, &left_collection)? + .vshard(); return Ok(vec![PhysicalTask { tenant_id, @@ -212,7 +214,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_aggregate( let collection = db_qualified(ctx.database_id, &raw_collection); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, &raw_collection); - let vshard = VShardId::from_collection_in_database(ctx.database_id, &collection); + let vshard = ctx.collection_key(&raw_collection).vshard(); let group_strs = group_by_to_strings(group_by); let agg_specs: Vec = aggregates.iter().map(agg_expr_to_spec).collect(); diff --git a/nodedb/src/control/planner/sql_plan_convert/aggregate/spec.rs b/nodedb/src/control/planner/sql_plan_convert/aggregate/spec.rs index 65f4497f4..fe69c2481 100644 --- a/nodedb/src/control/planner/sql_plan_convert/aggregate/spec.rs +++ b/nodedb/src/control/planner/sql_plan_convert/aggregate/spec.rs @@ -6,7 +6,7 @@ use nodedb_sql::types::{AggregateExpr, SqlExpr, SqlPlan}; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; @@ -242,7 +242,7 @@ pub(in crate::control::planner::sql_plan_convert) fn build_input_sourced_aggrega tenant_id, // Coordinator-local: empty collection keeps the task on the // coordinator vshard (the child's rows are not per-shard). - vshard_id: VShardId::from_collection_in_database(ctx.database_id, ""), + vshard_id: nodedb_types::CollectionKey::from_bare(ctx.database_id, "").vshard(), database_id: ctx.database_id, plan: PhysicalPlan::Query(QueryOp::Aggregate { collection: nodedb_types::QualifiedCollection::from_stored(raw_collection), diff --git a/nodedb/src/control/planner/sql_plan_convert/array_alter_convert.rs b/nodedb/src/control/planner/sql_plan_convert/array_alter_convert.rs index bd383f692..c742523c1 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_alter_convert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_alter_convert.rs @@ -8,7 +8,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::array_catalog::ArrayCatalogEntry; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::MetaOp; use nodedb_types::config::retention::BitemporalRetention; @@ -80,7 +80,7 @@ pub(super) fn convert_alter_array( // durably installed by the authorized dispatch boundary. let _updated = updated; - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, diff --git a/nodedb/src/control/planner/sql_plan_convert/array_convert/ddl.rs b/nodedb/src/control/planner/sql_plan_convert/array_convert/ddl.rs index a57d8075a..b60359d56 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_convert/ddl.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_convert/ddl.rs @@ -19,7 +19,7 @@ use nodedb_sql::types_array::{ use crate::bridge::envelope::PhysicalPlan; use crate::control::array_catalog::ArrayCatalogEntry; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::ArrayOp; use super::super::convert::ConvertContext; @@ -131,7 +131,7 @@ pub(in super::super) fn convert_create_array( // 4. Emit OpenArray so the authorized execution boundary can durably // register it immediately before opening the engine side. - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, @@ -177,7 +177,7 @@ pub(in super::super) fn convert_drop_array( detail: format!("DROP ARRAY {name}: not found"), }); }; - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, diff --git a/nodedb/src/control/planner/sql_plan_convert/array_convert/dml.rs b/nodedb/src/control/planner/sql_plan_convert/array_convert/dml.rs index 8730d2316..f29870c52 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_convert/dml.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_convert/dml.rs @@ -9,7 +9,7 @@ use nodedb_sql::types_array::{ArrayCoordLiteral, ArrayInsertRow}; use crate::bridge::envelope::PhysicalPlan; use crate::engine::array::wal::{ArrayDeleteCell, ArrayPutCell}; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::{ArrayOp, ClusterArrayOp}; use super::super::convert::ConvertContext; @@ -44,7 +44,7 @@ pub(in super::super) fn convert_insert_array( })?; let aid = ArrayId::in_database(tenant_id, ctx.database_id, name); - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); let system_now_ms = chrono::Utc::now().timestamp_millis(); if ctx.cluster_enabled { @@ -57,7 +57,7 @@ pub(in super::super) fn convert_insert_array( format: "msgpack".into(), detail: format!("array coord pk encode: {e}"), })?; - let surrogate = ctx.surrogate_for_pk(name, &pk_bytes)?; + let surrogate = ctx.surrogate_for_pk(ctx.collection_key(name), &pk_bytes)?; let hilbert = encode_hilbert_prefix(&schema, &coord).map_err(|e| crate::Error::PlanError { detail: format!("INSERT INTO ARRAY {name}: Hilbert prefix: {e}"), @@ -111,7 +111,7 @@ pub(in super::super) fn convert_insert_array( format: "msgpack".into(), detail: format!("array coord pk encode: {e}"), })?; - let surrogate = ctx.surrogate_for_pk(name, &pk_bytes)?; + let surrogate = ctx.surrogate_for_pk(ctx.collection_key(name), &pk_bytes)?; cells.push(ArrayPutCell { coord, attrs, @@ -175,7 +175,7 @@ pub(in super::super) fn convert_delete_array( })?; let aid = ArrayId::in_database(tenant_id, ctx.database_id, name); - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); let system_now_ms = chrono::Utc::now().timestamp_millis(); if ctx.cluster_enabled { diff --git a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/aggregate.rs b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/aggregate.rs index d52fc535b..7b6da2551 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/aggregate.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/aggregate.rs @@ -8,7 +8,7 @@ use nodedb_sql::temporal::TemporalScope; use nodedb_sql::types_array::ArrayReducerAst; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::{ArrayOp, ClusterArrayOp}; use super::super::convert::ConvertContext; @@ -54,7 +54,7 @@ pub(crate) fn convert_agg( super::helpers::resolve_array_temporal(temporal, "ARRAY_AGG")?; let mapped = map_reducer(reducer); let aid = ArrayId::in_database(tenant_id, ctx.database_id, name); - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); let plan = if ctx.cluster_enabled { // Encode the reducer for the wire. The coordinator decodes it diff --git a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/elementwise.rs b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/elementwise.rs index 713f5bacc..17fc2fd4f 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/elementwise.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/elementwise.rs @@ -6,7 +6,7 @@ use nodedb_array::types::ArrayId; use nodedb_sql::types_array::ArrayBinaryOpAst; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::ArrayOp; use super::super::convert::ConvertContext; @@ -44,7 +44,7 @@ pub(crate) fn convert_elementwise( } let left = ArrayId::in_database(tenant_id, ctx.database_id, left_name); let right = ArrayId::in_database(tenant_id, ctx.database_id, right_name); - let vshard = VShardId::from_collection_in_database(ctx.database_id, left_name); + let vshard = ctx.collection_key(left_name).vshard(); Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, diff --git a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/maint.rs b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/maint.rs index a17a422fe..c8e1d68f0 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/maint.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/maint.rs @@ -5,7 +5,7 @@ use nodedb_array::types::ArrayId; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::ArrayOp; use super::super::convert::ConvertContext; @@ -21,7 +21,7 @@ pub(crate) fn convert_flush( detail: "ARRAY_FLUSH: no WAL wired into convert context".into(), })?; let aid = ArrayId::in_database(tenant_id, ctx.database_id, name); - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); // A frontier read, not an allocation: `Flush` appends no WAL record, and // this LSN becomes the flushed segment's watermark — every cell it contains // was written by a record below the current frontier. The write funnel mints @@ -47,7 +47,7 @@ pub(crate) fn convert_compact( ) -> crate::Result> { let entry = super::helpers::load_entry(name, tenant_id, ctx)?; let aid = ArrayId::in_database(tenant_id, ctx.database_id, name); - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, diff --git a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/project.rs b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/project.rs index 23de4709a..79b2ed2ec 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/project.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/project.rs @@ -5,7 +5,7 @@ use nodedb_array::types::ArrayId; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::ArrayOp; use super::super::convert::ConvertContext; @@ -26,7 +26,7 @@ pub(crate) fn convert_project( }); } let aid = ArrayId::in_database(tenant_id, ctx.database_id, name); - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, diff --git a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/slice.rs b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/slice.rs index ed997b9f4..0f9bd496a 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/slice.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/slice.rs @@ -11,7 +11,7 @@ use nodedb_sql::temporal::TemporalScope; use nodedb_sql::types_array::ArraySliceAst; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::{ArrayOp, ClusterArrayOp}; use super::super::convert::ConvertContext; @@ -59,7 +59,7 @@ pub(crate) fn convert_slice( super::helpers::resolve_array_temporal_scope(temporal, "ARRAY_SLICE")?; let attr_indices = resolve_attr_indices(name, attr_projection, &schema)?; let aid = ArrayId::in_database(tenant_id, ctx.database_id, name); - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); let plan = if ctx.cluster_enabled { // In cluster mode emit a ClusterArray variant. The routing loop diff --git a/nodedb/src/control/planner/sql_plan_convert/convert.rs b/nodedb/src/control/planner/sql_plan_convert/convert.rs index 03a901dfc..40930d641 100644 --- a/nodedb/src/control/planner/sql_plan_convert/convert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/convert.rs @@ -80,9 +80,9 @@ pub struct ConvertContext { /// Per-tenant maximum vector dimension (0 = unlimited). Checked in /// `VectorPrimaryInsert` conversion before the task is built. pub max_vector_dim: u32, - /// Database scope for vShard computation. All `VShardId::from_collection_in_database` - /// calls must use this value so that collections in different databases are - /// routed to distinct shards and data-plane isolates them correctly. + /// Database scope for vShard computation. Every `CollectionKey` the + /// converter builds uses this value, so collections in different + /// databases route to distinct shards and the Data Plane isolates them. pub database_id: crate::types::DatabaseId, /// Tenant scope for surrogate identity. Threaded into every surrogate /// `assign`/`lookup` so two tenants with the same primary key in a @@ -136,11 +136,17 @@ impl ConvertContext { self.purpose == PlanningPurpose::Metadata } + /// The canonical key of `bare`, a catalog collection name in this + /// context's database. Placement and surrogate identity use this key. + pub fn collection_key<'a>(&self, bare: &'a str) -> nodedb_types::CollectionKey<'a> { + nodedb_types::CollectionKey::from_bare(self.database_id, bare) + } + /// Resolve an existing surrogate without creating a mapping while planning /// metadata. Execute planning retains the allocating assignment behavior. pub fn surrogate_for_pk( &self, - collection: &str, + key: nodedb_types::CollectionKey<'_>, pk_bytes: &[u8], ) -> crate::Result { let Some(assigner) = self.surrogate_assigner.as_ref() else { @@ -148,10 +154,10 @@ impl ConvertContext { }; if self.is_metadata() { return Ok(assigner - .lookup(self.database_id, self.tenant_id, collection, pk_bytes)? + .lookup(key, self.tenant_id, pk_bytes)? .unwrap_or(nodedb_types::Surrogate::ZERO)); } - assigner.assign(self.database_id, self.tenant_id, collection, pk_bytes) + assigner.assign(key, self.tenant_id, pk_bytes) } /// Resolve an EXISTING pk → surrogate binding read-only, yielding @@ -160,14 +166,14 @@ impl ConvertContext { /// a node-local phantom binding for a key no replica agrees on. pub fn surrogate_for_existing_pk( &self, - collection: &str, + key: nodedb_types::CollectionKey<'_>, pk_bytes: &[u8], ) -> crate::Result { let Some(assigner) = self.surrogate_assigner.as_ref() else { return Ok(nodedb_types::Surrogate::ZERO); }; Ok(assigner - .lookup(self.database_id, self.tenant_id, collection, pk_bytes)? + .lookup(key, self.tenant_id, pk_bytes)? .unwrap_or(nodedb_types::Surrogate::ZERO)) } @@ -180,7 +186,7 @@ impl ConvertContext { /// the same type the allocator renders through. pub fn fresh_surrogate( &self, - collection: &str, + key: nodedb_types::CollectionKey<'_>, ) -> crate::Result<(nodedb_types::Surrogate, String)> { let placeholder = || { let zero = nodedb_types::Surrogate::ZERO; @@ -193,7 +199,7 @@ impl ConvertContext { return placeholder(); } match self.surrogate_assigner.as_ref() { - Some(assigner) => assigner.assign_fresh(self.database_id, self.tenant_id, collection), + Some(assigner) => assigner.assign_fresh(key, self.tenant_id), None => placeholder(), } } @@ -329,26 +335,43 @@ mod tests { assert_eq!( metadata - .surrogate_for_pk("users", b"new-user") + .surrogate_for_pk(metadata.collection_key("users"), b"new-user") + .unwrap() + .as_u32(), + 0 + ); + assert_eq!( + metadata + .fresh_surrogate(metadata.collection_key("users")) .unwrap() + .0 .as_u32(), 0 ); - assert_eq!(metadata.fresh_surrogate("users").unwrap().0.as_u32(), 0); assert_eq!( assigner - .lookup(DatabaseId::DEFAULT, TenantId::new(1), "users", b"new-user") + .lookup( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + TenantId::new(1), + b"new-user", + ) .unwrap(), None ); assert_eq!(registry.read().expect("registry").current_hwm(), 0); let execute = context(PlanningPurpose::Execute, Arc::clone(&assigner)); - let allocated = execute.surrogate_for_pk("users", b"new-user").unwrap(); + let allocated = execute + .surrogate_for_pk(execute.collection_key("users"), b"new-user") + .unwrap(); assert_ne!(allocated.as_u32(), 0); assert_eq!( assigner - .lookup(DatabaseId::DEFAULT, TenantId::new(1), "users", b"new-user") + .lookup( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + TenantId::new(1), + b"new-user", + ) .unwrap(), Some(allocated) ); diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/balanced_gate.rs b/nodedb/src/control/planner/sql_plan_convert/dml/balanced_gate.rs index c0284b647..b1957fe90 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/balanced_gate.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/balanced_gate.rs @@ -62,6 +62,11 @@ pub(in crate::control::planner::sql_plan_convert::dml) struct WriteGates { /// split a balanced statement's boundary and refuse journals the constraint /// permits. An absent credential store or an absent collection row declares /// neither gate. +/// +/// `collection` may be bare or db-qualified: the catalog keys collections by +/// the bare name, so the lookup de-qualifies it. A qualified name looked up +/// as-is finds no row outside the default database, and the INSERT would +/// silently skip the CRDT and BALANCED routing. pub(in crate::control::planner::sql_plan_convert::dml) fn document_collection_write_gates( ctx: &ConvertContext, collection: &str, @@ -70,8 +75,10 @@ pub(in crate::control::planner::sql_plan_convert::dml) fn document_collection_wr return Ok(WriteGates::default()); }; let catalog = credentials.catalog(); + let bare = + crate::control::target_identity::naming::bare_collection_name(ctx.database_id, collection); Ok(catalog - .get_collection(ctx.database_id, ctx.tenant_id.as_u64(), collection)? + .get_collection(ctx.database_id, ctx.tenant_id.as_u64(), &bare)? .map(|c| WriteGates { crdt: c.crdt, balanced: c.balanced.is_some(), diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/crdt_gate.rs b/nodedb/src/control/planner/sql_plan_convert/dml/crdt_gate.rs index 9a6287856..24e632b96 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/crdt_gate.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/crdt_gate.rs @@ -20,8 +20,10 @@ use nodedb_sql::types::{SqlExpr, SqlValue}; use crate::control::planner::sql_plan_convert::convert::ConvertContext; use crate::control::planner::sql_plan_convert::value::row_to_msgpack; -/// `true` when `collection` (already db-qualified by the caller) is a CRDT -/// document collection. +/// `true` when `collection` is a CRDT document collection. +/// +/// `collection` may be bare or db-qualified: the catalog keys collections by +/// the bare name, so the lookup de-qualifies it. /// /// A genuine catalog READ error propagates: misrouting a write to the non-CRDT /// path would silently bypass CRDT convergence. An ABSENT credential store or @@ -36,8 +38,10 @@ pub(in crate::control::planner::sql_plan_convert::dml) fn document_collection_is return Ok(false); }; let catalog = credentials.catalog(); + let bare = + crate::control::target_identity::naming::bare_collection_name(ctx.database_id, collection); Ok(catalog - .get_collection(ctx.database_id, ctx.tenant_id.as_u64(), collection)? + .get_collection(ctx.database_id, ctx.tenant_id.as_u64(), &bare)? .map(|c| c.crdt) .unwrap_or(false)) } diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs index 332605e35..d59c80c02 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs @@ -4,7 +4,7 @@ use nodedb_sql::types::{SqlValue, WriteRoute}; use nodedb_types::Surrogate; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use super::super::super::convert::ConvertContext; @@ -45,10 +45,11 @@ pub(in super::super::super) fn convert_insert( tenant_id, ctx, } = args; + let key = ctx.collection_key(collection); let coll_qualified = super::super::super::convert::db_qualified(ctx.database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = key.vshard(); let mut tasks = Vec::new(); let mut columnar_rows: Vec<&Vec<(String, SqlValue)>> = Vec::new(); @@ -106,7 +107,7 @@ pub(in super::super::super) fn convert_insert( let value_bytes = row_to_msgpack(row)?; let (doc_id, surrogate) = resolve_doc_identity_with_declared( ctx, - collection, + key, primary_key, declared_pk.as_deref(), row, @@ -190,7 +191,7 @@ pub(in super::super::super) fn convert_insert( }; let surrogates = columnar_row_surrogates( ctx, - collection, + key, &columnar_rows, primary_key, declared_pk.as_deref(), diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs index 3d89fdbe0..9bf70f12c 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs @@ -1,7 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 use nodedb_sql::types::SqlValue; -use nodedb_types::Surrogate; +use nodedb_types::{CollectionKey, Surrogate}; use super::super::super::convert::ConvertContext; use super::super::super::value::sql_value_to_string; @@ -38,9 +38,13 @@ pub(in super::super::super) fn declared_primary_key_name( let Some(credentials) = ctx.credentials.as_ref() else { return Ok(None); }; + // `collection` may be bare or db-qualified. The catalog keys collections + // by the bare name. + let bare = + crate::control::target_identity::naming::bare_collection_name(ctx.database_id, collection); credentials .catalog() - .declared_primary_key(ctx.database_id, ctx.tenant_id.as_u64(), collection) + .declared_primary_key(ctx.database_id, ctx.tenant_id.as_u64(), &bare) } /// Resolve a row's document id and surrogate, refusing a NULL or omitted @@ -61,7 +65,7 @@ pub(in super::super::super) fn declared_primary_key_name( /// so no caller mints an identity without the NOT NULL check running first. pub(in super::super) fn resolve_doc_identity_with_declared( ctx: &ConvertContext, - collection: &str, + key: CollectionKey<'_>, primary_key: &str, declared: Option<&str>, row: &[(String, SqlValue)], @@ -71,7 +75,7 @@ pub(in super::super) fn resolve_doc_identity_with_declared( DocId::Present(_) => {} DocId::ExplicitNull | DocId::Absent => { return Err(crate::Error::RejectedConstraint { - collection: collection.to_string(), + collection: key.name().to_string(), constraint: "not_null".to_string(), detail: format!("primary key '{declared}' cannot be NULL or omitted"), }); @@ -80,17 +84,17 @@ pub(in super::super) fn resolve_doc_identity_with_declared( } if is_auto_rowid_pk(primary_key) { - let (s, pk) = assign_fresh(ctx, collection)?; + let (s, pk) = assign_fresh(ctx, key)?; return Ok((pk, s)); } let mint_key: &str = declared.unwrap_or(primary_key); match extract_doc_id(row, mint_key) { DocId::Present(id) => { - let s = assign_for_pk(ctx, collection, id.as_bytes())?; + let s = assign_for_pk(ctx, key, id.as_bytes())?; Ok((id, s)) } DocId::ExplicitNull | DocId::Absent => { - let (s, pk) = assign_fresh(ctx, collection)?; + let (s, pk) = assign_fresh(ctx, key)?; Ok((pk, s)) } } @@ -98,10 +102,10 @@ pub(in super::super) fn resolve_doc_identity_with_declared( pub(in super::super) fn assign_for_pk( ctx: &ConvertContext, - collection: &str, + key: CollectionKey<'_>, pk_bytes: &[u8], ) -> crate::Result { - ctx.surrogate_for_pk(collection, pk_bytes) + ctx.surrogate_for_pk(key, pk_bytes) } /// Allocate a fresh, unique surrogate for a row whose primary key is the @@ -114,9 +118,9 @@ pub(in super::super) fn assign_for_pk( /// Returns the bound identity string. The caller uses it verbatim. pub(super) fn assign_fresh( ctx: &ConvertContext, - collection: &str, + key: CollectionKey<'_>, ) -> crate::Result<(Surrogate, String)> { - ctx.fresh_surrogate(collection) + ctx.fresh_surrogate(key) } /// Whether a collection's declared primary key is the auto-generated `_rowid` @@ -137,7 +141,7 @@ pub(in super::super) fn is_auto_rowid_pk(primary_key: &str) -> bool { /// caller for the whole statement — not re-read here per row. pub(in super::super) fn columnar_row_surrogates( ctx: &ConvertContext, - collection: &str, + key: CollectionKey<'_>, columnar_rows: &[&Vec<(String, SqlValue)>], primary_key: &str, declared_pk: Option<&str>, @@ -151,7 +155,7 @@ pub(in super::super) fn columnar_row_surrogates( let mut out = Vec::with_capacity(columnar_rows.len()); for row in columnar_rows { let (_, surrogate) = - resolve_doc_identity_with_declared(ctx, collection, primary_key, declared, row)?; + resolve_doc_identity_with_declared(ctx, key, primary_key, declared, row)?; out.push(surrogate); } Ok(out) diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/kv_insert.rs b/nodedb/src/control/planner/sql_plan_convert/dml/kv_insert.rs index 7fb7e237d..e01cce8b6 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/kv_insert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/kv_insert.rs @@ -5,7 +5,7 @@ use nodedb_sql::types::{KvInsertIntent, SqlExpr, SqlValue}; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use super::super::convert::ConvertContext; @@ -25,6 +25,7 @@ pub(in super::super) fn convert_kv_insert( tenant_id: TenantId, ctx: &ConvertContext, ) -> crate::Result> { + let collection_key = ctx.collection_key(collection); let coll_qualified = super::super::convert::db_qualified(ctx.database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); let collection = coll_qualified.as_str(); @@ -33,7 +34,7 @@ pub(in super::super) fn convert_kv_insert( } else { assignments_to_update_values(on_conflict_updates)? }; - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = collection_key.vshard(); let ttl_ms = ttl_secs * 1000; let mut tasks = Vec::with_capacity(entries.len()); for (key_val, value_cols) in entries { @@ -59,7 +60,7 @@ pub(in super::super) fn convert_kv_insert( } buf }; - let surrogate = assign_for_pk(ctx, collection, &key)?; + let surrogate = assign_for_pk(ctx, collection_key, &key)?; let op = match intent { KvInsertIntent::Insert => KvOp::Insert { collection: qualified_collection.clone(), @@ -112,6 +113,7 @@ pub(in super::super) fn convert_kv_insert( // RLS injection pass. returning: None, rls_filters: Vec::new(), + provenance: None, }, }; tasks.push(PhysicalTask { diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/merge.rs b/nodedb/src/control/planner/sql_plan_convert/dml/merge.rs index 9d67da961..f5d83986c 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/merge.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/merge.rs @@ -5,7 +5,7 @@ use nodedb_sql::types::{MergeClauseKind, MergePlanAction, MergePlanClause, SqlExpr, SqlPlan}; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::DocumentOp; use nodedb_physical::physical_plan::UpdateValue; use nodedb_physical::physical_plan::document::merge_types::{ @@ -45,6 +45,7 @@ pub(in super::super) fn convert_merge( tenant_id, ctx, } = args; + let target_key = ctx.collection_key(target); let target_qualified = super::super::convert::db_qualified(ctx.database_id, target); let qualified_target = nodedb_types::QualifiedCollection::new(ctx.database_id, target); let target = target_qualified.as_str(); @@ -68,7 +69,7 @@ pub(in super::super) fn convert_merge( .map(convert_clause) .collect::>>()?; - let vshard = VShardId::from_collection_in_database(ctx.database_id, target); + let vshard = target_key.vshard(); // A declared PRIMARY KEY implies NOT NULL; the Data Plane checks a MATCHED // or NOT-MATCHED-BY-SOURCE UPDATE arm's post-image against this name. let declared_primary_key = super::declared_primary_key_name(ctx, target)?; diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs index 623db1e52..f5f073635 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs @@ -5,7 +5,7 @@ use nodedb_sql::types::{EngineType, Filter, SqlValue}; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use crate::control::planner::sql_plan_convert::convert::ConvertContext; @@ -25,13 +25,14 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_delete( tenant_id: TenantId, ctx: &ConvertContext, ) -> crate::Result> { + let collection_key = ctx.collection_key(collection); let coll_qualified = crate::control::planner::sql_plan_convert::convert::db_qualified( ctx.database_id, collection, ); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = collection_key.vshard(); if matches!(engine, EngineType::KeyValue) { // A KV collection has no document store; a WHERE with no primary key @@ -71,6 +72,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_delete( // Attached by `inject_returning_spec` after plan conversion. returning: None, rls_filters: Vec::new(), + provenance: None, }), post_set_op: PostSetOp::None, txn_id: None, @@ -167,7 +169,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_delete( // runs, an unbound row_key affects 0 rows, and the clone CoW // resolver intercepts the ZERO sentinel), but a key this statement // never creates must never mint a binding. - let surrogate = ctx.surrogate_for_existing_pk(collection, &pk_bytes)?; + let surrogate = ctx.surrogate_for_existing_pk(collection_key, &pk_bytes)?; let plan = if is_crdt { PhysicalPlan::Crdt(CrdtOp::DocDelete { collection: qualified_collection.clone(), diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/shared.rs b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/shared.rs index d077830a8..f71aa744b 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/shared.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/shared.rs @@ -25,9 +25,12 @@ pub(super) fn document_collection_is_edge_bearing( let Some(credentials) = ctx.credentials.as_ref() else { return Ok(false); }; + // The catalog keys collections by the bare name. let catalog = credentials.catalog(); + let bare = + crate::control::target_identity::naming::bare_collection_name(ctx.database_id, collection); Ok(catalog - .get_collection(ctx.database_id, ctx.tenant_id.as_u64(), collection)? + .get_collection(ctx.database_id, ctx.tenant_id.as_u64(), &bare)? .map(|c| c.has_implicit_edges) .unwrap_or(false)) } diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs index 96138c013..807b59fcb 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs @@ -5,7 +5,7 @@ use nodedb_sql::types::{EngineType, Filter, SqlExpr, SqlValue}; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use crate::control::planner::sql_plan_convert::convert::ConvertContext; @@ -44,13 +44,14 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update( tenant_id, ctx, } = params; + let collection_key = ctx.collection_key(collection); let coll_qualified = crate::control::planner::sql_plan_convert::convert::db_qualified( ctx.database_id, collection, ); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = collection_key.vshard(); let filter_bytes = serialize_filters(filters)?; let updates = assignments_to_update_values(assignments)?; @@ -122,7 +123,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update( let key_bytes = sql_value_to_bytes(key)?; // Content-addressed identity: keeps the surrogate the original insert assigned. // `Surrogate::ZERO` only when no assigner is wired (test / embedded-without-catalog). - let surrogate = ctx.surrogate_for_pk(collection, &key_bytes)?; + let surrogate = ctx.surrogate_for_pk(collection_key, &key_bytes)?; tasks.push(PhysicalTask { tenant_id, vshard_id: vshard, @@ -265,7 +266,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update( let plan = if let Some(fields_json) = crdt_fields_json.as_ref() { // An upsert CREATES the row when the key is absent, so it owns // a real identity and allocates one. - let surrogate = ctx.surrogate_for_pk(collection, &pk_bytes)?; + let surrogate = ctx.surrogate_for_pk(collection_key, &pk_bytes)?; PhysicalPlan::Crdt(CrdtOp::DocUpsert { collection: qualified_collection.clone(), document_id: pk_string, @@ -281,7 +282,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update( // still runs, an unbound row_key affects 0 rows, and the clone // CoW resolver intercepts the ZERO sentinel), but an UPDATE // creates no row, so it must never mint a binding. - let surrogate = ctx.surrogate_for_existing_pk(collection, &pk_bytes)?; + let surrogate = ctx.surrogate_for_existing_pk(collection_key, &pk_bytes)?; PhysicalPlan::Document(DocumentOp::PointUpdate { collection: qualified_collection.clone(), document_id: pk_string, diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update_from.rs b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update_from.rs index 0cd5ef6ce..602f6f6d0 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update_from.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update_from.rs @@ -15,7 +15,6 @@ use nodedb_physical::physical_plan::*; use crate::control::planner::sql_plan_convert::convert::ConvertContext; use crate::control::planner::sql_plan_convert::filter::serialize_filters; use crate::control::planner::sql_plan_convert::value::assignments_to_update_values_qualified; -use crate::types::VShardId; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; /// Parameters for [`convert_update_from`], bundled to avoid an unwieldy @@ -47,6 +46,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update_from( tenant_id, ctx, } = params; + let collection_key = ctx.collection_key(collection); let coll_qualified = crate::control::planner::sql_plan_convert::convert::db_qualified( ctx.database_id, collection, @@ -78,7 +78,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update_from( let updates = assignments_to_update_values_qualified(assignments)?; let target_filter_bytes = serialize_filters(target_filters)?; - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = collection_key.vshard(); // A declared PRIMARY KEY implies NOT NULL; the Data Plane checks the // post-image against this name once the SET expressions are evaluated. let declared_primary_key = super::super::declared_primary_key_name(ctx, collection)?; diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs b/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs index f37ecf15a..b632d3e76 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs @@ -9,7 +9,7 @@ use nodedb_sql::types::{SqlExpr, SqlValue, WriteRoute}; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::ColumnarInsertIntent; use nodedb_physical::physical_plan::*; @@ -49,10 +49,11 @@ pub(in super::super) fn convert_upsert( tenant_id, ctx, } = args; + let key = ctx.collection_key(collection); let coll_qualified = super::super::convert::db_qualified(ctx.database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = key.vshard(); let mut tasks = Vec::new(); // Detect CRDT document collections once. An explicit `ON CONFLICT DO UPDATE @@ -91,7 +92,7 @@ pub(in super::super) fn convert_upsert( let value_bytes = row_to_msgpack(row)?; let (doc_id, surrogate) = resolve_doc_identity_with_declared( ctx, - collection, + key, primary_key, declared_pk.as_deref(), row, @@ -143,7 +144,7 @@ pub(in super::super) fn convert_upsert( let payload = rows_to_msgpack_array(&columnar_rows)?; let surrogates = columnar_row_surrogates( ctx, - collection, + key, &columnar_rows, primary_key, declared_pk.as_deref(), diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/vector_primary.rs b/nodedb/src/control/planner/sql_plan_convert/dml/vector_primary.rs index a79cde8a2..e57e34c91 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/vector_primary.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/vector_primary.rs @@ -9,7 +9,7 @@ //! statement never created mints no binding. use nodedb_sql::types::{Filter, SqlExpr, SqlValue, VectorPrimaryInsertIntent, VectorPrimaryRow}; -use nodedb_types::{RlsWriteCheck, Surrogate}; +use nodedb_types::{CollectionKey, RlsWriteCheck, Surrogate}; use crate::bridge::envelope::PhysicalPlan; use crate::types::{TenantId, VShardId}; @@ -47,24 +47,27 @@ pub(in super::super) struct VectorPrimaryInsertArgs<'a> { } /// The routing every vector-primary task shares. -struct Routing { +struct Routing<'a> { + /// Canonical key: placement and surrogate identity. + key: CollectionKey<'a>, qualified: nodedb_types::QualifiedCollection, collection: String, vshard: VShardId, } -fn routing(ctx: &ConvertContext, collection: &str) -> Routing { +fn routing<'a>(ctx: &ConvertContext, collection: &'a str) -> Routing<'a> { + let key = ctx.collection_key(collection); let qualified = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); let collection = db_qualified(ctx.database_id, collection); - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection.as_str()); Routing { + key, qualified, collection, - vshard, + vshard: key.vshard(), } } -fn task(tenant_id: TenantId, r: &Routing, ctx: &ConvertContext, op: VectorOp) -> PhysicalTask { +fn task(tenant_id: TenantId, r: &Routing<'_>, ctx: &ConvertContext, op: VectorOp) -> PhysicalTask { PhysicalTask { tenant_id, vshard_id: r.vshard, @@ -150,7 +153,7 @@ pub(in super::super) fn convert_vector_primary_insert( .collect(); let (doc_id, surrogate) = resolve_doc_identity_with_declared( ctx, - collection, + r.key, primary_key, declared.as_deref(), &row_fields, @@ -222,7 +225,7 @@ pub(in super::super) fn convert_vector_primary_insert( /// serialize `filters` for the Data Plane to evaluate on the sidecar rows. fn write_targets( ctx: &ConvertContext, - collection: &str, + collection: CollectionKey<'_>, filters: &[Filter], target_keys: &[SqlValue], ) -> crate::Result { @@ -246,7 +249,7 @@ pub(in super::super) fn convert_vector_primary_delete( ctx: &ConvertContext, ) -> crate::Result> { let r = routing(ctx, collection); - let targets = write_targets(ctx, r.collection.as_str(), filters, target_keys)?; + let targets = write_targets(ctx, r.key, filters, target_keys)?; Ok(vec![task( tenant_id, &r, @@ -321,7 +324,7 @@ pub(in super::super) fn convert_vector_primary_update( }); } let r = routing(ctx, collection); - let targets = write_targets(ctx, r.collection.as_str(), filters, target_keys)?; + let targets = write_targets(ctx, r.key, filters, target_keys)?; let payload_patch: Vec<(String, UpdateValue)> = assignments_to_update_values(assignments)?; Ok(vec![task( tenant_id, diff --git a/nodedb/src/control/planner/sql_plan_convert/expr/bridge_expr.rs b/nodedb/src/control/planner/sql_plan_convert/expr/bridge_expr.rs index 4924eba8b..f90ab6949 100644 --- a/nodedb/src/control/planner/sql_plan_convert/expr/bridge_expr.rs +++ b/nodedb/src/control/planner/sql_plan_convert/expr/bridge_expr.rs @@ -235,31 +235,99 @@ fn convert_expr_inner(expr: &SqlExpr, qualify: bool) -> crate::bridge::expr_eval } } - // `ARRAY['a', 'b', ...]` — lower each element and, when all resolve to - // `BExpr::Literal`, fold into a single `Value::Array` literal so that - // functions like `pg_json_has_any_key` / `pg_json_has_all_keys` receive - // a proper `Value::Array` argument rather than `Value::Null`. + // `ARRAY[a, b, ...]`: an array of literals folds to one `Value::Array` + // literal, so functions like `pg_json_has_any_key` receive a real + // array argument. An array with any other element lowers to a + // `make_array` call, which builds the array per row from the + // evaluated elements. SqlExpr::ArrayLiteral(elems) => { - let mut values = Vec::with_capacity(elems.len()); - let mut all_literal = true; - for elem in elems { - match convert_expr_inner(elem, qualify) { - BExpr::Literal(v) => values.push(v), - other => { - all_literal = false; - // Non-literal element: fall back to Null for that slot. - let _ = other; - values.push(nodedb_types::Value::Null); + let lowered: Vec = elems + .iter() + .map(|elem| convert_expr_inner(elem, qualify)) + .collect(); + let literals: Option> = lowered + .iter() + .map(|elem| { + if let BExpr::Literal(v) = elem { + Some(v.clone()) + } else { + None } - } - } - if all_literal { - BExpr::Literal(nodedb_types::Value::Array(values)) - } else { - BExpr::Literal(nodedb_types::Value::Null) + }) + .collect(); + match literals { + Some(values) => BExpr::Literal(nodedb_types::Value::Array(values)), + None => BExpr::Function { + name: "make_array".into(), + args: lowered, + }, } } _ => BExpr::Literal(nodedb_types::Value::Null), } } + +#[cfg(test)] +mod tests { + use std::collections::HashMap; + + use nodedb_sql::types::SqlValue; + use nodedb_types::Value; + + use super::*; + + fn col(name: &str) -> SqlExpr { + SqlExpr::Column { + table: None, + name: name.into(), + } + } + + fn row() -> Value { + Value::Object(HashMap::from([ + ("n".to_string(), Value::Integer(7)), + ("name".to_string(), Value::String("Alice".into())), + ])) + } + + #[test] + fn an_array_with_a_column_element_is_built_per_row() { + let expr = SqlExpr::ArrayLiteral(vec![col("n"), SqlExpr::Literal(SqlValue::Int(1))]); + let lowered = sql_expr_to_bridge_expr(&expr); + assert_eq!( + lowered.eval(&row()).expect("eval"), + Value::Array(vec![Value::Integer(7), Value::Integer(1)]) + ); + } + + #[test] + fn an_array_of_literals_folds_to_one_literal() { + let expr = SqlExpr::ArrayLiteral(vec![ + SqlExpr::Literal(SqlValue::Int(1)), + SqlExpr::Literal(SqlValue::Int(2)), + ]); + assert_eq!( + sql_expr_to_bridge_expr(&expr), + crate::bridge::expr_eval::SqlExpr::Literal(Value::Array(vec![ + Value::Integer(1), + Value::Integer(2) + ])) + ); + } + + #[test] + fn like_and_not_like_evaluate_through_the_like_function() { + let like = |negated, case_insensitive, pattern: &str| SqlExpr::Like { + expr: Box::new(col("name")), + pattern: Box::new(SqlExpr::Literal(SqlValue::String(pattern.into()))), + negated, + case_insensitive, + }; + let eval = |e: SqlExpr| sql_expr_to_bridge_expr(&e).eval(&row()).expect("eval"); + assert_eq!(eval(like(false, false, "Al%")), Value::Bool(true)); + assert_eq!(eval(like(true, false, "Al%")), Value::Bool(false)); + assert_eq!(eval(like(false, true, "al%")), Value::Bool(true)); + assert_eq!(eval(like(false, false, "al%")), Value::Bool(false)); + } +} diff --git a/nodedb/src/control/planner/sql_plan_convert/kv_counter_shape.rs b/nodedb/src/control/planner/sql_plan_convert/kv_counter_shape.rs new file mode 100644 index 000000000..4fe129c2b --- /dev/null +++ b/nodedb/src/control/planner/sql_plan_convert/kv_counter_shape.rs @@ -0,0 +1,60 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Resolve the [`KvCounterShape`] a KV counter op carries: the row an absent +//! key becomes. + +use std::sync::Arc; + +use nodedb_physical::physical_plan::KvCounterShape; +use nodedb_sql::SqlCatalog; +use nodedb_sql::planner::dml_helpers::{ + KvCounterFreshRow, KvCounterKind, plan_kv_counter_fresh_row, +}; + +use super::value::{write_msgpack_map_header, write_msgpack_str, write_msgpack_value}; +use crate::control::planner::catalog_adapter::OriginCatalog; +use crate::control::planner::plan_error_map::map_plan_error; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId}; + +/// The shape a SQL counter op on `collection` gives an absent `key`. +/// +/// A typed collection creates the row `INSERT (key, column) VALUES (key, n)` +/// stores, DEFAULTs included. A raw collection, or one the catalog does not +/// hold, stores decimal text. +pub(crate) fn kv_counter_shape( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection: &str, + key: &str, + kind: KvCounterKind, +) -> crate::Result { + let catalog = OriginCatalog::new( + Arc::clone(&state.credentials), + tenant_id.as_u64(), + database_id, + Some(Arc::clone(&state.retention_policy_registry)), + ) + .with_sequence_registry(Arc::clone(&state.sequence_registry)); + let info = catalog + .get_collection(database_id, collection) + .map_err(|e| map_plan_error(e.into(), tenant_id))?; + let Some(info) = info else { + return Ok(KvCounterShape::Raw); + }; + let fresh = plan_kv_counter_fresh_row(&info, key, kind, &catalog) + .map_err(|e| map_plan_error(e, tenant_id))?; + Ok(match fresh { + KvCounterFreshRow::Raw => KvCounterShape::Raw, + KvCounterFreshRow::Typed { column, cells } => { + let mut template = Vec::with_capacity(cells.len() * 32); + write_msgpack_map_header(&mut template, cells.len()); + for (name, value) in &cells { + write_msgpack_str(&mut template, name); + write_msgpack_value(&mut template, value); + } + KvCounterShape::Typed { column, template } + } + }) +} diff --git a/nodedb/src/control/planner/sql_plan_convert/mod.rs b/nodedb/src/control/planner/sql_plan_convert/mod.rs index c08ece90b..db0b41a2a 100644 --- a/nodedb/src/control/planner/sql_plan_convert/mod.rs +++ b/nodedb/src/control/planner/sql_plan_convert/mod.rs @@ -12,6 +12,7 @@ pub mod expr; pub mod filter; pub mod filter_scan_side; pub mod group_key_name; +pub mod kv_counter_shape; pub mod lateral; pub mod output_schema; pub mod output_schema_types; diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/core.rs b/nodedb/src/control/planner/sql_plan_convert/scan/core.rs index e2256a3ad..bd1e39c4e 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/core.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/core.rs @@ -6,7 +6,7 @@ use nodedb_sql::types::{EngineType, Filter, SqlValue}; use nodedb_types::SystemTimeScope; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use super::super::aggregate::{ @@ -54,7 +54,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_scan( let sort = convert_sort_keys(sort_keys); return Ok(vec![PhysicalTask { tenant_id, - vshard_id: VShardId::from_collection_in_database(database_id, ""), + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, "").vshard(), database_id, plan: PhysicalPlan::Query(QueryOp::ProviderScan { provider: Some(collection.to_string()), @@ -73,13 +73,12 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_scan( }]); } - let coll_qualified = super::super::convert::db_qualified(database_id, collection); + let collection_key = nodedb_types::CollectionKey::from_bare(database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(database_id, collection); - let collection = coll_qualified.as_str(); let filter_bytes = serialize_filters(filters)?; let proj_names = extract_projection_names(projection, window_functions); let sort = convert_sort_keys(sort_keys); - let vshard = VShardId::from_collection_in_database(database_id, collection); + let vshard = collection_key.vshard(); let physical = match engine { EngineType::Timeseries => { @@ -217,12 +216,11 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_document_index_look tenant_id, database_id, } = args; - let coll_qualified = super::super::convert::db_qualified(database_id, collection); + let collection_key = nodedb_types::CollectionKey::from_bare(database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(database_id, collection); - let collection = coll_qualified.as_str(); let filter_bytes = serialize_filters(filters)?; let proj_names = extract_projection_names(projection, &[]); - let vshard = VShardId::from_collection_in_database(database_id, collection); + let vshard = collection_key.vshard(); let physical = PhysicalPlan::Document(DocumentOp::IndexedFetch { collection: qualified_collection, path: field.into(), @@ -250,10 +248,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_point_get( tenant_id: TenantId, ctx: &super::super::convert::ConvertContext, ) -> crate::Result> { - let coll_qualified = super::super::convert::db_qualified(ctx.database_id, collection); + let collection_key = nodedb_types::CollectionKey::from_bare(ctx.database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = collection_key.vshard(); let physical = match engine { EngineType::KeyValue => PhysicalPlan::Kv(KvOp::Get { collection: qualified_collection.clone(), @@ -265,7 +262,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_point_get( let pk_string = sql_value_to_string(key_value); let pk_bytes = pk_string.clone().into_bytes(); let surrogate = match ctx.surrogate_assigner.as_ref() { - Some(a) => match a.lookup(ctx.database_id, ctx.tenant_id, collection, &pk_bytes)? { + Some(a) => match a.lookup(collection_key, ctx.tenant_id, &pk_bytes)? { Some(s) => s, None => { // No surrogate bound in the target database yet. diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/join.rs b/nodedb/src/control/planner/sql_plan_convert/scan/join.rs index 2523ad394..8234f110d 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/join.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/join.rs @@ -7,7 +7,7 @@ use nodedb_sql::planner::bitmap_emit::predicate::BitmapHint; use nodedb_sql::types::SqlPlan; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{DatabaseId, VShardId}; +use crate::types::DatabaseId; use nodedb_physical::physical_plan::*; use super::super::aggregate::{ @@ -176,7 +176,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_join( let left_bitmap = raw_left_bm.and_then(|h| bitmap_hint_to_plan(&h, db_id)); let right_bitmap = raw_right_bm.and_then(|h| bitmap_hint_to_plan(&h, db_id)); - let vshard = VShardId::from_collection_in_database(p.ctx.database_id, &left_collection); + let vshard = + nodedb_types::CollectionKey::from_qualified_str(p.ctx.database_id, &left_collection)? + .vshard(); // Shuffle eligibility. A whole-join shuffle is only *structurally* valid // when BOTH sides are plain sharded user collections scanned by name — i.e. diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/recursive.rs b/nodedb/src/control/planner/sql_plan_convert/scan/recursive.rs index a49744ebf..dbcd37f11 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/recursive.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/recursive.rs @@ -13,10 +13,9 @@ use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; pub(in crate::control::planner::sql_plan_convert) fn convert_recursive_scan( p: RecursiveScanParams<'_>, ) -> crate::Result> { - let coll_qualified = super::super::convert::db_qualified(p.database_id, p.collection); + let collection_key = nodedb_types::CollectionKey::from_bare(p.database_id, p.collection); let qualified_collection = nodedb_types::QualifiedCollection::new(p.database_id, p.collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(p.database_id, collection); + let vshard = collection_key.vshard(); Ok(vec![PhysicalTask { tenant_id: p.tenant_id, vshard_id: vshard, diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/search.rs b/nodedb/src/control/planner/sql_plan_convert/scan/search.rs index 79aab126b..7df0e480f 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/search.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/search.rs @@ -4,7 +4,7 @@ //! builder shared across them. use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use super::super::filter::serialize_filters; @@ -17,11 +17,10 @@ use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; pub(in crate::control::planner::sql_plan_convert) fn convert_vector_search( p: VectorSearchParams<'_>, ) -> crate::Result> { - let coll_qualified = super::super::convert::db_qualified(p.ctx.database_id, p.collection); + let collection_key = nodedb_types::CollectionKey::from_bare(p.ctx.database_id, p.collection); let qualified_collection = nodedb_types::QualifiedCollection::new(p.ctx.database_id, p.collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(p.ctx.database_id, collection); + let vshard = collection_key.vshard(); let filter_bytes = serialize_filters(p.filters)?; let inline_prefilter_plan = match p.array_prefilter { Some(pref) => Some(Box::new(build_array_prefilter_plan( @@ -63,10 +62,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_vector_search( pub(in crate::control::planner::sql_plan_convert) fn convert_sparse_search( p: SparseSearchParams<'_>, ) -> crate::Result> { - let coll_qualified = super::super::convert::db_qualified(p.database_id, p.collection); + let collection_key = nodedb_types::CollectionKey::from_bare(p.database_id, p.collection); let qualified_collection = nodedb_types::QualifiedCollection::new(p.database_id, p.collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(p.database_id, collection); + let vshard = collection_key.vshard(); Ok(vec![PhysicalTask { tenant_id: p.tenant_id, vshard_id: vshard, @@ -184,10 +182,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_text_search( ) -> crate::Result> { use nodedb_sql::fts_types::FtsQuery; - let coll_qualified = super::super::convert::db_qualified(database_id, collection); + let collection_key = nodedb_types::CollectionKey::from_bare(database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(database_id, collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(database_id, collection); + let vshard = collection_key.vshard(); // Phrase queries emit a dedicated PhraseSearch op rather than going // through the BM25 plain-string path. Score alias is not meaningful @@ -285,10 +282,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_hybrid_search( tenant_id, database_id, } = p; - let coll_qualified = super::super::convert::db_qualified(database_id, collection); + let collection_key = nodedb_types::CollectionKey::from_bare(database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(database_id, collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(database_id, collection); + let vshard = collection_key.vshard(); Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, @@ -328,10 +324,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_hybrid_search_tripl tenant_id, database_id, } = p; - let coll_qualified = super::super::convert::db_qualified(database_id, collection); + let collection_key = nodedb_types::CollectionKey::from_bare(database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(database_id, collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(database_id, collection); + let vshard = collection_key.vshard(); Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/spatial.rs b/nodedb/src/control/planner/sql_plan_convert/scan/spatial.rs index fc12aa38f..0c6d2be70 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/spatial.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/spatial.rs @@ -3,7 +3,6 @@ //! Spatial scan converter. use crate::bridge::envelope::PhysicalPlan; -use crate::types::VShardId; use nodedb_physical::physical_plan::*; use super::super::aggregate::extract_projection_names; @@ -26,10 +25,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_spatial_scan( tenant_id, database_id, } = p; - let coll_qualified = super::super::convert::db_qualified(database_id, collection); + let collection_key = nodedb_types::CollectionKey::from_bare(database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(database_id, collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(database_id, collection); + let vshard = collection_key.vshard(); let attr_bytes = serialize_filters(attribute_filters)?; let proj_names = extract_projection_names(projection, &[]); let sp = match predicate { diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs b/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs index 0d8521248..9a56408cf 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs @@ -5,7 +5,7 @@ use nodedb_sql::types::SqlValue; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use super::super::aggregate::{ @@ -37,16 +37,21 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_timeseries_scan( ctx, temporal, } = p; - let coll_qualified = super::super::convert::db_qualified(ctx.database_id, collection); + let collection_key = nodedb_types::CollectionKey::from_bare(ctx.database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); - let collection = coll_qualified.as_str(); let filter_bytes = serialize_filters(filters)?; let agg_pairs: Vec<(String, String)> = aggregates.iter().map(agg_expr_to_pair).collect(); // AUTO_TIER: split query across retention tiers if enabled. if *tiered && let Some(registry) = &ctx.retention_registry - && let Some(policy) = registry.get(ctx.database_id.as_u64(), tenant_id.as_u64(), collection) + // Policies are keyed by policy name. A collection's policy is found + // by the bare collection name it targets. + && let Some(policy) = registry.get_for_collection( + ctx.database_id.as_u64(), + tenant_id.as_u64(), + collection_key.name(), + ) && policy.auto_tier { return Ok(super::super::super::auto_tier::plan_tiered_scan( @@ -65,7 +70,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_timeseries_scan( let proj_names = extract_projection_names(projection, &[]); let computed_bytes = extract_computed_columns(projection, &[], false)?; - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = collection_key.vshard(); Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, @@ -97,10 +102,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_timeseries_ingest( tenant_id: TenantId, ctx: &super::super::convert::ConvertContext, ) -> crate::Result> { - let coll_qualified = super::super::convert::db_qualified(ctx.database_id, collection); + let collection_key = nodedb_types::CollectionKey::from_bare(ctx.database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = collection_key.vshard(); let mut payload = Vec::with_capacity(rows.len() * 128); write_msgpack_array_header(&mut payload, rows.len()); let mut surrogates: Vec = Vec::with_capacity(rows.len()); @@ -119,7 +123,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_timeseries_ingest( // PK collapses every row onto `Surrogate::ZERO` and merges distinct // rows. Nothing looks a timeseries row up by this binding, so the // identity string is discarded. - let (s, _) = ctx.fresh_surrogate(collection)?; + let (s, _) = ctx.fresh_surrogate(collection_key)?; surrogates.push(s); } Ok(vec![PhysicalTask { diff --git a/nodedb/src/control/planner/sql_plan_convert/set_ops.rs b/nodedb/src/control/planner/sql_plan_convert/set_ops.rs index 0141aac85..5e9c892d0 100644 --- a/nodedb/src/control/planner/sql_plan_convert/set_ops.rs +++ b/nodedb/src/control/planner/sql_plan_convert/set_ops.rs @@ -5,7 +5,7 @@ use nodedb_sql::types::{EngineType, Projection, SortKey, SqlExpr, SqlPlan, SqlValue, WindowSpec}; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use super::body::convert_body_to_single_plan; @@ -55,7 +55,7 @@ pub(super) fn convert_constant_result( })?; Ok(vec![PhysicalTask { tenant_id, - vshard_id: VShardId::from_collection_in_database(ctx.database_id, ""), + vshard_id: nodedb_types::CollectionKey::from_bare(ctx.database_id, "").vshard(), database_id: ctx.database_id, plan: PhysicalPlan::Query(QueryOp::ProviderScan { provider: None, @@ -87,10 +87,11 @@ pub(super) fn convert_truncate( tenant_id: TenantId, ctx: &ConvertContext, ) -> crate::Result> { + let collection_key = nodedb_types::CollectionKey::from_bare(ctx.database_id, collection); let coll_qualified = super::convert::db_qualified(ctx.database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = collection_key.vshard(); let plan = match engine { EngineType::DocumentSchemaless | EngineType::DocumentStrict => { PhysicalPlan::Document(DocumentOp::Truncate { @@ -209,6 +210,7 @@ pub(super) fn convert_insert_select( tenant_id: TenantId, ctx: &ConvertContext, ) -> crate::Result> { + let target_key = nodedb_types::CollectionKey::from_bare(ctx.database_id, target); let target_qualified = super::convert::db_qualified(ctx.database_id, target); let qualified_target = nodedb_types::QualifiedCollection::new(ctx.database_id, target); let target = target_qualified.as_str(); @@ -254,7 +256,7 @@ pub(super) fn convert_insert_select( let filter_bytes = super::filter::serialize_filters(filters)?; let column_map_bytes = super::aggregate::serialize_column_map(column_map)?; - let vshard = VShardId::from_collection_in_database(ctx.database_id, target); + let vshard = target_key.vshard(); let qualified_source = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); Ok(vec![PhysicalTask { @@ -330,7 +332,7 @@ pub(super) fn convert_subquery( tenant_id, // Coordinator-local: resolved to a `ProviderScan` over the gathered // rows (empty collection, like a constant result), dispatched once. - vshard_id: VShardId::from_collection_in_database(ctx.database_id, ""), + vshard_id: nodedb_types::CollectionKey::from_bare(ctx.database_id, "").vshard(), database_id: ctx.database_id, plan: PhysicalPlan::Query(QueryOp::PostProcess { input: Box::new(child), diff --git a/nodedb/src/control/request_tracker.rs b/nodedb/src/control/request_tracker.rs index 27d641b21..6fac98f4c 100644 --- a/nodedb/src/control/request_tracker.rs +++ b/nodedb/src/control/request_tracker.rs @@ -3,12 +3,12 @@ use std::collections::HashMap; use std::sync::{Mutex, MutexGuard}; -use tokio::sync::mpsc; +use tokio::sync::{mpsc, oneshot}; use crate::bridge::envelope::Response; use crate::types::RequestId; -/// Per-request channel capacity. A streaming scan produces at most +/// Per-request partial-response capacity. A streaming scan produces at most /// `ceil(rows / STREAM_CHUNK_SIZE)` partials — a few hundred for the /// largest realistic queries. Capacity here bounds how many chunks can /// sit in RAM while the Control-Plane session's TCP write buffer is @@ -16,19 +16,83 @@ use crate::types::RequestId; /// observe backpressure instead of silently growing RSS. pub const REQUEST_CHANNEL_CAPACITY: usize = 256; +/// The sending ends of one tracked request. +struct PendingRequest { + partials: mpsc::Sender, + final_tx: oneshot::Sender, +} + +/// The receiving end of one tracked request. +/// +/// Partial responses arrive through a bounded channel. The final response +/// has a slot of its own, so a full partial channel never drops it. +pub struct ResponseReceiver { + partials: mpsc::Receiver, + final_rx: Option>, +} + +impl ResponseReceiver { + /// The next response: every buffered partial in order, then the final + /// one. `None` once the final response was taken, or once the request + /// ended without one. + /// + /// Cancel-safe: a dropped `recv` future loses no response. + pub async fn recv(&mut self) -> Option { + if let Some(response) = self.partials.recv().await { + return Some(response); + } + let final_rx = self.final_rx.as_mut()?; + let answer = final_rx.await; + self.final_rx = None; + answer.ok() + } + + /// The next response when one is ready, without waiting. `None` when + /// nothing is ready, or once the request ended. + pub fn try_recv(&mut self) -> Option { + match self.partials.try_recv() { + Ok(response) => return Some(response), + Err(mpsc::error::TryRecvError::Empty) => return None, + Err(mpsc::error::TryRecvError::Disconnected) => {} + } + let final_rx = self.final_rx.as_mut()?; + match final_rx.try_recv() { + Ok(response) => { + self.final_rx = None; + Some(response) + } + Err(oneshot::error::TryRecvError::Empty) => None, + Err(oneshot::error::TryRecvError::Closed) => { + self.final_rx = None; + None + } + } + } + + /// A receiver fed by `partials` alone: every response, final included, + /// arrives on it in order. + #[cfg(test)] + pub(crate) fn from_channel(partials: mpsc::Receiver) -> Self { + Self { + partials, + final_rx: None, + } + } +} + /// Routes Data Plane responses back to the waiting Control Plane session. /// -/// Each dispatched request registers an mpsc sender here. The background +/// Each dispatched request registers its senders here. The background /// response poller forwards responses as they arrive. For streaming queries, /// multiple partial responses arrive before the final one. /// /// - Partial responses (`response.partial == true`): forwarded but request /// stays in the map for more chunks. -/// - Final response (`response.partial == false`): forwarded and request -/// removed from the map. +/// - Final response (`response.partial == false`): forwarded into the +/// request's final slot and request removed from the map. #[derive(Default)] pub struct RequestTracker { - pending: Mutex>>, + pending: Mutex>, } impl RequestTracker { @@ -38,48 +102,58 @@ impl RequestTracker { } } - fn lock_pending(&self) -> MutexGuard<'_, HashMap>> { + fn lock_pending(&self) -> MutexGuard<'_, HashMap> { match self.pending.lock() { Ok(guard) => guard, Err(poisoned) => poisoned.into_inner(), } } - /// Register a pending request. Returns a bounded receiver the session awaits. + /// Register a pending request. Returns the receiver the session awaits. /// /// For non-streaming requests, exactly one response arrives. /// For streaming requests, multiple partial responses arrive before the final one. - /// Channel capacity applies backpressure when the session is slow. - pub fn register(&self, id: RequestId) -> mpsc::Receiver { - let (tx, rx) = mpsc::channel(REQUEST_CHANNEL_CAPACITY); - self.lock_pending().insert(id, tx); - rx + /// Partial capacity applies backpressure when the session is slow. + pub fn register(&self, id: RequestId) -> ResponseReceiver { + let (partials_tx, partials_rx) = mpsc::channel(REQUEST_CHANNEL_CAPACITY); + let (final_tx, final_rx) = oneshot::channel(); + self.lock_pending().insert( + id, + PendingRequest { + partials: partials_tx, + final_tx, + }, + ); + ResponseReceiver { + partials: partials_rx, + final_rx: Some(final_rx), + } } /// Forward a response from the Data Plane to the waiting session. /// /// - If `response.partial` is true: sends the chunk but keeps the /// request in the map for subsequent chunks. - /// - If `response.partial` is false: sends the final chunk and - /// removes the request from the map. + /// - If `response.partial` is false: puts the response in the final + /// slot and removes the request from the map. A full partial channel + /// never refuses it. /// - /// Returns `false` if the request was cancelled, timed out, or the - /// session buffer is full (backpressure signal — the Data Plane should - /// stop producing further chunks for this request). + /// Returns `false` if the request was cancelled or its receiver dropped, + /// or if a partial found the session buffer full (backpressure signal — + /// the Data Plane must stop producing further chunks for this request). pub fn complete(&self, response: Response) -> bool { let is_final = !response.partial; let mut pending = self.lock_pending(); if is_final { - if let Some(tx) = pending.remove(&response.request_id) { - tx.try_send(response).is_ok() - } else { - false + match pending.remove(&response.request_id) { + Some(request) => request.final_tx.send(response).is_ok(), + None => false, } } else { let request_id = response.request_id; - if let Some(tx) = pending.get(&request_id) { - match tx.try_send(response) { + if let Some(request) = pending.get(&request_id) { + match request.partials.try_send(response) { Ok(()) => true, Err(_) => { // Full channel (session stalled) or closed (cancelled): @@ -205,4 +279,40 @@ mod tests { // Entry was evicted on first full-channel hit. assert_eq!(tracker.in_flight(), 0); } + + /// A session that stalls with its partial buffer full still receives + /// the final response, after every buffered partial. + #[tokio::test] + async fn a_full_partial_buffer_never_drops_the_final_response() { + let tracker = RequestTracker::new(); + let mut rx = tracker.register(RequestId::new(11)); + for i in 0..REQUEST_CHANNEL_CAPACITY { + assert!(tracker.complete(make_partial(11, &format!("chunk-{i}")))); + } + + assert!(tracker.complete(make_response(11))); + + for _ in 0..REQUEST_CHANNEL_CAPACITY { + let partial = rx.recv().await.expect("buffered partial"); + assert!(partial.partial); + } + let last = rx.recv().await.expect("final response"); + assert!(!last.partial); + assert!(rx.recv().await.is_none()); + } + + /// A `recv` dropped while it waits keeps the final response for the + /// next `recv`. + #[tokio::test] + async fn a_cancelled_recv_keeps_the_final_response() { + let tracker = RequestTracker::new(); + let mut rx = tracker.register(RequestId::new(12)); + let waited = tokio::time::timeout(std::time::Duration::from_millis(10), rx.recv()).await; + assert!(waited.is_err(), "nothing has arrived yet"); + + assert!(tracker.complete(make_response(12))); + + let last = rx.recv().await.expect("final response"); + assert_eq!(last.request_id, RequestId::new(12)); + } } diff --git a/nodedb/src/control/security/auth_fence/cluster.rs b/nodedb/src/control/security/auth_fence/cluster.rs new file mode 100644 index 000000000..a99dfb790 --- /dev/null +++ b/nodedb/src/control/security/auth_fence/cluster.rs @@ -0,0 +1,106 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Cluster helpers shared by the planning view and the authorization lease: +//! hosting checks, confirmed read indexes and applied-index waits. + +use std::time::{Duration, Instant}; + +use nodedb_cluster::WaitOutcome; + +use crate::control::cluster::read_index::ReadIndexRefusal; +use crate::control::state::SharedState; + +/// Refusal for a statement whose authorization state is not current. +pub(crate) fn behind(detail: impl Into) -> crate::Error { + crate::Error::AuthorizationStateBehind { + detail: detail.into(), + } +} + +/// Whether this node replicates `group_id`, as a voter or a learner. +pub(crate) fn hosts_group(state: &SharedState, group_id: u64) -> bool { + let Some(routing) = state.cluster_routing.as_ref() else { + return false; + }; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + routing.group_info(group_id).is_some_and(|info| { + info.members.contains(&state.node_id) || info.learners.contains(&state.node_id) + }) +} + +/// The data group that homes `vshard_id`, from this node's routing table. +pub(crate) fn group_of_vshard(state: &SharedState, vshard_id: u32) -> crate::Result { + let Some(routing) = state.cluster_routing.as_ref() else { + return Err(behind("no routing table on this node")); + }; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + routing + .group_for_vshard(vshard_id) + .map_err(|e| behind(format!("raft group of vShard {vshard_id}: {e}"))) +} + +/// Every data group in this node's routing table. +pub(crate) fn routed_groups(state: &SharedState) -> Vec { + let Some(routing) = state.cluster_routing.as_ref() else { + return Vec::new(); + }; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + routing.group_ids() +} + +/// A read index of `group_id` confirmed by its leader against a quorum, +/// taken by a probe that started after this call. +pub(crate) async fn confirmed_read_index( + state: &SharedState, + group_id: u64, + timeout: Duration, +) -> crate::Result { + let Some(gate) = state.raft_read_gate.get() else { + return Err(behind(format!( + "no read index service for raft group {group_id} yet" + ))); + }; + let deadline = Instant::now() + timeout; + state + .authorization_fence + .read_index_coalescer(group_id) + .read_index(deadline, || gate.read_index(group_id, timeout)) + .await + .map_err(|refusal| match refusal { + ReadIndexRefusal::NotLeader => { + behind(format!("raft group {group_id} has no reachable leader")) + } + ReadIndexRefusal::Timeout { waited_ms } => behind(format!( + "raft group {group_id} confirmed no read index within {waited_ms}ms" + )), + }) +} + +/// Wait until this node's applied index of `group_id` reaches `target`. +pub(crate) async fn wait_applied( + state: &SharedState, + group_id: u64, + target: u64, + timeout: Duration, +) -> crate::Result<()> { + let watcher = state.applied_index_watcher(group_id); + if watcher.current() >= target { + return Ok(()); + } + // The watcher parks its caller on a condition variable, so the wait runs + // on the blocking pool. + let outcome = tokio::task::spawn_blocking(move || watcher.wait_for(target, timeout)) + .await + .map_err(|e| crate::Error::Internal { + detail: format!("applied-index wait for raft group {group_id} did not finish: {e}"), + })?; + match outcome { + WaitOutcome::Reached => Ok(()), + WaitOutcome::TimedOut => Err(behind(format!( + "raft group {group_id} was not applied through index {target} in time" + ))), + WaitOutcome::GroupGone => Err(behind(format!( + "raft group {group_id} left this node while it caught up" + ))), + } +} diff --git a/nodedb/src/control/security/auth_fence/mod.rs b/nodedb/src/control/security/auth_fence/mod.rs new file mode 100644 index 000000000..ea63f5a9f --- /dev/null +++ b/nodedb/src/control/security/auth_fence/mod.rs @@ -0,0 +1,11 @@ +// SPDX-License-Identifier: BUSL-1.1 + +pub mod cluster; +pub mod read_index; +pub mod state; +pub mod tree_defs; +pub mod view; + +pub use state::AuthorizationFence; +pub use tree_defs::{PendingTreeDefs, TreeDefChange}; +pub use view::permission_view; diff --git a/nodedb/src/control/security/auth_fence/read_index.rs b/nodedb/src/control/security/auth_fence/read_index.rs new file mode 100644 index 000000000..07cf96fe2 --- /dev/null +++ b/nodedb/src/control/security/auth_fence/read_index.rs @@ -0,0 +1,251 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Coalesce concurrent read-index requests for one Raft group. +//! +//! A read index may answer a request only when the probe that produced it +//! started after the request arrived. A probe that started earlier can carry +//! a commit index below an entry acknowledged just before the request. So a +//! request that finds a probe in flight waits for the next one, and one probe +//! then answers every request that arrived while the previous probe ran. + +use std::future::Future; +use std::sync::Mutex; +use std::time::Instant; + +use tokio::sync::Notify; + +use crate::control::cluster::read_index::ReadIndexRefusal; + +/// Probe numbering and the last answer. +#[derive(Debug, Default)] +struct CoalescerState { + /// The probe running now, if any. + in_flight: Option, + /// The highest probe that finished. + completed: u64, + /// The answer of probe `completed`. + result: Option>, +} + +/// Coalesces read-index probes for one group. +#[derive(Debug, Default)] +pub struct ReadIndexCoalescer { + state: Mutex, + done: Notify, +} + +/// Clears the in-flight probe when its runner stops without an answer, so a +/// cancelled runner never blocks later requests. +struct RunGuard<'a> { + coalescer: &'a ReadIndexCoalescer, + probe: u64, + finished: bool, +} + +impl Drop for RunGuard<'_> { + fn drop(&mut self) { + if self.finished { + return; + } + { + let mut state = self.coalescer.lock(); + if state.in_flight == Some(self.probe) { + state.in_flight = None; + } + } + self.coalescer.done.notify_waiters(); + } +} + +impl ReadIndexCoalescer { + pub fn new() -> Self { + Self::default() + } + + fn lock(&self) -> std::sync::MutexGuard<'_, CoalescerState> { + self.state.lock().unwrap_or_else(|p| p.into_inner()) + } + + /// A read index taken by a probe that started after this call, running + /// `probe` when no such probe is in flight. Refuses with a timeout once + /// `deadline` passes. + pub async fn read_index( + &self, + deadline: Instant, + probe: F, + ) -> Result + where + F: Fn() -> Fut, + Fut: Future>, + { + let started = Instant::now(); + let target = { + let state = self.lock(); + match state.in_flight { + Some(running) => running + 1, + None => state.completed + 1, + } + }; + loop { + let notified = self.done.notified(); + tokio::pin!(notified); + notified.as_mut().enable(); + + let run = { + let mut state = self.lock(); + if state.completed >= target + && let Some(result) = state.result + { + return result; + } + if state.in_flight.is_none() { + state.in_flight = Some(target); + true + } else { + false + } + }; + + if run { + let mut guard = RunGuard { + coalescer: self, + probe: target, + finished: false, + }; + let result = probe().await; + { + let mut state = self.lock(); + state.completed = state.completed.max(target); + state.result = Some(result); + state.in_flight = None; + } + guard.finished = true; + self.done.notify_waiters(); + return result; + } + + let remaining = deadline.saturating_duration_since(Instant::now()); + if remaining.is_zero() { + return Err(ReadIndexRefusal::Timeout { + waited_ms: u64::try_from(started.elapsed().as_millis()).unwrap_or(u64::MAX), + }); + } + let _ = tokio::time::timeout(remaining, notified).await; + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + use std::sync::atomic::{AtomicU64, Ordering}; + use std::time::Duration; + + use super::*; + + fn deadline() -> Instant { + Instant::now() + Duration::from_secs(5) + } + + #[tokio::test] + async fn a_request_with_no_probe_in_flight_runs_one() { + let coalescer = ReadIndexCoalescer::new(); + let probes = AtomicU64::new(0); + let index = coalescer + .read_index(deadline(), || async { + Ok(probes.fetch_add(1, Ordering::SeqCst) + 10) + }) + .await; + assert_eq!(index, Ok(10)); + assert_eq!(probes.load(Ordering::SeqCst), 1); + } + + /// A request that arrives while a probe runs never takes that probe's + /// answer: it waits for a probe that started after it. + #[tokio::test] + async fn a_request_arriving_during_a_probe_waits_for_the_next_one() { + let coalescer = Arc::new(ReadIndexCoalescer::new()); + let release = Arc::new(Notify::new()); + let probes = Arc::new(AtomicU64::new(0)); + + let first = { + let coalescer = Arc::clone(&coalescer); + let release = Arc::clone(&release); + let probes = Arc::clone(&probes); + tokio::spawn(async move { + coalescer + .read_index(deadline(), || { + let release = Arc::clone(&release); + let probes = Arc::clone(&probes); + async move { + let n = probes.fetch_add(1, Ordering::SeqCst) + 1; + if n == 1 { + release.notified().await; + } + Ok(n * 100) + } + }) + .await + }) + }; + while probes.load(Ordering::SeqCst) == 0 { + tokio::task::yield_now().await; + } + + let second = { + let coalescer = Arc::clone(&coalescer); + let probes = Arc::clone(&probes); + tokio::spawn(async move { + coalescer + .read_index(deadline(), || { + let probes = Arc::clone(&probes); + async move { Ok((probes.fetch_add(1, Ordering::SeqCst) + 1) * 100) } + }) + .await + }) + }; + tokio::task::yield_now().await; + release.notify_one(); + + assert_eq!(first.await.expect("first"), Ok(100)); + assert_eq!(second.await.expect("second"), Ok(200)); + assert_eq!(probes.load(Ordering::SeqCst), 2); + } + + /// A runner dropped mid-probe leaves no probe marked in flight. + #[tokio::test] + async fn a_cancelled_runner_does_not_block_later_requests() { + let coalescer = Arc::new(ReadIndexCoalescer::new()); + let stuck = { + let coalescer = Arc::clone(&coalescer); + tokio::spawn( + async move { coalescer.read_index(deadline(), std::future::pending).await }, + ) + }; + tokio::task::yield_now().await; + stuck.abort(); + let _ = stuck.await; + + let index = coalescer.read_index(deadline(), || async { Ok(7) }).await; + assert_eq!(index, Ok(7)); + } + + #[tokio::test] + async fn a_request_past_its_deadline_times_out() { + let coalescer = Arc::new(ReadIndexCoalescer::new()); + let holder = { + let coalescer = Arc::clone(&coalescer); + tokio::spawn( + async move { coalescer.read_index(deadline(), std::future::pending).await }, + ) + }; + tokio::task::yield_now().await; + let refused = coalescer + .read_index(Instant::now() + Duration::from_millis(20), || async { + Ok(1) + }) + .await; + assert!(matches!(refused, Err(ReadIndexRefusal::Timeout { .. }))); + holder.abort(); + } +} diff --git a/nodedb/src/control/security/auth_fence/state.rs b/nodedb/src/control/security/auth_fence/state.rs new file mode 100644 index 000000000..fd010136b --- /dev/null +++ b/nodedb/src/control/security/auth_fence/state.rs @@ -0,0 +1,154 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Shared state of the authorization fence. + +use std::collections::HashMap; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::{Arc, Mutex, OnceLock}; + +use tokio::sync::Notify; + +use crate::control::cluster::calvin::scheduler::AppliedMirrors; +use crate::control::security::auth_lease::{ + CalvinAckCoverage, LeaderLeaseService, LeaseHolder, LeaseTiming, +}; +use crate::control::security::permission_tree::SourceIndex; +use crate::event::progress::CoreEmitProgress; + +use super::read_index::ReadIndexCoalescer; +use super::tree_defs::PendingTreeDefs; + +/// State the authorization fence and the authorization lease share. +#[derive(Debug)] +pub struct AuthorizationFence { + /// One counter per Data Plane core, installed when the Event Plane starts. + emit_progress: OnceLock>>, + /// Woken after the permission step or a reload advances the cache. + permission_applied: Notify, + /// Read-index coalescers, by Raft group. + read_index: Mutex>>, + /// Tree-definition changes the metadata applier committed. + tree_defs: PendingTreeDefs, + /// The permission cache's source collections, readable without its lock. + sources: Arc, + /// Which Calvin positions this node's schedulers applied. + calvin_mirrors: AppliedMirrors, + /// Sequencer completion acks not yet settled against local schedulers. + calvin_acks: CalvinAckCoverage, + /// This node's authorization lease. + holder: LeaseHolder, + /// Lease timing, installed when the node joins a cluster. Absent on a + /// single node, which plans without a lease. + timing: OnceLock, + /// The leader-side lease service, installed with the Raft loop. + leader: OnceLock>, + /// A Raft snapshot was installed since the permission cache last + /// reloaded for one. + snapshot_installed: AtomicBool, +} + +impl AuthorizationFence { + /// State sharing `sources` with the permission cache. + pub fn new(sources: Arc) -> Self { + Self { + emit_progress: OnceLock::new(), + permission_applied: Notify::new(), + read_index: Mutex::new(HashMap::new()), + tree_defs: PendingTreeDefs::default(), + sources, + calvin_mirrors: AppliedMirrors::default(), + calvin_acks: CalvinAckCoverage::default(), + holder: LeaseHolder::default(), + timing: OnceLock::new(), + leader: OnceLock::new(), + snapshot_installed: AtomicBool::new(false), + } + } + + /// Record that a Raft snapshot replaced data-group state. The rows it + /// brought emitted no events. + pub fn note_snapshot_installed(&self) { + self.snapshot_installed.store(true, Ordering::Release); + } + + /// Whether a snapshot was installed since the last call. + pub fn take_snapshot_installed(&self) -> bool { + self.snapshot_installed.swap(false, Ordering::AcqRel) + } + + /// Install the emitted-event counters, one per core in core order. + /// Returns `false` when counters were already installed. + pub fn install_emit_progress(&self, progress: Vec>) -> bool { + self.emit_progress.set(progress).is_ok() + } + + /// The emitted-event counter of every core, read now. `None` before the + /// Event Plane starts: no permission step runs, so the cache cannot track + /// writes and a coverage wait reloads it. + pub fn emitted_snapshot(&self) -> Option> { + self.emit_progress + .get() + .map(|cores| cores.iter().map(|core| core.emitted()).collect()) + } + + /// The wake-up the permission step and a reload fire. + pub fn permission_applied(&self) -> &Notify { + &self.permission_applied + } + + /// The tree-definition changes waiting for the cache. + pub fn tree_defs(&self) -> &PendingTreeDefs { + &self.tree_defs + } + + /// The permission cache's source collections. + pub fn sources(&self) -> &SourceIndex { + &self.sources + } + + /// Which Calvin positions this node's schedulers applied. + pub fn calvin_mirrors(&self) -> &AppliedMirrors { + &self.calvin_mirrors + } + + /// Sequencer completion acks not yet settled against local schedulers. + pub fn calvin_acks(&self) -> &CalvinAckCoverage { + &self.calvin_acks + } + + /// This node's authorization lease. + pub fn holder(&self) -> &LeaseHolder { + &self.holder + } + + /// Install the lease timing. Returns `false` when already installed. + pub fn install_timing(&self, timing: LeaseTiming) -> bool { + self.timing.set(timing).is_ok() + } + + /// The lease timing, when this node runs in a cluster. + pub fn timing(&self) -> Option { + self.timing.get().copied() + } + + /// Install the leader-side lease service. Returns `false` when already + /// installed. + pub fn install_leader(&self, service: Arc) -> bool { + self.leader.set(service).is_ok() + } + + /// The leader-side lease service, once installed. + pub fn leader(&self) -> Option<&Arc> { + self.leader.get() + } + + /// The read-index coalescer of `group_id`. + pub fn read_index_coalescer(&self, group_id: u64) -> Arc { + let mut coalescers = self.read_index.lock().unwrap_or_else(|p| p.into_inner()); + Arc::clone( + coalescers + .entry(group_id) + .or_insert_with(|| Arc::new(ReadIndexCoalescer::new())), + ) + } +} diff --git a/nodedb/src/control/security/auth_fence/tree_defs.rs b/nodedb/src/control/security/auth_fence/tree_defs.rs new file mode 100644 index 000000000..728d35745 --- /dev/null +++ b/nodedb/src/control/security/auth_fence/tree_defs.rs @@ -0,0 +1,163 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Permission-tree definition changes the metadata applier committed but the +//! permission cache has not taken yet. +//! +//! The applier runs synchronously and cannot take the cache's async lock, so +//! it queues each change here. The planning view and the lease coverage +//! apply the queue before they read the cache. Changes apply in commit +//! order. + +use std::sync::Mutex; + +use crate::control::security::catalog::StoredCollection; +use crate::control::security::permission_tree::{PermissionCache, PermissionTreeDef, SourceIndex}; +use crate::types::DatabaseId; + +/// One committed change to a collection's tree definition. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum TreeDefChange { + Register { + tenant_id: u64, + collection: String, + def: PermissionTreeDef, + }, + Unregister { + tenant_id: u64, + collection: String, + }, +} + +impl TreeDefChange { + /// The change a committed collection descriptor makes. Tree definitions + /// live on default-database collections only, as the DDL writes them. + /// An inactive collection governs nothing. + pub fn from_collection(stored: &StoredCollection) -> crate::Result> { + if stored.database_id != DatabaseId::DEFAULT { + return Ok(None); + } + let tenant_id = stored.tenant_id; + let collection = stored.name.clone(); + let def = match (&stored.permission_tree_def, stored.is_active) { + (Some(json), true) => json, + _ => { + return Ok(Some(Self::Unregister { + tenant_id, + collection, + })); + } + }; + let def: PermissionTreeDef = + sonic_rs::from_str(def).map_err(|e| crate::Error::Serialization { + format: "json".into(), + detail: format!("PERMISSION_TREE of collection '{collection}': {e}"), + })?; + Ok(Some(Self::Register { + tenant_id, + collection, + def, + })) + } + + /// Record this committed change in the source index, ahead of the cache. + pub fn note_committed(&self, sources: &SourceIndex) { + match self { + Self::Register { + tenant_id, + collection, + def, + } => sources.note_committed(*tenant_id, collection, Some(def)), + Self::Unregister { + tenant_id, + collection, + } => sources.note_committed(*tenant_id, collection, None), + } + } + + /// Apply this change to `cache`. + pub fn apply(self, cache: &mut PermissionCache) { + match self { + Self::Register { + tenant_id, + collection, + def, + } => cache.register_tree_def(tenant_id, &collection, def), + Self::Unregister { + tenant_id, + collection, + } => cache.unregister_tree_def(tenant_id, &collection), + } + } +} + +/// The queue of committed changes not yet in the cache. +#[derive(Debug, Default)] +pub struct PendingTreeDefs { + changes: Mutex>, +} + +impl PendingTreeDefs { + /// Queue a change the applier committed. + pub fn push(&self, change: TreeDefChange) { + self.changes + .lock() + .unwrap_or_else(|p| p.into_inner()) + .push(change); + } + + /// Whether a change waits. + pub fn is_empty(&self) -> bool { + self.changes + .lock() + .unwrap_or_else(|p| p.into_inner()) + .is_empty() + } + + /// Apply every queued change to `cache`. The caller holds the cache's + /// write lock, so two callers cannot apply out of order. + pub fn apply_to(&self, cache: &mut PermissionCache) { + let changes = std::mem::take(&mut *self.changes.lock().unwrap_or_else(|p| p.into_inner())); + for change in changes { + change.apply(cache); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn def() -> PermissionTreeDef { + sonic_rs::from_str( + r#"{"resource_column":"id","graph_index":"tree","permission_table":"grants"}"#, + ) + .expect("tree def") + } + + #[test] + fn queued_changes_apply_in_commit_order() { + let pending = PendingTreeDefs::default(); + pending.push(TreeDefChange::Register { + tenant_id: 1, + collection: "docs".into(), + def: def(), + }); + pending.push(TreeDefChange::Unregister { + tenant_id: 1, + collection: "docs".into(), + }); + pending.push(TreeDefChange::Register { + tenant_id: 1, + collection: "notes".into(), + def: def(), + }); + assert!(!pending.is_empty()); + + let mut cache = PermissionCache::new(); + pending.apply_to(&mut cache); + + assert!(pending.is_empty()); + assert!(cache.get_tree_def(1, "docs").is_none()); + assert_eq!(cache.get_tree_def(1, "notes"), Some(&def())); + } +} diff --git a/nodedb/src/control/security/auth_fence/view.rs b/nodedb/src/control/security/auth_fence/view.rs new file mode 100644 index 000000000..595db3fdf --- /dev/null +++ b/nodedb/src/control/security/auth_fence/view.rs @@ -0,0 +1,84 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The read of authorization state that every planning path takes. +//! +//! Planning uses local state only, and every check here is local: +//! +//! 1. Tree-definition changes the metadata applier queued move into the +//! permission cache. +//! 2. A stale cache reloads from this node's own cores: no reload covers it +//! yet (startup, or a tree definition changed), or a core lost an event. +//! A writer's acknowledgement never reloads, so a write covered only by a +//! reload binds this statement through this step. +//! 3. A tenant whose tree rows live in a Raft group this node does not +//! replicate is refused: this node never covers that group. +//! 4. In a cluster, the node must hold a valid authorization lease. A writer +//! acknowledges an authorization change only after every lease holder +//! covered it or its lease expired, so a valid lease means the state read +//! here holds every change acknowledged before this point. A node that +//! leads the metadata group as its only voter holds a pinned lease, which +//! never expires: every barrier waits for its coverage instead. +//! +//! The lease is checked after the cache guard is taken. The guard fixes the +//! cache for the whole plan, and a change acknowledged after the check was +//! acknowledged after the statement started planning. + +use std::time::Instant; + +use tokio::sync::RwLockReadGuard; + +use crate::control::security::auth_lease::lease_status; +use crate::control::security::permission_tree::{PermissionCache, reload}; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId}; + +use super::cluster::{behind, group_of_vshard, hosts_group}; + +/// Read the permission cache for planning a statement of `tenant_id`. +pub async fn permission_view( + state: &SharedState, + tenant_id: TenantId, +) -> crate::Result> { + apply_committed_tree_defs(state).await; + reload::reload_if_stale(state).await?; + + let cache = state.permission_cache.read().await; + if state.cluster_routing.is_some() && cache.has_tree_defs_for_tenant(tenant_id.as_u64()) { + for source in cache + .tree_sources() + .into_iter() + .filter(|source| source.tenant_id == tenant_id.as_u64()) + { + let vshard = + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &source.collection) + .vshard(); + let group_id = group_of_vshard(state, vshard.as_u32())?; + if !hosts_group(state, group_id) { + return Err(behind(format!( + "this node does not replicate raft group {group_id}, which homes \ + permission source '{}'; run the statement on a node that does", + source.collection + ))); + } + } + } + if !lease_status(state, Instant::now()).admits_planning() { + return Err(behind( + "this node holds no valid authorization lease; it has not confirmed the latest \ + authorization changes", + )); + } + Ok(cache) +} + +/// Move the tree-definition changes the metadata applier committed into the +/// cache. The applier queues each change before it advances the applied +/// index, so the queue holds every change this node applied. +pub(crate) async fn apply_committed_tree_defs(state: &SharedState) { + let pending = state.authorization_fence.tree_defs(); + if pending.is_empty() { + return; + } + let mut cache = state.permission_cache.write().await; + pending.apply_to(&mut cache); +} diff --git a/nodedb/src/control/security/auth_lease/barrier.rs b/nodedb/src/control/security/auth_lease/barrier.rs new file mode 100644 index 000000000..b96676cdb --- /dev/null +++ b/nodedb/src/control/security/auth_lease/barrier.rs @@ -0,0 +1,168 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The writer's side: hold an authorization change's acknowledgement until +//! no node can plan against the state before it. +//! +//! - **In a cluster** the metadata leader holds the barrier until every node +//! with an unexpired lease covered the targets, or its lease expired. The +//! writing node is a lease holder too, so its own Event Plane lag closes +//! the same way. +//! - **On a single node** there is no lease. The barrier waits until the +//! local permission cache reflects every event the cores emitted, which +//! covers the change just applied. + +use std::time::{Duration, Instant}; + +use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use nodedb_cluster::{AuthBarrierOutcome, AuthBarrierRequest, GroupCoverage, RaftRpc}; +use tokio::runtime::RuntimeFlavor; + +use crate::control::security::auth_fence::cluster::behind; +use crate::control::state::SharedState; + +use super::coverage::permission_step_covers_now; +use super::leadership::{metadata_leader, send_to_leader}; + +/// Hold the acknowledgement of a change until it binds every node. +/// +/// `targets` name the change: per group, the index it committed at. The +/// change itself is committed; an error means only that the barrier did not +/// release within the request deadline. +/// +/// Latency: in a cluster, every authorization-bearing write waits about one +/// renewal interval (the Raft heartbeat) before it is acknowledged. Each +/// holder reports its coverage only with its next renewal. This covers every +/// DDL that bears authorization, `CREATE COLLECTION` included. A holder that +/// cannot renew adds up to one lease duration (the election timeout), until +/// its lease expires. A pinned holder, the leader that is the only voter of +/// the metadata group, has no expiry: the barrier waits for its next renewal +/// however late it runs. On a single node the wait is the permission step's +/// lag only. +pub async fn authorization_barrier( + state: &SharedState, + targets: Vec, +) -> crate::Result<()> { + let deadline_secs = state.tuning.network.default_deadline_secs; + let deadline = Instant::now() + Duration::from_secs(deadline_secs); + let Some(timing) = state.authorization_fence.timing() else { + return await_local_coverage(state, deadline).await; + }; + loop { + let remaining = deadline.saturating_duration_since(Instant::now()); + if remaining.is_zero() { + return Err(committed_but_pending(format!( + "the barrier did not release within {deadline_secs}s" + ))); + } + let request = AuthBarrierRequest { + targets: targets.clone(), + timeout_ms: u64::try_from(remaining.as_millis()).unwrap_or(u64::MAX), + }; + let outcome = match metadata_leader(state).filter(|(leader, _)| *leader != 0) { + None => None, + Some((leader_id, _)) if leader_id == state.node_id => { + match state.authorization_fence.leader() { + Some(service) => Some(service.hold_barrier(request).await.outcome), + None => None, + } + } + Some((leader_id, _)) => match send_to_leader( + state, + leader_id, + RaftRpc::AuthBarrierRequest(request), + remaining + timing.lease, + ) + .await + { + Ok(RaftRpc::AuthBarrierResponse(response)) => Some(response.outcome), + Ok(other) => { + tracing::warn!( + leader_id, + "authorization barrier: unexpected reply {other:?}" + ); + None + } + Err(error) => { + tracing::debug!(%error, "authorization barrier: not delivered"); + None + } + }, + }; + match outcome { + Some(AuthBarrierOutcome::Released) => return Ok(()), + // No leader known, a leader change, or a lost message: ask the + // leader again. A new leader holds the barrier to its own floors. + Some(AuthBarrierOutcome::NotLeader { .. }) + | Some(AuthBarrierOutcome::Timeout { .. }) + | None => tokio::time::sleep(timing.renew_every).await, + } + } +} + +/// Hold the acknowledgement of a Calvin transaction that wrote a +/// permission-tree source. +/// +/// Its completion acks sit in the sequencer log, at or below the sequencer +/// group's commit index now. A node covers that index only once its own +/// replicas applied every acknowledged transaction below it. +pub async fn calvin_write_barrier(state: &SharedState) -> crate::Result<()> { + if state.authorization_fence.timing().is_none() { + let deadline = + Instant::now() + Duration::from_secs(state.tuning.network.default_deadline_secs); + return await_local_coverage(state, deadline).await; + } + let commit_index = state + .raft_status_fn + .get() + .and_then(|status| { + status() + .into_iter() + .find(|group| group.group_id == SEQUENCER_GROUP_ID) + .map(|group| group.commit_index) + }) + .ok_or_else(|| committed_but_pending("this node does not replicate the sequencer group"))?; + authorization_barrier( + state, + vec![GroupCoverage { + group_id: SEQUENCER_GROUP_ID, + through: commit_index, + }], + ) + .await +} + +/// Run [`authorization_barrier`] from synchronous code on a Tokio worker. +pub fn block_on_barrier(state: &SharedState, targets: Vec) -> crate::Result<()> { + let handle = tokio::runtime::Handle::try_current().map_err(|_| crate::Error::Internal { + detail: "authorization barrier: called outside a Tokio runtime".into(), + })?; + if handle.runtime_flavor() != RuntimeFlavor::MultiThread { + return Err(crate::Error::Internal { + detail: "authorization barrier: synchronous callers need a multi-thread runtime".into(), + }); + } + tokio::task::block_in_place(|| handle.block_on(authorization_barrier(state, targets))) +} + +/// Wait until the permission step covers every event the cores emitted +/// before this call, which includes the write just applied. +/// +/// This is a writer's acknowledgement wait: it only waits, and never reloads +/// or dispatches. A cache that needs a reload is reloaded by the next +/// statement's planning, before it reads the cache. +pub async fn await_local_coverage(state: &SharedState, deadline: Instant) -> crate::Result<()> { + let remaining = deadline.saturating_duration_since(Instant::now()); + if permission_step_covers_now(state, remaining).await { + return Ok(()); + } + Err(committed_but_pending( + "the permission cache did not catch up with the change", + )) +} + +fn committed_but_pending(detail: impl std::fmt::Display) -> crate::Error { + behind(format!( + "the authorization change is committed, but {detail}; nodes still planning against \ + the previous state refuse until they catch up" + )) +} diff --git a/nodedb/src/control/security/auth_lease/calvin_acks.rs b/nodedb/src/control/security/auth_lease/calvin_acks.rs new file mode 100644 index 000000000..af6596b51 --- /dev/null +++ b/nodedb/src/control/security/auth_lease/calvin_acks.rs @@ -0,0 +1,164 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Coverage of the sequencer group. +//! +//! A Calvin transaction is acknowledged once every participant vShard's +//! leader proposed a `CompletionAck` to the sequencer log. Every other +//! replica of the vShard applies the transaction in its own time. This node +//! covers sequencer index `i` when it applied the sequencer log through `i` +//! and, for every ack at or below `i` naming a vShard this node replicates, +//! its own scheduler applied that transaction too. +//! +//! The sequencer state machine records each ack it applies. This tracker +//! takes them in log order and settles each against the local scheduler's +//! applied mirror. The first unsettled ack bounds the coverage. + +use std::collections::VecDeque; +use std::sync::Mutex; + +use nodedb_cluster::calvin::{AppliedCompletionAck, CalvinCompletionRegistry}; + +use crate::control::cluster::calvin::scheduler::AppliedMirrors; + +/// Acks taken from the sequencer log and not yet settled. +#[derive(Debug, Default)] +pub struct CalvinAckCoverage { + pending: Mutex>, +} + +impl CalvinAckCoverage { + /// The sequencer index this node covers, given that it applied the + /// sequencer log through `applied`, read before this call. + /// + /// `replicates` answers whether this node replicates a vShard. An ack for + /// a vShard it does not replicate settles at once: this node never plans + /// against that vShard's rows. + pub fn covered_through( + &self, + registry: &CalvinCompletionRegistry, + mirrors: &AppliedMirrors, + applied: u64, + replicates: impl Fn(u32) -> bool, + ) -> u64 { + let mut pending = self.pending.lock().unwrap_or_else(|p| p.into_inner()); + pending.extend(registry.applied_acks.drain()); + while let Some(ack) = pending.front() { + let settled = !replicates(ack.vshard_id) + || mirrors + .get(ack.vshard_id) + .is_some_and(|mirror| mirror.is_applied(ack.txn.epoch, ack.txn.position)); + if !settled { + break; + } + pending.pop_front(); + } + match pending.front() { + Some(ack) => applied.min(ack.index.saturating_sub(1)), + None => applied, + } + } +} + +#[cfg(test)] +mod tests { + use std::collections::BTreeSet; + + use nodedb_cluster::calvin::TxnId; + + use super::*; + use crate::control::cluster::calvin::scheduler::NOT_YET_APPLIED_EPOCH; + + fn ack(index: u64, epoch: u64, vshard_id: u32) -> AppliedCompletionAck { + AppliedCompletionAck { + index, + txn: TxnId::new(epoch, 0), + vshard_id, + } + } + + #[test] + fn an_ack_the_local_replica_has_not_applied_bounds_the_coverage() { + let registry = CalvinCompletionRegistry::new_detached(); + registry.applied_acks.enable(); + let mirrors = AppliedMirrors::default(); + let mirror = mirrors.register(7, NOT_YET_APPLIED_EPOCH, &BTreeSet::new()); + let coverage = CalvinAckCoverage::default(); + + registry.applied_acks.record(ack(4, 1, 7)); + registry.applied_acks.record(ack(6, 2, 9)); + // vShard 7 is replicated here and its scheduler has not applied epoch 1. + let replicates = |vshard: u32| vshard == 7; + assert_eq!( + coverage.covered_through(®istry, &mirrors, 8, replicates), + 3 + ); + + mirror.mark(1, 0); + // The ack for vShard 9 settles at once: it is not replicated here. + assert_eq!( + coverage.covered_through(®istry, &mirrors, 8, replicates), + 8 + ); + } + + #[test] + fn a_replicated_vshard_without_a_scheduler_yet_is_unsettled() { + let registry = CalvinCompletionRegistry::new_detached(); + registry.applied_acks.enable(); + let mirrors = AppliedMirrors::default(); + let coverage = CalvinAckCoverage::default(); + registry.applied_acks.record(ack(2, 1, 5)); + assert_eq!( + coverage.covered_through(®istry, &mirrors, 3, |_| true), + 1 + ); + } + + /// A restart loses the in-memory mirror and, after a checkpoint, the WAL + /// markers of transactions applied before it. The mirror rebuilt from the + /// state the checkpoint saved settles every ack the sequencer log replays, + /// so the coverage reaches the sequencer log's applied index. + #[test] + fn a_mirror_rebuilt_from_saved_state_settles_replayed_acks() { + use crate::control::cluster::calvin::scheduler::recover_applied; + use crate::control::security::catalog::SystemCatalog; + use crate::control::security::catalog::calvin_applied::StoredCalvinApplied; + + let wal_dir = tempfile::tempdir().expect("wal dir"); + // The WAL a checkpoint truncated holds no applied marker. + let wal = crate::wal::manager::WalManager::open(wal_dir.path(), false).expect("wal"); + let catalog_dir = tempfile::tempdir().expect("catalog dir"); + let catalog = + SystemCatalog::open(&catalog_dir.path().join("system.redb")).expect("catalog"); + catalog + .save_calvin_applied(vec![StoredCalvinApplied { + vshard_id: 7, + fully_applied_epoch: NOT_YET_APPLIED_EPOCH, + tail: [(0, 0), (3, 1)].into_iter().collect(), + }]) + .expect("save"); + + let recovered = recover_applied(&wal, &catalog, 7).expect("recover"); + let mirrors = AppliedMirrors::default(); + mirrors.register(7, recovered.fully_applied_epoch, &recovered.applied_tail); + + let registry = CalvinCompletionRegistry::new_detached(); + registry.applied_acks.enable(); + registry.applied_acks.record(AppliedCompletionAck { + index: 5, + txn: TxnId::new(0, 0), + vshard_id: 7, + }); + registry.applied_acks.record(AppliedCompletionAck { + index: 9, + txn: TxnId::new(3, 1), + vshard_id: 7, + }); + let coverage = CalvinAckCoverage::default(); + assert_eq!( + coverage.covered_through(®istry, &mirrors, 19, |_| true), + 19, + "every replayed ack settles against the rebuilt mirror" + ); + } +} diff --git a/nodedb/src/control/security/auth_lease/coverage.rs b/nodedb/src/control/security/auth_lease/coverage.rs new file mode 100644 index 000000000..0bf20c5c4 --- /dev/null +++ b/nodedb/src/control/security/auth_lease/coverage.rs @@ -0,0 +1,196 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! What this node's authorization state covers, per Raft group. +//! +//! - **Metadata group:** roles, grants, RLS policies, scope grants and tree +//! definitions. The metadata applier updates each store before it advances +//! the applied index, and queues tree-definition changes, which this step +//! moves into the permission cache. The applied index is covered as read. +//! - **Data groups:** the permission cache holds tree rows through the +//! Event Plane's permission step. A group's applied index read before the +//! emitted-event counters is covered once the step reaches those counters. +//! - **Sequencer group:** Calvin writes. Covered as [`super::calvin_acks`] +//! settles, then through the permission step like a data group. +//! - **A group this node does not replicate** is reported at `u64::MAX`: +//! planning here refuses any tenant whose tree rows live in it. +//! +//! Indexes are read first, then the counters, then the cache is checked. A +//! group's writes at or below the index read emitted their events before the +//! counters were read, so a cache that reached the counters holds them. + +use std::time::{Duration, Instant}; + +use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use nodedb_cluster::{GroupCoverage, METADATA_GROUP_ID}; + +use crate::control::security::auth_fence::cluster::{group_of_vshard, hosts_group, routed_groups}; +use crate::control::security::auth_fence::view::apply_committed_tree_defs; +use crate::control::security::permission_tree::reload; +use crate::control::state::SharedState; + +/// Raw indexes, read before the emitted-event counters. +struct RawCoverage { + metadata: u64, + /// Data groups and the sequencer group, which the permission step covers. + event_backed: Vec, +} + +fn read_raw(state: &SharedState) -> RawCoverage { + let metadata = state.applied_index_watcher(METADATA_GROUP_ID).current(); + let mut event_backed: Vec = routed_groups(state) + .into_iter() + .filter(|group_id| *group_id != METADATA_GROUP_ID && *group_id != SEQUENCER_GROUP_ID) + .map(|group_id| GroupCoverage { + group_id, + through: if hosts_group(state, group_id) { + state.applied_index_watcher(group_id).current() + } else { + u64::MAX + }, + }) + .collect(); + event_backed.push(GroupCoverage { + group_id: SEQUENCER_GROUP_ID, + through: sequencer_coverage(state), + }); + RawCoverage { + metadata, + event_backed, + } +} + +/// The sequencer index this node's Calvin replicas cover. +fn sequencer_coverage(state: &SharedState) -> u64 { + let hosts_sequencer = state.raft_status_fn.get().is_some_and(|status| { + status() + .iter() + .any(|group| group.group_id == SEQUENCER_GROUP_ID) + }); + let Some(registry) = state.calvin_completion_registry.get() else { + return u64::MAX; + }; + if !hosts_sequencer { + // No sequencer replica runs here, so no local scheduler applies a + // Calvin write and none can be planned against. + return u64::MAX; + } + let applied = state.applied_index_watcher(SEQUENCER_GROUP_ID).current(); + let fence = &state.authorization_fence; + fence + .calvin_acks() + .covered_through(registry, fence.calvin_mirrors(), applied, |vshard_id| { + group_of_vshard(state, vshard_id).is_ok_and(|g| hosts_group(state, g)) + }) +} + +/// This node's coverage, confirmed through the permission step. +/// +/// The metadata group is always current. The event-backed groups are taken +/// from this snapshot when the permission step reaches it within `wait`; +/// otherwise `previous` stands for them, which an earlier call confirmed. +pub async fn confirmed_coverage( + state: &SharedState, + previous: &[GroupCoverage], + wait: Duration, +) -> crate::Result> { + let raw = read_raw(state); + apply_committed_tree_defs(state).await; + if state.authorization_fence.take_snapshot_installed() { + // The snapshot rows emitted no events. A reload after the indexes + // were read holds them. + reload::reload_all(state, None).await?; + } + reload::reload_if_stale(state).await?; + + let caught_up = permission_step_reaches_now(state, wait).await?; + let mut coverage = vec![GroupCoverage { + group_id: METADATA_GROUP_ID, + through: raw.metadata, + }]; + if caught_up { + coverage.extend(raw.event_backed); + } else { + coverage.extend( + previous + .iter() + .filter(|report| report.group_id != METADATA_GROUP_ID) + .copied(), + ); + } + Ok(coverage) +} + +/// Whether the permission cache reflects every event the cores emitted +/// before this call, within `wait`. A core that lost an event is reloaded. +pub(crate) async fn permission_step_reaches_now( + state: &SharedState, + wait: Duration, +) -> crate::Result { + let fence = &state.authorization_fence; + let Some(targets) = fence.emitted_snapshot() else { + // No permission step runs, so only a reload reflects the writes. + reload::reload_all(state, None).await?; + return Ok(true); + }; + let until = Instant::now() + wait; + loop { + let notified = fence.permission_applied().notified(); + tokio::pin!(notified); + notified.as_mut().enable(); + let needs_reload = { + let cache = state.permission_cache.read().await; + if cache.progress().caught_up(&targets) { + return Ok(true); + } + cache.progress().needs_reload_for(&targets) + }; + if needs_reload { + reload::reload_all(state, Some(&targets)).await?; + continue; + } + let now = Instant::now(); + if now >= until { + return Ok(false); + } + let _ = tokio::time::timeout(until - now, notified).await; + } +} + +/// Whether the permission step covers every event the cores emitted before +/// this call, within `wait`. A writer holding its acknowledgement calls this. +/// +/// It never reloads. A cache that only a reload can bring to the targets is +/// stale, and [`reload::reload_if_stale`] reloads it before the next +/// statement plans. That reload reads each core after the write applied, so +/// the write already binds every later plan. Before the Event Plane starts no +/// permission step counts writes, so the cache is marked for a reload. +pub(crate) async fn permission_step_covers_now(state: &SharedState, wait: Duration) -> bool { + let fence = &state.authorization_fence; + let Some(targets) = fence.emitted_snapshot() else { + state + .permission_cache + .write() + .await + .progress_mut() + .mark_reload_needed(); + return true; + }; + let until = Instant::now() + wait; + loop { + let notified = fence.permission_applied().notified(); + tokio::pin!(notified); + notified.as_mut().enable(); + { + let cache = state.permission_cache.read().await; + let progress = cache.progress(); + if progress.caught_up(&targets) || progress.needs_reload_for(&targets) { + return true; + } + } + let now = Instant::now(); + if now >= until { + return false; + } + let _ = tokio::time::timeout(until - now, notified).await; + } +} diff --git a/nodedb/src/control/security/auth_lease/holder.rs b/nodedb/src/control/security/auth_lease/holder.rs new file mode 100644 index 000000000..0d940b3e4 --- /dev/null +++ b/nodedb/src/control/security/auth_lease/holder.rs @@ -0,0 +1,76 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The holder side of this node's authorization lease. +//! +//! A statement is planned against local authorization state only while the +//! lease is valid. The lease ends on this node's clock before it ends on the +//! leader's (see [`super::timing`]), so once the leader treats it as expired +//! no statement here can still plan under it. +//! +//! A node that leads the metadata group as its only voter also plans under a +//! pinned lease, which has no expiry (see [`super::status`]). + +use std::sync::Mutex; +use std::time::Instant; + +/// The end of this node's lease, if it holds one. +#[derive(Debug, Default)] +pub struct LeaseHolder { + valid_until: Mutex>, +} + +impl LeaseHolder { + /// Extend the lease to `until`. A grant never shortens a lease already + /// held: the leader granted each one against the state it covers. + pub fn install(&self, until: Instant) { + let mut valid_until = self.valid_until.lock().unwrap_or_else(|p| p.into_inner()); + if valid_until.is_none_or(|current| current < until) { + *valid_until = Some(until); + } + } + + /// Whether the lease is valid at `now`. + pub fn is_valid_at(&self, now: Instant) -> bool { + self.valid_until + .lock() + .unwrap_or_else(|p| p.into_inner()) + .is_some_and(|until| now < until) + } + + /// When the lease ends, if one was granted. + pub fn valid_until(&self) -> Option { + *self.valid_until.lock().unwrap_or_else(|p| p.into_inner()) + } +} + +#[cfg(test)] +mod tests { + use std::time::Duration; + + use super::*; + use crate::control::security::auth_lease::LeaseTiming; + + #[test] + fn a_lease_is_valid_until_its_margin_adjusted_end() { + let timing = LeaseTiming::from_raft(Duration::from_millis(150), Duration::from_millis(50)) + .expect("timing"); + let holder = LeaseHolder::default(); + let sent_at = Instant::now(); + assert!(!holder.is_valid_at(sent_at)); + + holder.install(timing.holder_expiry(sent_at, timing.lease)); + assert!(holder.is_valid_at(sent_at + Duration::from_millis(99))); + // The leader's lease still runs at 100ms, but the holder's has ended. + assert!(!holder.is_valid_at(sent_at + Duration::from_millis(100))); + assert!(!holder.is_valid_at(sent_at + timing.lease)); + } + + #[test] + fn an_older_grant_never_shortens_the_lease() { + let holder = LeaseHolder::default(); + let now = Instant::now(); + holder.install(now + Duration::from_millis(200)); + holder.install(now + Duration::from_millis(100)); + assert!(holder.is_valid_at(now + Duration::from_millis(150))); + } +} diff --git a/nodedb/src/control/security/auth_lease/leadership.rs b/nodedb/src/control/security/auth_lease/leadership.rs new file mode 100644 index 000000000..086537480 --- /dev/null +++ b/nodedb/src/control/security/auth_lease/leadership.rs @@ -0,0 +1,75 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Who leads the metadata group, from this node's Raft status. + +use std::collections::BTreeSet; +use std::time::Duration; + +use nodedb_cluster::{METADATA_GROUP_ID, RaftRpc}; + +use crate::control::cluster::warm_peers::register_peers_from_topology; +use crate::control::state::SharedState; + +/// The metadata group's leader and term, as this node sees them. A leader +/// id of `0` means none is known. +pub(crate) fn metadata_leader(state: &SharedState) -> Option<(u64, u64)> { + let status = state.raft_status_fn.get()?; + status() + .into_iter() + .find(|group| group.group_id == METADATA_GROUP_ID) + .map(|group| (group.leader_id, group.term)) +} + +/// The term this node leads the metadata group in, if it does. +pub(crate) fn leading_term(state: &SharedState) -> Option { + metadata_leader(state) + .filter(|(leader_id, _)| *leader_id == state.node_id) + .map(|(_, term)| term) +} + +/// The term this node leads the metadata group in, when it is also the +/// group's only voter. +/// +/// A single-voter group commits a configuration change inside the propose +/// call and applies it at once. A second voter therefore ends this before it +/// can vote. No other node can lead the group while this returns a term. +pub(crate) fn sole_voter_term(state: &SharedState) -> Option { + let status = state.raft_status_fn.get()?; + status() + .into_iter() + .find(|group| group.group_id == METADATA_GROUP_ID) + .filter(|group| { + group.role == "Leader" && group.leader_id == state.node_id && group.member_count == 1 + }) + .map(|group| group.term) +} + +/// The leader hint to send back with a refusal. +pub(crate) fn leader_hint(state: &SharedState) -> Option { + metadata_leader(state) + .map(|(leader_id, _)| leader_id) + .filter(|leader_id| *leader_id != 0) +} + +/// Send `rpc` to the metadata leader `leader_id` and return its answer. +pub(crate) async fn send_to_leader( + state: &SharedState, + leader_id: u64, + rpc: RaftRpc, + timeout: Duration, +) -> crate::Result { + let Some(transport) = state.cluster_transport.as_ref() else { + return Err(crate::Error::Internal { + detail: "authorization lease: no cluster transport on this node".into(), + }); + }; + let mut targets = BTreeSet::new(); + targets.insert(leader_id); + register_peers_from_topology(state, transport, &targets); + transport + .send_rpc_with_read_timeout(leader_id, rpc, timeout) + .await + .map_err(|e| crate::Error::Internal { + detail: format!("authorization lease: rpc to metadata leader {leader_id}: {e}"), + }) +} diff --git a/nodedb/src/control/security/auth_lease/mod.rs b/nodedb/src/control/security/auth_lease/mod.rs new file mode 100644 index 000000000..907385bad --- /dev/null +++ b/nodedb/src/control/security/auth_lease/mod.rs @@ -0,0 +1,22 @@ +// SPDX-License-Identifier: BUSL-1.1 + +pub mod barrier; +pub mod calvin_acks; +pub mod coverage; +pub mod holder; +pub mod leadership; +pub mod renew_loop; +pub mod service; +pub mod status; +pub mod table; +pub mod timing; +pub mod withheld_warn; + +pub use barrier::{ + authorization_barrier, await_local_coverage, block_on_barrier, calvin_write_barrier, +}; +pub use calvin_acks::CalvinAckCoverage; +pub use holder::LeaseHolder; +pub use service::LeaderLeaseService; +pub use status::{LeaseStatus, await_planning_admitted, lease_status}; +pub use timing::LeaseTiming; diff --git a/nodedb/src/control/security/auth_lease/renew_loop.rs b/nodedb/src/control/security/auth_lease/renew_loop.rs new file mode 100644 index 000000000..10a9f879e --- /dev/null +++ b/nodedb/src/control/security/auth_lease/renew_loop.rs @@ -0,0 +1,123 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The holder's renewal loop. +//! +//! Every renewal interval the node computes its confirmed coverage (see +//! [`super::coverage`]) and sends it to the metadata leader. A grant extends +//! the lease from the moment the request left. A withheld or failed renewal +//! extends nothing, so the lease lapses unless a later renewal succeeds. + +use std::sync::Arc; +use std::time::Instant; + +use nodedb_cluster::{ + AuthLeaseRenewOutcome, AuthLeaseRenewRequest, AuthLeaseRenewResponse, GroupCoverage, RaftRpc, +}; + +use crate::control::shutdown::ShutdownReceiver; +use crate::control::state::SharedState; + +use super::coverage::confirmed_coverage; +use super::leadership::{metadata_leader, send_to_leader}; +use super::timing::LeaseTiming; + +/// Renew this node's lease until shutdown. +pub async fn run_renew_loop( + state: Arc, + timing: LeaseTiming, + mut shutdown: ShutdownReceiver, +) { + let mut confirmed: Vec = Vec::new(); + loop { + // Shutdown ends a renewal in flight too: a coverage read or a renew + // RPC holds the node's state until it returns. + tokio::select! { + _ = renew_round(&state, timing, &mut confirmed) => {} + _ = shutdown.wait_cancelled() => return, + } + tokio::select! { + _ = tokio::time::sleep(timing.renew_every) => {} + _ = shutdown.wait_cancelled() => return, + } + } +} + +/// Compute this node's confirmed coverage and renew the lease with it. +async fn renew_round(state: &SharedState, timing: LeaseTiming, confirmed: &mut Vec) { + match confirmed_coverage(state, confirmed, timing.renew_every).await { + Ok(coverage) => { + *confirmed = coverage; + renew_once(state, timing, confirmed).await; + } + Err(error) => { + tracing::warn!(%error, "authorization lease: coverage could not be computed"); + } + } +} + +/// Send one renewal and install a granted lease. +async fn renew_once(state: &SharedState, timing: LeaseTiming, coverage: &[GroupCoverage]) { + let Some((leader_id, _)) = metadata_leader(state).filter(|(leader, _)| *leader != 0) else { + return; + }; + let request = AuthLeaseRenewRequest { + node_id: state.node_id, + coverage: coverage.to_vec(), + }; + let sent_at = Instant::now(); + let response = if leader_id == state.node_id { + match state.authorization_fence.leader() { + Some(service) => service.renew_lease(request).await, + None => return, + } + } else { + match send_to_leader( + state, + leader_id, + RaftRpc::AuthLeaseRenewRequest(request), + timing.lease, + ) + .await + { + Ok(RaftRpc::AuthLeaseRenewResponse(response)) => response, + Ok(other) => { + tracing::warn!( + leader_id, + "authorization lease: unexpected renewal reply {other:?}" + ); + return; + } + Err(error) => { + tracing::debug!(%error, "authorization lease: renewal not delivered"); + return; + } + } + }; + install(state, timing, sent_at, response); +} + +fn install( + state: &SharedState, + timing: LeaseTiming, + sent_at: Instant, + response: AuthLeaseRenewResponse, +) { + match response.outcome { + AuthLeaseRenewOutcome::Granted { lease_ms } => { + let granted = std::time::Duration::from_millis(lease_ms); + state + .authorization_fence + .holder() + .install(timing.holder_expiry(sent_at, granted)); + } + AuthLeaseRenewOutcome::Withheld => { + tracing::debug!("authorization lease: renewal withheld until coverage catches up"); + } + AuthLeaseRenewOutcome::NotLeader { leader_hint } => { + tracing::debug!( + ?leader_hint, + "authorization lease: renewal reached a non-leader" + ); + } + } +} diff --git a/nodedb/src/control/security/auth_lease/service.rs b/nodedb/src/control/security/auth_lease/service.rs new file mode 100644 index 000000000..98f8abad4 --- /dev/null +++ b/nodedb/src/control/security/auth_lease/service.rs @@ -0,0 +1,346 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The metadata leader's side of the authorization lease. +//! +//! Only the current metadata leader answers renewals and barriers. Each +//! leadership term starts a fresh [`LeaseTable`], whose floors are loaded +//! before anything is granted or released: +//! +//! - **Metadata group:** the leader's own confirmed read index. +//! - **Sequencer group and every group homing a tree source:** a read index +//! from each group's leader. +//! +//! A read index is at or above every entry committed before it was taken, so +//! the floors cover every change acknowledged in an earlier term. +//! +//! A grant and a release are answered only after the leader confirms its +//! leadership against a quorum, taken after the decision. A leader deposed +//! meanwhile answers `NotLeader`, so no lease it grants and no barrier it +//! releases outlives its term unseen. +//! +//! A leader that is the only voter of the metadata group pins its own lease +//! (see [`super::table`]). No other node can lead the group then, so every +//! barrier releases here, and each one waits for this node's coverage. +//! Planning on this node then needs no lease that expires on the clock. + +use std::collections::HashSet; +use std::sync::{Mutex, Weak}; +use std::time::{Duration, Instant}; + +use futures::future::try_join_all; +use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use nodedb_cluster::{ + AuthBarrierOutcome, AuthBarrierRequest, AuthBarrierResponse, AuthLeaseRenewOutcome, + AuthLeaseRenewRequest, AuthLeaseRenewResponse, GroupCoverage, METADATA_GROUP_ID, +}; +use tokio::sync::Notify; + +use crate::control::security::auth_fence::cluster::{ + confirmed_read_index, group_of_vshard, wait_applied, +}; +use crate::control::security::auth_fence::view::apply_committed_tree_defs; +use crate::control::state::SharedState; + +use super::leadership::{leader_hint, leading_term, sole_voter_term}; +use super::table::{BarrierState, LeaseTable, RenewDecision}; +use super::timing::LeaseTiming; +use super::withheld_warn::WithheldWarnings; + +/// Answers lease renewals and barriers while this node leads the metadata +/// group. +pub struct LeaderLeaseService { + /// Held weakly: the service lives on `SharedState`. + state: Weak, + timing: LeaseTiming, + table: Mutex>, + /// Woken on every renewal and floor load, for waiting barriers. + changed: Notify, + /// One floor load at a time. + floors_loading: tokio::sync::Mutex<()>, + /// Rate limit of the withheld-renewal warning. + withheld_warnings: WithheldWarnings, +} + +impl std::fmt::Debug for LeaderLeaseService { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("LeaderLeaseService") + .field("timing", &self.timing) + .finish_non_exhaustive() + } +} + +impl LeaderLeaseService { + pub fn new(state: Weak, timing: LeaseTiming) -> Self { + Self { + state, + timing, + table: Mutex::new(None), + changed: Notify::new(), + floors_loading: tokio::sync::Mutex::new(()), + withheld_warnings: WithheldWarnings::default(), + } + } + + fn table(&self) -> std::sync::MutexGuard<'_, Option> { + self.table.lock().unwrap_or_else(|p| p.into_inner()) + } + + /// Whether the table of `term` pins the lease of `node_id`. + /// + /// The caller passes the term in which it read that this node is the + /// only voter of the metadata group. + pub fn holds_pinned_lease(&self, term: u64, node_id: u64) -> bool { + self.table() + .as_ref() + .is_some_and(|table| table.term() == term && table.is_pinned(node_id)) + } + + /// Run `edit` on the current table, creating one for `term` first. + #[cfg(test)] + pub(crate) fn edit_table(&self, term: u64, now: Instant, edit: impl FnOnce(&mut LeaseTable)) { + let mut table = self.table(); + let table = table.get_or_insert_with(|| LeaseTable::new(term, now)); + edit(table); + } + + /// Make the table of `term` current and load its floors. + async fn table_ready(&self, state: &SharedState, term: u64) -> crate::Result<()> { + { + let mut table = self.table(); + if table.as_ref().is_none_or(|t| t.term() != term) { + *table = Some(LeaseTable::new(term, Instant::now())); + } + if table.as_ref().is_some_and(LeaseTable::floors_ready) { + return Ok(()); + } + } + let _loading = self.floors_loading.lock().await; + if self + .table() + .as_ref() + .is_some_and(|t| t.term() == term && t.floors_ready()) + { + return Ok(()); + } + let floors = self.load_floors(state).await?; + if let Some(table) = self.table().as_mut().filter(|t| t.term() == term) { + table.load_floors(&floors); + } + self.changed.notify_waiters(); + Ok(()) + } + + /// The floors a new term starts from. + async fn load_floors(&self, state: &SharedState) -> crate::Result> { + let timeout = self.timing.lease; + let metadata = confirmed_read_index(state, METADATA_GROUP_ID, timeout).await?; + // The source set comes from tree definitions, which live in the + // metadata group. Apply it through the read index first. + wait_applied(state, METADATA_GROUP_ID, metadata, timeout).await?; + apply_committed_tree_defs(state).await; + + let mut groups: HashSet = HashSet::new(); + for vshard_id in state.authorization_fence.sources().source_vshards() { + groups.insert(group_of_vshard(state, vshard_id)?); + } + groups.insert(SEQUENCER_GROUP_ID); + let group_floors = try_join_all(groups.into_iter().map(|group_id| async move { + confirmed_read_index(state, group_id, timeout) + .await + .map(|through| GroupCoverage { group_id, through }) + })) + .await?; + + let mut floors = vec![GroupCoverage { + group_id: METADATA_GROUP_ID, + through: metadata, + }]; + floors.extend(group_floors); + Ok(floors) + } + + /// Whether this node still leads the metadata group in `term`, confirmed + /// against a quorum now. + async fn confirm_leadership(&self, state: &SharedState, term: u64) -> bool { + confirmed_read_index(state, METADATA_GROUP_ID, self.timing.lease) + .await + .is_ok() + && leading_term(state) == Some(term) + } + + fn not_leader_renewal(state: Option<&SharedState>) -> AuthLeaseRenewResponse { + AuthLeaseRenewResponse { + outcome: AuthLeaseRenewOutcome::NotLeader { + leader_hint: state.and_then(leader_hint), + }, + } + } + + fn not_leader_barrier(state: Option<&SharedState>) -> AuthBarrierResponse { + AuthBarrierResponse { + outcome: AuthBarrierOutcome::NotLeader { + leader_hint: state.and_then(leader_hint), + }, + } + } + + /// Grant or withhold the lease of the node that sent `req`. + pub async fn renew_lease(&self, req: AuthLeaseRenewRequest) -> AuthLeaseRenewResponse { + let Some(state) = self.state.upgrade() else { + return Self::not_leader_renewal(None); + }; + let Some(term) = leading_term(&state) else { + return Self::not_leader_renewal(Some(&state)); + }; + if let Err(error) = self.table_ready(&state, term).await { + if self + .withheld_warnings + .should_warn(req.node_id, Instant::now()) + { + tracing::warn!( + node_id = req.node_id, + %error, + "authorization lease: renewal withheld: the floors of this term are not loaded" + ); + } + return AuthLeaseRenewResponse { + outcome: AuthLeaseRenewOutcome::Withheld, + }; + } + let sole_voter = sole_voter_term(&state) == Some(term); + let (decision, shortfall) = { + let mut table = self.table(); + match table.as_mut().filter(|t| t.term() == term) { + Some(table) => { + table.observe_sole_voter(sole_voter); + let decision = table.renew( + req.node_id, + &req.coverage, + Instant::now(), + self.timing.lease, + ); + if decision == RenewDecision::Granted + && sole_voter + && req.node_id == state.node_id + { + table.pin(req.node_id); + } + let shortfall = match decision { + RenewDecision::Withheld => table.shortfall(&req.coverage), + RenewDecision::Granted => Vec::new(), + }; + (decision, shortfall) + } + None => return Self::not_leader_renewal(Some(&state)), + } + }; + if decision == RenewDecision::Withheld + && self + .withheld_warnings + .should_warn(req.node_id, Instant::now()) + { + // Each entry: (group_id, floor, reported coverage or None). + tracing::warn!( + node_id = req.node_id, + term, + short_groups = ?shortfall, + "authorization lease: renewal withheld: the node's coverage is below the floor \ + of each group listed as (group_id, floor, reported)" + ); + } + self.changed.notify_waiters(); + let outcome = match decision { + RenewDecision::Withheld => AuthLeaseRenewOutcome::Withheld, + RenewDecision::Granted if self.confirm_leadership(&state, term).await => { + AuthLeaseRenewOutcome::Granted { + lease_ms: u64::try_from(self.timing.lease.as_millis()).unwrap_or(u64::MAX), + } + } + RenewDecision::Granted => { + return Self::not_leader_renewal(Some(&state)); + } + }; + AuthLeaseRenewResponse { outcome } + } + + /// Answer once no lease holder can plan against state older than the + /// request's targets. + pub async fn hold_barrier(&self, req: AuthBarrierRequest) -> AuthBarrierResponse { + let started = Instant::now(); + let deadline = started + Duration::from_millis(req.timeout_ms); + let timed_out = || AuthBarrierResponse { + outcome: AuthBarrierOutcome::Timeout { + waited_ms: u64::try_from(started.elapsed().as_millis()).unwrap_or(u64::MAX), + }, + }; + loop { + let Some(state) = self.state.upgrade() else { + return Self::not_leader_barrier(None); + }; + let Some(term) = leading_term(&state) else { + return Self::not_leader_barrier(Some(&state)); + }; + if let Err(error) = self.table_ready(&state, term).await { + tracing::debug!(%error, "authorization barrier: floors not loaded yet"); + if Instant::now() >= deadline { + return timed_out(); + } + tokio::time::sleep(self.timing.renew_every).await; + continue; + } + let sole_voter = sole_voter_term(&state) == Some(term); + let notified = self.changed.notified(); + tokio::pin!(notified); + notified.as_mut().enable(); + let status = { + let mut table = self.table(); + match table.as_mut().filter(|t| t.term() == term) { + Some(table) => { + table.observe_sole_voter(sole_voter); + table.raise_floors(&req.targets); + table.barrier(&req.targets, Instant::now(), self.timing.lease) + } + None => continue, + } + }; + match status { + BarrierState::Released => { + return if self.confirm_leadership(&state, term).await { + AuthBarrierResponse { + outcome: AuthBarrierOutcome::Released, + } + } else { + Self::not_leader_barrier(Some(&state)) + }; + } + BarrierState::NotReady => tokio::time::sleep(self.timing.renew_every).await, + BarrierState::Waiting { until } => { + // A pinned holder releases the barrier by renewing, which + // wakes `notified`. A check every renewal interval also + // sees the pin end when a second voter joins. + let until = until.unwrap_or_else(|| Instant::now() + self.timing.renew_every); + let wake = tokio::time::Instant::from_std(until.min(deadline)); + drop(state); + tokio::select! { + _ = notified => {} + _ = tokio::time::sleep_until(wake) => {} + } + } + } + if Instant::now() >= deadline { + return timed_out(); + } + } + } +} + +#[async_trait::async_trait] +impl nodedb_cluster::AuthLeaseService for LeaderLeaseService { + async fn renew(&self, req: AuthLeaseRenewRequest) -> AuthLeaseRenewResponse { + self.renew_lease(req).await + } + + async fn barrier(&self, req: AuthBarrierRequest) -> AuthBarrierResponse { + self.hold_barrier(req).await + } +} diff --git a/nodedb/src/control/security/auth_lease/status.rs b/nodedb/src/control/security/auth_lease/status.rs new file mode 100644 index 000000000..bec4b3a68 --- /dev/null +++ b/nodedb/src/control/security/auth_lease/status.rs @@ -0,0 +1,301 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! This node's lease state, as `/healthz` and the metrics endpoint report it. +//! +//! Boot waits for the first lease before the gates open. A node that later +//! loses its lease refuses every permission-checked statement until it +//! renews. It still serves every other path, so readiness reports it as +//! degraded, with the reason, for as long as it holds no valid lease. + +use std::fmt::Write as _; +use std::time::{Duration, Instant}; + +use crate::control::security::auth_fence::AuthorizationFence; +use crate::control::state::SharedState; + +use super::leadership::sole_voter_term; + +/// Whether this node can plan permission-checked statements now. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum LeaseStatus { + /// A single node: no lease exists and none is needed. + NotRequired, + /// The lease is valid for `remaining`. + Valid { remaining: Duration }, + /// This node leads the metadata group as its only voter and holds a + /// pinned lease, which has no expiry (see [`super::table`]). + SoleVoter, + /// No valid lease. `expired_for` is how long ago the last one ended, or + /// `None` when no lease was ever granted. + Invalid { expired_for: Option }, +} + +impl LeaseStatus { + /// Whether this node can plan permission-checked statements. + pub fn admits_planning(&self) -> bool { + !matches!(self, Self::Invalid { .. }) + } +} + +/// The lease state of this node at `now`. +pub fn lease_status(state: &SharedState, now: Instant) -> LeaseStatus { + lease_status_of( + &state.authorization_fence, + || sole_voter_term(state), + state.node_id, + now, + ) +} + +/// The lease state of node `node_id` at `now`. +/// +/// `sole_voter_term` returns the term in which this node leads the metadata +/// group as its only voter. It runs only when the bounded lease is not +/// valid, so the planning path reads the Raft status only after a lapse. +pub(crate) fn lease_status_of( + fence: &AuthorizationFence, + sole_voter_term: impl FnOnce() -> Option, + node_id: u64, + now: Instant, +) -> LeaseStatus { + if fence.timing().is_none() { + return LeaseStatus::NotRequired; + } + let valid_until = fence.holder().valid_until(); + if let Some(until) = valid_until.filter(|until| now < *until) { + return LeaseStatus::Valid { + remaining: until - now, + }; + } + let pinned = sole_voter_term().is_some_and(|term| { + fence + .leader() + .is_some_and(|service| service.holds_pinned_lease(term, node_id)) + }); + if pinned { + return LeaseStatus::SoleVoter; + } + LeaseStatus::Invalid { + expired_for: valid_until.map(|until| now.saturating_duration_since(until)), + } +} + +/// Wait until this node can plan permission-checked statements, polling +/// every `poll`, or refuse once `timeout` passes. +pub async fn await_planning_admitted( + state: &SharedState, + timeout: Duration, + poll: Duration, +) -> crate::Result<()> { + let deadline = Instant::now() + timeout; + while !lease_status(state, Instant::now()).admits_planning() { + if Instant::now() >= deadline { + return Err(crate::Error::AuthorizationStateBehind { + detail: format!("no authorization lease was granted within {timeout:?}"), + }); + } + tokio::time::sleep(poll).await; + } + Ok(()) +} + +/// Append the lease gauges. A single node holds no lease and emits none. +/// +/// - `nodedb_authorization_lease_valid`: 1 while the lease is valid, else 0. +/// - `nodedb_authorization_lease_remaining_seconds`: time left on the lease, +/// 0 without one, `+Inf` for a pinned lease. +pub fn render_prometheus(state: &SharedState, out: &mut String) { + render_status(lease_status(state, Instant::now()), out); +} + +fn render_status(status: LeaseStatus, out: &mut String) { + let (valid, remaining) = match status { + LeaseStatus::NotRequired => return, + LeaseStatus::Valid { remaining } => (1, remaining.as_secs_f64()), + LeaseStatus::SoleVoter => (1, f64::INFINITY), + LeaseStatus::Invalid { .. } => (0, 0.0), + }; + let _ = writeln!( + out, + "# HELP nodedb_authorization_lease_valid Whether this node holds a valid \ + authorization lease and can plan permission-checked statements\n\ + # TYPE nodedb_authorization_lease_valid gauge\n\ + nodedb_authorization_lease_valid {valid}\n\ + # HELP nodedb_authorization_lease_remaining_seconds Time left on this \ + node's authorization lease\n\ + # TYPE nodedb_authorization_lease_remaining_seconds gauge\n\ + nodedb_authorization_lease_remaining_seconds {}", + prometheus_float(remaining) + ); +} + +/// A gauge value in the Prometheus text format, which spells infinity `+Inf`. +fn prometheus_float(value: f64) -> String { + if value.is_infinite() { + "+Inf".to_string() + } else { + value.to_string() + } +} + +#[cfg(test)] +mod tests { + use std::sync::{Arc, Weak}; + + use nodedb_cluster::GroupCoverage; + + use super::*; + use crate::control::security::auth_lease::table::RenewDecision; + use crate::control::security::auth_lease::{LeaderLeaseService, LeaseTiming}; + use crate::control::security::permission_tree::SourceIndex; + + const NODE: u64 = 1; + const TERM: u64 = 3; + + fn timing() -> LeaseTiming { + LeaseTiming::from_raft(Duration::from_millis(1000), Duration::from_millis(100)) + .expect("timing") + } + + fn fence() -> AuthorizationFence { + AuthorizationFence::new(Arc::new(SourceIndex::default())) + } + + /// A fence whose leader service granted this node's lease at `granted_at` + /// in `TERM`, pinned when `sole_voter` holds, as one renewal round does. + fn fence_granted_at(granted_at: Instant, sole_voter: bool) -> AuthorizationFence { + let fence = fence(); + let timing = timing(); + assert!(fence.install_timing(timing)); + let service = Arc::new(LeaderLeaseService::new(Weak::new(), timing)); + let coverage = [GroupCoverage { + group_id: 0, + through: 10, + }]; + service.edit_table(TERM, granted_at, |table| { + table.load_floors(&coverage); + table.observe_sole_voter(sole_voter); + assert_eq!( + table.renew(NODE, &coverage, granted_at, timing.lease), + RenewDecision::Granted + ); + if sole_voter { + table.pin(NODE); + } + }); + assert!(fence.install_leader(service)); + fence + .holder() + .install(timing.holder_expiry(granted_at, timing.lease)); + fence + } + + #[test] + fn a_node_without_timing_needs_no_lease() { + let fence = fence(); + let status = lease_status_of(&fence, || None, NODE, Instant::now()); + assert_eq!(status, LeaseStatus::NotRequired); + let mut out = String::new(); + render_status(status, &mut out); + assert!(out.is_empty()); + } + + #[test] + fn a_lapsed_lease_is_invalid_and_reports_zero() { + let fence = fence(); + assert!(fence.install_timing(timing())); + let now = Instant::now(); + assert_eq!( + lease_status_of(&fence, || None, NODE, now), + LeaseStatus::Invalid { expired_for: None } + ); + + fence.holder().install(now + Duration::from_secs(2)); + let status = lease_status_of(&fence, || None, NODE, now); + assert_eq!( + status, + LeaseStatus::Valid { + remaining: Duration::from_secs(2) + } + ); + let mut out = String::new(); + render_status(status, &mut out); + assert!(out.contains("nodedb_authorization_lease_valid 1")); + + let later = now + Duration::from_secs(3); + let status = lease_status_of(&fence, || None, NODE, later); + assert_eq!( + status, + LeaseStatus::Invalid { + expired_for: Some(Duration::from_secs(1)) + } + ); + assert!(!status.admits_planning()); + } + + /// The renewal task of a sole-voter node is starved for a whole lease + /// duration. Its bounded lease lapses, yet the node still plans: its + /// lease is pinned, and every barrier waits for its coverage. + #[test] + fn a_starved_renewal_on_a_sole_voter_keeps_planning() { + let granted_at = Instant::now(); + let fence = fence_granted_at(granted_at, true); + let starved = granted_at + timing().lease; + assert!( + !fence.holder().is_valid_at(starved), + "the bounded lease lapsed" + ); + + let status = lease_status_of(&fence, || Some(TERM), NODE, starved); + assert_eq!(status, LeaseStatus::SoleVoter); + assert!(status.admits_planning()); + let mut out = String::new(); + render_status(status, &mut out); + assert!(out.contains("nodedb_authorization_lease_valid 1")); + assert!(out.contains("nodedb_authorization_lease_remaining_seconds +Inf")); + } + + /// A cluster node whose renewal lapses cannot confirm it holds the latest + /// authorization state. It refuses, whatever its table once recorded. + #[test] + fn a_node_with_a_peer_voter_refuses_once_its_lease_lapses() { + let granted_at = Instant::now(); + let starved = granted_at + timing().lease; + + // A lease granted while a peer voter existed was never pinned. + let fence = fence_granted_at(granted_at, false); + let status = lease_status_of(&fence, || None, NODE, starved); + assert!(matches!(status, LeaseStatus::Invalid { .. })); + assert!(!status.admits_planning()); + + // A pinned lease stops counting the moment a peer voter joins. + let fence = fence_granted_at(granted_at, true); + let status = lease_status_of(&fence, || None, NODE, starved); + assert!(!status.admits_planning()); + } + + /// A barrier that runs once a peer voter joined removes the pin. The node + /// then refuses even after it is the only voter again, until a new grant. + #[test] + fn a_pin_removed_by_a_barrier_stays_removed() { + let granted_at = Instant::now(); + let fence = fence_granted_at(granted_at, true); + let starved = granted_at + timing().lease; + fence + .leader() + .expect("leader service") + .edit_table(TERM, starved, |table| table.observe_sole_voter(false)); + let status = lease_status_of(&fence, || Some(TERM), NODE, starved); + assert!(!status.admits_planning()); + } + + /// A pin belongs to the leadership term that granted it. + #[test] + fn a_pin_from_an_earlier_term_does_not_count() { + let granted_at = Instant::now(); + let fence = fence_granted_at(granted_at, true); + let starved = granted_at + timing().lease; + let status = lease_status_of(&fence, || Some(TERM + 1), NODE, starved); + assert!(!status.admits_planning()); + } +} diff --git a/nodedb/src/control/security/auth_lease/table.rs b/nodedb/src/control/security/auth_lease/table.rs new file mode 100644 index 000000000..161f6152a --- /dev/null +++ b/nodedb/src/control/security/auth_lease/table.rs @@ -0,0 +1,412 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The metadata leader's lease table. +//! +//! One table exists per leadership term. It records, for each node that +//! renewed with this leader, when its lease ends and what its last report +//! covered. It also records the floors: per group, the highest index of any +//! authorization change a barrier registered. A lease is granted only to a +//! report that covers every floor. +//! +//! A barrier releases once, for every target, each node holding an unexpired +//! lease reported coverage of it. Leases granted by an earlier leader are not +//! in the table. They end within one lease duration of this leader taking +//! over, so a barrier also waits until then. +//! +//! A leader that is the only voter of the metadata group can pin its own +//! lease. A pinned lease has no expiry: every barrier waits for the pinned +//! node's coverage, however late its renewal runs. The caller reports on +//! every renewal and barrier whether the leader is still the only voter. The +//! first report that it is not removes the pin. +//! +//! The table is pure: callers pass the clock, so every rule is testable. + +use std::collections::HashMap; +use std::time::{Duration, Instant}; + +use nodedb_cluster::GroupCoverage; + +/// What a node's last renewal reported and when its lease ends. +#[derive(Debug, Default)] +struct HolderRecord { + /// End of the lease this leader granted, if any. + expires_at: Option, + /// Coverage by group, from the last renewal. + coverage: HashMap, +} + +impl HolderRecord { + fn covers(&self, group_id: u64, index: u64) -> bool { + self.coverage + .get(&group_id) + .is_some_and(|through| *through >= index) + } +} + +/// The answer to a renewal. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum RenewDecision { + Granted, + Withheld, +} + +/// Where a barrier stands. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum BarrierState { + /// No node can plan against state older than the targets. + Released, + /// Waiting for a report or an expiry. Nothing changes on its own before + /// `until`, except a renewal. `until` is `None` while the pinned holder + /// has not covered the targets: only its renewal, or the end of the pin, + /// can release the barrier then. + Waiting { until: Option }, + /// The floors of this term are not loaded yet. + NotReady, +} + +/// The lease table of one leadership term. +#[derive(Debug)] +pub struct LeaseTable { + term: u64, + leader_since: Instant, + floors_ready: bool, + floors: HashMap, + holders: HashMap, + /// The holder whose lease has no expiry while the leader stays the only + /// voter of the metadata group. + pinned: Option, +} + +impl LeaseTable { + /// A table for `term`, whose leadership this node observed at `now`. + pub fn new(term: u64, now: Instant) -> Self { + Self { + term, + leader_since: now, + floors_ready: false, + floors: HashMap::new(), + holders: HashMap::new(), + pinned: None, + } + } + + pub fn term(&self) -> u64 { + self.term + } + + pub fn floors_ready(&self) -> bool { + self.floors_ready + } + + /// Load the floors this term starts from: an index per group at or above + /// every change acknowledged before the term. + pub fn load_floors(&mut self, floors: &[GroupCoverage]) { + self.raise_floors(floors); + self.floors_ready = true; + } + + /// Raise the floors to cover `targets`. + pub fn raise_floors(&mut self, targets: &[GroupCoverage]) { + for target in targets { + let floor = self.floors.entry(target.group_id).or_insert(0); + *floor = (*floor).max(target.through); + } + } + + /// Record a renewal from `node_id` and decide on its lease. + pub fn renew( + &mut self, + node_id: u64, + coverage: &[GroupCoverage], + now: Instant, + lease: Duration, + ) -> RenewDecision { + let record = self.holders.entry(node_id).or_default(); + record.coverage = coverage + .iter() + .map(|report| (report.group_id, report.through)) + .collect(); + let covered = self + .floors + .iter() + .all(|(group_id, floor)| record.covers(*group_id, *floor)); + if !self.floors_ready || !covered { + return RenewDecision::Withheld; + } + record.expires_at = Some(now + lease); + RenewDecision::Granted + } + + /// Record whether the leader is still the only voter of the metadata + /// group. A leader that is not removes the pin. + pub fn observe_sole_voter(&mut self, sole_voter: bool) { + if !sole_voter { + self.pinned = None; + } + } + + /// Pin the lease of `node_id`, which a renewal just granted while the + /// leader was the only voter. + pub fn pin(&mut self, node_id: u64) { + self.pinned = Some(node_id); + } + + /// Whether the lease of `node_id` is pinned. + pub fn is_pinned(&self, node_id: u64) -> bool { + self.pinned == Some(node_id) + } + + /// Every group whose floor `coverage` does not reach, as + /// `(group_id, floor, reported)`. `reported` is `None` for a group the + /// report omits. + pub fn shortfall(&self, coverage: &[GroupCoverage]) -> Vec<(u64, u64, Option)> { + let mut short: Vec<(u64, u64, Option)> = self + .floors + .iter() + .filter_map(|(group_id, floor)| { + let reported = coverage + .iter() + .find(|report| report.group_id == *group_id) + .map(|report| report.through); + (reported.is_none_or(|through| through < *floor)) + .then_some((*group_id, *floor, reported)) + }) + .collect(); + short.sort_unstable(); + short + } + + /// Where a barrier on `targets` stands at `now`. + pub fn barrier( + &self, + targets: &[GroupCoverage], + now: Instant, + lease: Duration, + ) -> BarrierState { + if !self.floors_ready { + return BarrierState::NotReady; + } + let mut until: Option = None; + let mut wait_for = |instant: Instant| { + until = Some(until.map_or(instant, |current: Instant| current.min(instant))); + }; + let mut wait_for_pinned = false; + let earlier_leases_end = self.leader_since + lease; + if now < earlier_leases_end { + wait_for(earlier_leases_end); + } + for (node_id, record) in &self.holders { + let pinned = self.pinned == Some(*node_id); + let live_until = record.expires_at.filter(|end| *end > now); + if !pinned && live_until.is_none() { + continue; + } + let covered = targets + .iter() + .all(|target| record.covers(target.group_id, target.through)); + if covered { + continue; + } + match live_until { + Some(expires_at) if !pinned => wait_for(expires_at), + _ => wait_for_pinned = true, + } + } + if wait_for_pinned { + return BarrierState::Waiting { until: None }; + } + match until { + Some(until) => BarrierState::Waiting { until: Some(until) }, + None => BarrierState::Released, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + const LEASE: Duration = Duration::from_millis(150); + + fn cover(group_id: u64, through: u64) -> GroupCoverage { + GroupCoverage { group_id, through } + } + + /// A table past the window of earlier leaders' leases. + fn settled_table(start: Instant) -> LeaseTable { + let mut table = LeaseTable::new(4, start); + table.load_floors(&[cover(0, 10)]); + table + } + + #[test] + fn nothing_is_granted_or_released_before_the_floors_load() { + let now = Instant::now(); + let mut table = LeaseTable::new(1, now); + assert_eq!( + table.renew(2, &[cover(0, 99)], now, LEASE), + RenewDecision::Withheld + ); + assert_eq!(table.barrier(&[], now, LEASE), BarrierState::NotReady); + } + + #[test] + fn the_shortfall_names_each_uncovered_floor() { + let mut table = settled_table(Instant::now()); + table.raise_floors(&[cover(7, 19)]); + assert_eq!( + table.shortfall(&[cover(0, 10), cover(7, 12)]), + vec![(7, 19, Some(12))] + ); + assert_eq!(table.shortfall(&[cover(7, 19)]), vec![(0, 10, None)]); + assert!(table.shortfall(&[cover(0, 10), cover(7, 19)]).is_empty()); + } + + #[test] + fn a_report_below_a_floor_is_withheld() { + let start = Instant::now(); + let mut table = settled_table(start); + assert_eq!( + table.renew(2, &[cover(0, 9)], start, LEASE), + RenewDecision::Withheld + ); + assert_eq!( + table.renew(2, &[cover(0, 10)], start, LEASE), + RenewDecision::Granted + ); + // A group the report omits counts as uncovered. + table.raise_floors(&[cover(5, 1)]); + assert_eq!( + table.renew(2, &[cover(0, 10)], start, LEASE), + RenewDecision::Withheld + ); + } + + #[test] + fn a_barrier_waits_out_the_leases_of_earlier_leaders() { + let start = Instant::now(); + let table = settled_table(start); + assert_eq!( + table.barrier(&[cover(0, 5)], start, LEASE), + BarrierState::Waiting { + until: Some(start + LEASE) + } + ); + assert_eq!( + table.barrier(&[cover(0, 5)], start + LEASE, LEASE), + BarrierState::Released + ); + } + + #[test] + fn a_barrier_releases_on_coverage_or_expiry() { + let start = Instant::now(); + let mut table = settled_table(start); + let now = start + LEASE; + assert_eq!( + table.renew(2, &[cover(0, 10)], now, LEASE), + RenewDecision::Granted + ); + let target = [cover(0, 12)]; + table.raise_floors(&target); + // Node 2 holds a lease and has not covered index 12. + assert_eq!( + table.barrier(&target, now, LEASE), + BarrierState::Waiting { + until: Some(now + LEASE) + } + ); + // Its renewal below the new floor is withheld, and its lease is not + // extended. + assert_eq!( + table.renew(2, &[cover(0, 11)], now, LEASE), + RenewDecision::Withheld + ); + // Covering the target releases the barrier at once. + assert_eq!( + table.renew(2, &[cover(0, 12)], now, LEASE), + RenewDecision::Granted + ); + assert_eq!(table.barrier(&target, now, LEASE), BarrierState::Released); + + // A node that never covers releases the barrier when its lease ends. + let later = [cover(0, 20)]; + table.raise_floors(&later); + assert_eq!( + table.barrier(&later, now, LEASE), + BarrierState::Waiting { + until: Some(now + LEASE) + } + ); + assert_eq!( + table.barrier(&later, now + LEASE, LEASE), + BarrierState::Released + ); + // Once expired, it gets no lease back without covering the floor. + assert_eq!( + table.renew(2, &[cover(0, 12)], now + LEASE, LEASE), + RenewDecision::Withheld + ); + } + + /// A pinned lease holds a barrier past its bounded expiry. Its renewal + /// can run arbitrarily late without any barrier releasing behind it. + #[test] + fn a_pinned_lease_holds_a_barrier_past_its_expiry() { + let start = Instant::now(); + let mut table = settled_table(start); + let now = start + LEASE; + table.observe_sole_voter(true); + assert_eq!( + table.renew(1, &[cover(0, 10)], now, LEASE), + RenewDecision::Granted + ); + table.pin(1); + let target = [cover(0, 12)]; + table.raise_floors(&target); + + // Ten lease durations pass with no renewal. The bounded lease ended + // long ago, but the barrier still waits for node 1. + let starved = now + LEASE * 10; + table.observe_sole_voter(true); + assert_eq!( + table.barrier(&target, starved, LEASE), + BarrierState::Waiting { until: None } + ); + + // A late renewal that covers the target releases it. + assert_eq!( + table.renew(1, &[cover(0, 12)], starved, LEASE), + RenewDecision::Granted + ); + assert_eq!( + table.barrier(&target, starved, LEASE), + BarrierState::Released + ); + } + + /// Once the leader is not the only voter, the pin ends and the lease + /// expires on the clock again. + #[test] + fn a_second_voter_ends_the_pin() { + let start = Instant::now(); + let mut table = settled_table(start); + let now = start + LEASE; + assert_eq!( + table.renew(1, &[cover(0, 10)], now, LEASE), + RenewDecision::Granted + ); + table.pin(1); + assert!(table.is_pinned(1)); + let target = [cover(0, 12)]; + table.raise_floors(&target); + + let starved = now + LEASE * 10; + table.observe_sole_voter(false); + assert!(!table.is_pinned(1)); + assert_eq!( + table.barrier(&target, starved, LEASE), + BarrierState::Released + ); + } +} diff --git a/nodedb/src/control/security/auth_lease/timing.rs b/nodedb/src/control/security/auth_lease/timing.rs new file mode 100644 index 000000000..5cebf13ad --- /dev/null +++ b/nodedb/src/control/security/auth_lease/timing.rs @@ -0,0 +1,97 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Lease durations derived from the Raft timing. +//! +//! - **Lease:** the minimum election timeout. A leader that loses its +//! quorum is replaced no sooner than that, the same bound Raft leader +//! leases rely on. +//! - **Skew margin:** the heartbeat interval. The holder ends its lease this +//! much before the leader does, so a holder clock that runs slow by up to +//! the heartbeat-to-election ratio never outlives the leader's view. +//! - **Renewal:** every heartbeat interval. A holder that covers a change +//! renews within one heartbeat, so an acknowledgement waits about one +//! heartbeat when every node is healthy. + +use std::time::{Duration, Instant}; + +/// Lease durations of one cluster. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct LeaseTiming { + /// How long a granted lease runs on the leader's clock. + pub lease: Duration, + /// How much earlier a holder ends its lease than the leader. + pub skew_margin: Duration, + /// How often a holder renews. + pub renew_every: Duration, +} + +impl LeaseTiming { + /// Derive the timing from the Raft election timeout and heartbeat. + pub fn from_raft( + election_timeout_min: Duration, + heartbeat_interval: Duration, + ) -> crate::Result { + if heartbeat_interval.is_zero() || heartbeat_interval >= election_timeout_min { + return Err(crate::Error::Config { + detail: format!( + "authorization lease: heartbeat interval {heartbeat_interval:?} must be \ + non-zero and below the minimum election timeout {election_timeout_min:?}" + ), + }); + } + Ok(Self { + lease: election_timeout_min, + skew_margin: heartbeat_interval, + renew_every: heartbeat_interval, + }) + } + + /// When a lease the leader granted for `granted`, requested at `sent_at` + /// on the holder's clock, ends on the holder's clock. + /// + /// The leader starts its lease no earlier than it received the request, + /// which is after `sent_at`. Ending at `sent_at + granted - skew_margin` + /// therefore ends first on any holder clock that runs no slower than the + /// margin allows. + pub fn holder_expiry(&self, sent_at: Instant, granted: Duration) -> Instant { + sent_at + granted.saturating_sub(self.skew_margin) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn timing_follows_the_raft_configuration() { + let timing = LeaseTiming::from_raft(Duration::from_millis(150), Duration::from_millis(50)) + .expect("timing"); + assert_eq!(timing.lease, Duration::from_millis(150)); + assert_eq!(timing.skew_margin, Duration::from_millis(50)); + assert_eq!(timing.renew_every, Duration::from_millis(50)); + } + + #[test] + fn a_heartbeat_at_or_above_the_election_timeout_is_refused() { + assert!( + LeaseTiming::from_raft(Duration::from_millis(50), Duration::from_millis(50)).is_err() + ); + assert!(LeaseTiming::from_raft(Duration::from_millis(50), Duration::ZERO).is_err()); + } + + /// The holder ends its lease a skew margin before the leader does, even + /// when the leader granted at the very moment the request left. + #[test] + fn the_holder_expires_early_by_the_skew_margin() { + let timing = LeaseTiming::from_raft(Duration::from_millis(150), Duration::from_millis(50)) + .expect("timing"); + let sent_at = Instant::now(); + let holder_end = timing.holder_expiry(sent_at, timing.lease); + let leader_end = sent_at + timing.lease; + assert_eq!(leader_end - holder_end, timing.skew_margin); + // A holder clock slow by a third of the lease still ends first: 100ms + // of holder time is at most 133ms of real time, inside 150ms. + let slow_real_elapsed = (holder_end - sent_at).mul_f64(4.0 / 3.0); + assert!(sent_at + slow_real_elapsed <= leader_end); + } +} diff --git a/nodedb/src/control/security/auth_lease/withheld_warn.rs b/nodedb/src/control/security/auth_lease/withheld_warn.rs new file mode 100644 index 000000000..bc2ce5b18 --- /dev/null +++ b/nodedb/src/control/security/auth_lease/withheld_warn.rs @@ -0,0 +1,50 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Rate limit for the leader's warning on a withheld lease renewal. +//! +//! A node whose coverage stays below a floor renews every interval, and each +//! renewal is withheld. The warning names the floor and the reported coverage +//! of every short group, once per node per window. + +use std::collections::HashMap; +use std::sync::Mutex; +use std::time::{Duration, Instant}; + +/// At most one warning per node in this window. +const WARN_WINDOW: Duration = Duration::from_secs(10); + +/// When each node's last warning was logged. +#[derive(Debug, Default)] +pub struct WithheldWarnings { + last: Mutex>, +} + +impl WithheldWarnings { + /// Whether a warning about `node_id` logs at `now`. A `true` starts the + /// node's next window. + pub fn should_warn(&self, node_id: u64, now: Instant) -> bool { + let mut last = self.last.lock().unwrap_or_else(|p| p.into_inner()); + match last.get(&node_id) { + Some(at) if now.duration_since(*at) < WARN_WINDOW => false, + _ => { + last.insert(node_id, now); + true + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn one_warning_per_node_per_window() { + let warnings = WithheldWarnings::default(); + let start = Instant::now(); + assert!(warnings.should_warn(2, start)); + assert!(!warnings.should_warn(2, start + Duration::from_secs(1))); + assert!(warnings.should_warn(3, start + Duration::from_secs(1))); + assert!(warnings.should_warn(2, start + WARN_WINDOW)); + } +} diff --git a/nodedb/src/control/security/catalog/bootstrap_tables.rs b/nodedb/src/control/security/catalog/bootstrap_tables.rs index 441777ef5..86f5ae1fa 100644 --- a/nodedb/src/control/security/catalog/bootstrap_tables.rs +++ b/nodedb/src/control/security/catalog/bootstrap_tables.rs @@ -75,6 +75,8 @@ pub(super) const BOOTSTRAP_TABLES: &[BootstrapTable] = bootstrap_tables![ "collections" => COLLECTIONS, "metadata" => METADATA, "wal_tombstones" => WAL_TOMBSTONES, + "tenant_group_marks" => super::tenant_group_marks::TENANT_GROUP_MARKS, + "calvin_applied" => super::calvin_applied::CALVIN_APPLIED, "l2_cleanup_queue" => L2_CLEANUP_QUEUE, "pending_reclaim" => PENDING_RECLAIM, "column_stats" => COLUMN_STATS, diff --git a/nodedb/src/control/security/catalog/calvin_applied.rs b/nodedb/src/control/security/catalog/calvin_applied.rs new file mode 100644 index 000000000..db18de836 --- /dev/null +++ b/nodedb/src/control/security/catalog/calvin_applied.rs @@ -0,0 +1,181 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Persistent Calvin applied state backing `_system.calvin_applied`. +//! +//! A Calvin scheduler learns at boot which `(epoch, position)` it already +//! applied from the applied markers in the WAL. A checkpoint deletes the WAL +//! segments that hold them, while the sequencer log still holds the entries +//! and delivers them again after a restart. So before each truncation the +//! checkpoint saves every scheduler's applied state here, and boot recovery +//! reads it together with the markers the WAL still holds. + +use std::collections::BTreeSet; + +use redb::{ReadableDatabase, ReadableTable, TableDefinition}; + +use super::types::{SystemCatalog, catalog_err}; + +/// Table: `vshard_id` -> `(fully_applied_epoch, msgpack of the applied tail)`. +pub(super) const CALVIN_APPLIED: TableDefinition = + TableDefinition::new("_system.calvin_applied"); + +/// Sentinel `fully_applied_epoch`: no epoch is fully applied. +const NONE_FULLY_APPLIED: u64 = u64::MAX; + +/// One vShard's applied state. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct StoredCalvinApplied { + pub vshard_id: u32, + /// Every position of every epoch at or below this is applied. `u64::MAX` + /// means none is. + pub fully_applied_epoch: u64, + /// Applied `(epoch, position)` pairs above the watermark. + pub tail: BTreeSet<(u64, u32)>, +} + +impl StoredCalvinApplied { + /// The union of two states of one vShard: the higher watermark, and the + /// tail entries of both above it. + fn merge(self, other: StoredCalvinApplied) -> StoredCalvinApplied { + let fully_applied_epoch = match (self.fully_applied_epoch, other.fully_applied_epoch) { + (NONE_FULLY_APPLIED, w) | (w, NONE_FULLY_APPLIED) => w, + (a, b) => a.max(b), + }; + let tail = self + .tail + .into_iter() + .chain(other.tail) + .filter(|(epoch, _)| { + fully_applied_epoch == NONE_FULLY_APPLIED || *epoch > fully_applied_epoch + }) + .collect(); + StoredCalvinApplied { + vshard_id: self.vshard_id, + fully_applied_epoch, + tail, + } + } +} + +fn encode_tail(tail: &BTreeSet<(u64, u32)>) -> crate::Result> { + let pairs: Vec<(u64, u32)> = tail.iter().copied().collect(); + zerompk::to_msgpack_vec(&pairs).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("encode calvin applied tail: {e}"), + }) +} + +fn decode_tail(bytes: &[u8]) -> crate::Result> { + let pairs: Vec<(u64, u32)> = + zerompk::from_msgpack(bytes).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("decode calvin applied tail: {e}"), + })?; + Ok(pairs.into_iter().collect()) +} + +impl SystemCatalog { + /// The saved applied state of `vshard_id`, if any. + pub fn load_calvin_applied( + &self, + vshard_id: u32, + ) -> crate::Result> { + let read_txn = self + .db + .begin_read() + .map_err(|e| catalog_err("load_calvin_applied read txn", e))?; + let table = read_txn + .open_table(CALVIN_APPLIED) + .map_err(|e| catalog_err("open calvin_applied", e))?; + let Some(row) = table + .get(vshard_id) + .map_err(|e| catalog_err("get calvin_applied", e))? + else { + return Ok(None); + }; + let (fully_applied_epoch, tail) = row.value(); + Ok(Some(StoredCalvinApplied { + vshard_id, + fully_applied_epoch, + tail: decode_tail(tail)?, + })) + } + + /// Save `states` in one transaction. Each is merged with the state + /// already saved for its vShard, so the saved state only grows. + pub fn save_calvin_applied(&self, states: Vec) -> crate::Result<()> { + if states.is_empty() { + return Ok(()); + } + let write_txn = self + .db + .begin_write() + .map_err(|e| catalog_err("save_calvin_applied txn", e))?; + { + let mut table = write_txn + .open_table(CALVIN_APPLIED) + .map_err(|e| catalog_err("open calvin_applied", e))?; + for state in states { + let saved = match table + .get(state.vshard_id) + .map_err(|e| catalog_err("get calvin_applied", e))? + { + Some(row) => { + let (fully_applied_epoch, tail) = row.value(); + Some(StoredCalvinApplied { + vshard_id: state.vshard_id, + fully_applied_epoch, + tail: decode_tail(tail)?, + }) + } + None => None, + }; + let merged = match saved { + Some(saved) => saved.merge(state), + None => state, + }; + let tail = encode_tail(&merged.tail)?; + table + .insert( + merged.vshard_id, + (merged.fully_applied_epoch, tail.as_slice()), + ) + .map_err(|e| catalog_err("insert calvin_applied", e))?; + } + } + write_txn + .commit() + .map_err(|e| catalog_err("commit calvin_applied", e)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn state(vshard_id: u32, w: u64, tail: &[(u64, u32)]) -> StoredCalvinApplied { + StoredCalvinApplied { + vshard_id, + fully_applied_epoch: w, + tail: tail.iter().copied().collect(), + } + } + + #[test] + fn saved_state_reloads_and_only_grows() { + let dir = tempfile::tempdir().expect("tempdir"); + let catalog = SystemCatalog::open(&dir.path().join("system.redb")).expect("catalog"); + assert_eq!(catalog.load_calvin_applied(3).expect("load"), None); + + catalog + .save_calvin_applied(vec![state(3, NONE_FULLY_APPLIED, &[(0, 0), (2, 1)])]) + .expect("save"); + catalog + .save_calvin_applied(vec![state(3, 1, &[(4, 0)])]) + .expect("save"); + assert_eq!( + catalog.load_calvin_applied(3).expect("load"), + Some(state(3, 1, &[(2, 1), (4, 0)])) + ); + } +} diff --git a/nodedb/src/control/security/catalog/collections.rs b/nodedb/src/control/security/catalog/collections.rs index 426b92c62..0cfe2bbd9 100644 --- a/nodedb/src/control/security/catalog/collections.rs +++ b/nodedb/src/control/security/catalog/collections.rs @@ -119,7 +119,9 @@ impl SystemCatalog { .insert((database_id.as_u64(), inner_key.as_str()), bytes.as_slice()) .map_err(|e| catalog_err("insert collection", e))?; } - write_txn.commit().map_err(|e| catalog_err("commit", e)) + write_txn.commit().map_err(|e| catalog_err("commit", e))?; + self.event_defs.install(database_id, coll); + Ok(()) } /// Insert a collection only when its catalog key is absent. @@ -157,6 +159,9 @@ impl SystemCatalog { } }; write_txn.commit().map_err(|e| catalog_err("commit", e))?; + if inserted { + self.event_defs.install(database_id, coll); + } Ok(inserted) } @@ -254,6 +259,9 @@ impl SystemCatalog { .is_some(); } write_txn.commit().map_err(|e| catalog_err("commit", e))?; + if removed { + self.event_defs.remove(database_id, tenant_id, name); + } Ok(removed) } @@ -398,7 +406,9 @@ impl SystemCatalog { } write_txn .commit() - .map_err(|e| catalog_err("migrate_collections commit", e)) + .map_err(|e| catalog_err("migrate_collections commit", e))?; + // The migration wrote rows outside `put_collection`. + self.reload_event_definitions() } } @@ -493,6 +503,73 @@ mod tests { assert_eq!(fetched.unwrap().name, "users"); } + fn with_event(tenant_id: u64, name: &str) -> StoredCollection { + let mut c = make_coll(tenant_id, name); + c.event_defs = vec![super::super::collection_constraints::EventDefinition { + name: "ev".into(), + collection: name.into(), + when_condition: "INSERT".into(), + then_action: "SELECT 1".into(), + }]; + c + } + + #[test] + fn committed_writes_keep_the_event_index_in_step() { + let (_dir, cat) = open_catalog(); + let db = DatabaseId::DEFAULT; + cat.put_collection(db, &with_event(1, "orders")).unwrap(); + assert_eq!( + cat.event_definitions(db, 1, "orders").map(|d| d.len()), + Some(1) + ); + + cat.put_collection(db, &make_coll(1, "orders")).unwrap(); + assert!(cat.event_definitions(db, 1, "orders").is_none()); + + cat.put_collection(db, &with_event(1, "orders")).unwrap(); + assert!(cat.delete_collection(db, 1, "orders").unwrap()); + assert!(cat.event_definitions(db, 1, "orders").is_none()); + } + + #[test] + fn a_skipped_insert_leaves_the_event_index_unchanged() { + let (_dir, cat) = open_catalog(); + let db = DatabaseId::DEFAULT; + cat.put_collection(db, &make_coll(1, "orders")).unwrap(); + assert!( + !cat.put_collection_if_absent(db, &with_event(1, "orders")) + .unwrap() + ); + assert!(cat.event_definitions(db, 1, "orders").is_none()); + } + + #[test] + fn a_failed_write_leaves_the_event_index_unchanged() { + let (_dir, cat) = open_catalog(); + let db = DatabaseId::DEFAULT; + cat.fail_next_collection_write_for_test(); + assert!(cat.put_collection(db, &with_event(1, "orders")).is_err()); + assert!(cat.event_definitions(db, 1, "orders").is_none()); + } + + #[test] + fn reopening_the_catalog_loads_the_event_index() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("system.redb"); + { + let cat = SystemCatalog::open(&path).unwrap(); + cat.put_collection(DatabaseId::DEFAULT, &with_event(1, "orders")) + .unwrap(); + } + let cat = SystemCatalog::open(&path).unwrap(); + assert_eq!( + cat.event_definitions(DatabaseId::DEFAULT, 1, "orders") + .map(|d| d.len()), + Some(1) + ); + } + #[test] fn missing_returns_none() { let (_dir, cat) = open_catalog(); diff --git a/nodedb/src/control/security/catalog/event_defs_index.rs b/nodedb/src/control/security/catalog/event_defs_index.rs new file mode 100644 index 000000000..35116e546 --- /dev/null +++ b/nodedb/src/control/security/catalog/event_defs_index.rs @@ -0,0 +1,238 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! In-memory index of each collection's DEFINE EVENT definitions. +//! +//! The Event Plane reads a collection's event definitions for every write +//! event. It must not read redb, so it reads this index instead. +//! +//! [`SystemCatalog`](super::system_catalog::SystemCatalog) owns the index and +//! keeps it in step with the committed `COLLECTIONS` table: +//! - `open` and `migrate_collections` rebuild it from every committed +//! collection row (`reload_event_definitions`); +//! - `put_collection` and `put_collection_if_absent` install a row after the +//! redb commit succeeds; +//! - `delete_collection` removes a row after the redb commit succeeds. +//! +//! Every collection writer goes through those functions: the replicated +//! catalog apply, the single-node fallback, transaction COMMIT, catalog +//! restore, and maintenance writers. DDL a transaction stages lives in the +//! catalog overlay and never reaches the table before COMMIT, so the index +//! holds committed definitions only. + +use std::collections::HashMap; +use std::sync::{Arc, RwLock}; + +use nodedb_types::DatabaseId; + +use redb::{ReadableDatabase, ReadableTable}; + +use super::collection::StoredCollection; +use super::collection_constraints::EventDefinition; +use super::system_catalog::SystemCatalog; +use super::tables::COLLECTIONS; +use super::types::catalog_err; + +/// `(database, tenant, collection)`. +type IndexKey = (DatabaseId, u64, String); + +/// Committed event definitions by collection. Send + Sync. A reader clones +/// the definitions out and holds no lock afterwards. +#[derive(Debug, Default)] +pub struct EventDefsIndex { + by_collection: RwLock>>, +} + +impl EventDefsIndex { + pub fn new() -> Self { + Self::default() + } + + /// Replace the whole index with the definitions of `rows`, each keyed + /// under the database its table row is stored in. + pub fn load_all(&self, rows: &[(DatabaseId, StoredCollection)]) { + let mut map = HashMap::new(); + for (database_id, row) in rows { + if let Some(defs) = active_defs(row) { + map.insert(key_of(*database_id, row), defs); + } + } + *self.write() = map; + } + + /// Record `row`, committed under `database_id`. An inactive row, or a + /// row with no event definitions, removes the collection's entry. + pub fn install(&self, database_id: DatabaseId, row: &StoredCollection) { + let key = key_of(database_id, row); + let mut map = self.write(); + match active_defs(row) { + Some(defs) => { + map.insert(key, defs); + } + None => { + map.remove(&key); + } + } + } + + /// Forget a collection whose row was deleted. + pub fn remove(&self, database_id: DatabaseId, tenant_id: u64, collection: &str) { + self.write() + .remove(&(database_id, tenant_id, collection.to_owned())); + } + + /// The committed event definitions of a collection. `None` when it has + /// none. + pub fn get( + &self, + database_id: DatabaseId, + tenant_id: u64, + collection: &str, + ) -> Option> { + let map = self + .by_collection + .read() + .unwrap_or_else(|poisoned| poisoned.into_inner()); + map.get(&(database_id, tenant_id, collection.to_owned())) + .cloned() + } + + fn write(&self) -> std::sync::RwLockWriteGuard<'_, HashMap>> { + self.by_collection + .write() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + } +} + +impl SystemCatalog { + /// Rebuild the event-definition index from every committed collection + /// row, keyed by the database each row is stored under. + pub fn reload_event_definitions(&self) -> crate::Result<()> { + let read_txn = self + .db + .begin_read() + .map_err(|e| catalog_err("read txn", e))?; + let table = read_txn + .open_table(COLLECTIONS) + .map_err(|e| catalog_err("open collections", e))?; + let mut rows = Vec::new(); + for entry in table + .iter() + .map_err(|e| catalog_err("iterate collections", e))? + { + let (key, value) = entry.map_err(|e| catalog_err("read collection", e))?; + let (database_id, _) = key.value(); + let row: StoredCollection = zerompk::from_msgpack(value.value()) + .map_err(|e| catalog_err("deser collection", e))?; + rows.push((DatabaseId::new(database_id), row)); + } + self.event_defs.load_all(&rows); + Ok(()) + } + + /// The committed DEFINE EVENT definitions of a collection. Reads memory + /// only. `None` when the collection has none. + pub fn event_definitions( + &self, + database_id: DatabaseId, + tenant_id: u64, + collection: &str, + ) -> Option> { + self.event_defs.get(database_id, tenant_id, collection) + } +} + +fn key_of(database_id: DatabaseId, row: &StoredCollection) -> IndexKey { + (database_id, row.tenant_id, row.name.clone()) +} + +/// The definitions an active row carries. `None` for an inactive row or an +/// empty list. +fn active_defs(row: &StoredCollection) -> Option> { + (row.is_active && !row.event_defs.is_empty()).then(|| Arc::from(row.event_defs.as_slice())) +} + +#[cfg(test)] +mod tests { + use super::*; + + const DB: DatabaseId = DatabaseId::DEFAULT; + + fn def(name: &str) -> EventDefinition { + EventDefinition { + name: name.into(), + collection: "orders".into(), + when_condition: "INSERT".into(), + then_action: "SELECT 1".into(), + } + } + + fn row(defs: Vec) -> StoredCollection { + let mut row = StoredCollection::new(7, "orders", "admin"); + row.event_defs = defs; + row + } + + fn names(index: &EventDefsIndex, row: &StoredCollection) -> Vec { + index + .get(DB, row.tenant_id, &row.name) + .map(|defs| defs.iter().map(|d| d.name.clone()).collect()) + .unwrap_or_default() + } + + #[test] + fn install_records_the_definitions() { + let index = EventDefsIndex::new(); + let r = row(vec![def("a")]); + index.install(DB, &r); + assert_eq!(names(&index, &r), vec!["a".to_string()]); + } + + #[test] + fn install_replaces_the_previous_definitions() { + let index = EventDefsIndex::new(); + index.install(DB, &row(vec![def("a")])); + let r = row(vec![def("b"), def("c")]); + index.install(DB, &r); + assert_eq!(names(&index, &r), vec!["b".to_string(), "c".to_string()]); + } + + #[test] + fn a_row_with_no_definitions_removes_the_entry() { + let index = EventDefsIndex::new(); + index.install(DB, &row(vec![def("a")])); + let r = row(Vec::new()); + index.install(DB, &r); + assert!(index.get(DB, r.tenant_id, &r.name).is_none()); + } + + #[test] + fn a_dropped_collection_fires_no_event() { + let index = EventDefsIndex::new(); + index.install(DB, &row(vec![def("a")])); + let mut dropped = row(vec![def("a")]); + dropped.is_active = false; + index.install(DB, &dropped); + assert!(index.get(DB, dropped.tenant_id, &dropped.name).is_none()); + } + + #[test] + fn remove_forgets_a_purged_collection() { + let index = EventDefsIndex::new(); + let r = row(vec![def("a")]); + index.install(DB, &r); + index.remove(DB, r.tenant_id, &r.name); + assert!(index.get(DB, r.tenant_id, &r.name).is_none()); + } + + #[test] + fn load_all_skips_inactive_rows() { + let index = EventDefsIndex::new(); + let live = row(vec![def("a")]); + let mut gone = StoredCollection::new(7, "gone", "admin"); + gone.event_defs = vec![def("x")]; + gone.is_active = false; + index.load_all(&[(DB, live.clone()), (DB, gone.clone())]); + assert_eq!(names(&index, &live), vec!["a".to_string()]); + assert!(index.get(DB, gone.tenant_id, &gone.name).is_none()); + } +} diff --git a/nodedb/src/control/security/catalog/index_registry.rs b/nodedb/src/control/security/catalog/index_registry.rs index 7c88b21cc..d62a3cb72 100644 --- a/nodedb/src/control/security/catalog/index_registry.rs +++ b/nodedb/src/control/security/catalog/index_registry.rs @@ -146,11 +146,26 @@ impl SystemCatalog { Ok(removed) } - /// Every index record of one (database, tenant), in key order. + /// Every index record of one (database, tenant), in key order, with the + /// calling connection's buffered transactional DDL merged in. pub fn list_index_records( &self, database_id: u64, tenant_id: u64, + ) -> crate::Result> { + let committed = self.list_committed_index_records(database_id, tenant_id)?; + Ok(crate::control::catalog_overlay::resolve_index_records( + database_id, + tenant_id, + committed, + )) + } + + /// Committed-only listing, bypassing the transaction DDL overlay. + pub fn list_committed_index_records( + &self, + database_id: u64, + tenant_id: u64, ) -> crate::Result> { let prefix = tenant_prefix(database_id, tenant_id); let read_txn = self diff --git a/nodedb/src/control/security/catalog/mod.rs b/nodedb/src/control/security/catalog/mod.rs index d41f6649b..e316a06a8 100644 --- a/nodedb/src/control/security/catalog/mod.rs +++ b/nodedb/src/control/security/catalog/mod.rs @@ -7,6 +7,7 @@ pub mod auth_types; pub mod auth_users; pub mod blacklist; pub mod bootstrap_tables; +pub mod calvin_applied; pub mod change_streams; pub mod checkpoint; pub mod checkpoints; @@ -27,6 +28,7 @@ pub mod database_grants; pub mod database_quotas; pub mod database_types; pub mod dependencies; +pub mod event_defs_index; pub mod function_types; pub mod functions; pub mod index_record; @@ -63,6 +65,7 @@ pub mod sync_producer; pub mod synonym_groups; pub mod system_catalog; pub mod tables; +pub mod tenant_group_marks; pub mod tenant_id_hwm; pub mod tenant_quotas; pub mod topics; diff --git a/nodedb/src/control/security/catalog/surrogate_pk.rs b/nodedb/src/control/security/catalog/surrogate_pk.rs index 008c2a4db..5c85bc5d6 100644 --- a/nodedb/src/control/security/catalog/surrogate_pk.rs +++ b/nodedb/src/control/security/catalog/surrogate_pk.rs @@ -8,7 +8,8 @@ //! //! The compound key is `(database_id, tenant_id, collection, pk_bytes)` //! (forward) and `(database_id, tenant_id, collection, surrogate)` (reverse), -//! scoping the PK map to its database + tenant boundary. +//! scoping the PK map to its database + tenant boundary. Every entry point +//! takes a [`CollectionKey`], so `collection` is always the bare catalog name. //! //! ## Migration //! @@ -20,7 +21,7 @@ //! second key component. Both are idempotent: each skips if its target is //! already non-empty. -use nodedb_types::{DatabaseId, Surrogate, TenantId}; +use nodedb_types::{CollectionKey, DatabaseId, Surrogate, TenantId}; use redb::{ReadableDatabase, ReadableTable, ReadableTableMetadata}; #[allow(unused_imports)] // SURROGATE_PK_REV_LEGACY is used only in #[cfg(test)] helpers @@ -37,14 +38,14 @@ impl SystemCatalog { /// pk_bytes)` to the same surrogate is a no-op-on-disk overwrite. pub fn put_surrogate( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, pk_bytes: &[u8], surrogate: Surrogate, ) -> crate::Result<()> { - let db_id = database_id.as_u64(); + let db_id = key.database_id().as_u64(); let tid = tenant_id.as_u64(); + let collection = key.name(); let txn = self .db .begin_write() @@ -69,13 +70,13 @@ impl SystemCatalog { /// collection, pk_bytes)`. Returns `None` if no binding exists. pub fn get_surrogate_for_pk( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, pk_bytes: &[u8], ) -> crate::Result> { - let db_id = database_id.as_u64(); + let db_id = key.database_id().as_u64(); let tid = tenant_id.as_u64(); + let collection = key.name(); let txn = self .db .begin_read() @@ -96,13 +97,13 @@ impl SystemCatalog { /// surrogate)`. Returns `None` if no binding exists. pub fn get_pk_for_surrogate( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, surrogate: Surrogate, ) -> crate::Result>> { - let db_id = database_id.as_u64(); + let db_id = key.database_id().as_u64(); let tid = tenant_id.as_u64(); + let collection = key.name(); let txn = self .db .begin_read() @@ -122,13 +123,13 @@ impl SystemCatalog { /// Remove a surrogate ↔ PK binding atomically. Idempotent. pub fn delete_surrogate( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, pk_bytes: &[u8], ) -> crate::Result<()> { - let db_id = database_id.as_u64(); + let db_id = key.database_id().as_u64(); let tid = tenant_id.as_u64(); + let collection = key.name(); let txn = self .db .begin_write() @@ -157,12 +158,12 @@ impl SystemCatalog { /// Returns `Vec<(pk_bytes, surrogate)>` in redb's natural key order. pub fn scan_surrogates_for_collection( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, ) -> crate::Result, Surrogate)>> { - let db_id = database_id.as_u64(); + let db_id = key.database_id().as_u64(); let tid = tenant_id.as_u64(); + let collection = key.name(); let txn = self .db .begin_read() @@ -233,16 +234,16 @@ impl SystemCatalog { /// collection)` triple. Drains both forward and reverse tables. Idempotent. pub fn delete_all_surrogates_for_collection( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, ) -> crate::Result<()> { - let to_remove = self.scan_surrogates_for_collection(database_id, tenant_id, collection)?; + let to_remove = self.scan_surrogates_for_collection(key, tenant_id)?; if to_remove.is_empty() { return Ok(()); } - let db_id = database_id.as_u64(); + let db_id = key.database_id().as_u64(); let tid = tenant_id.as_u64(); + let collection = key.name(); let txn = self .db .begin_write() @@ -444,25 +445,22 @@ mod max_bound_surrogate_tests { fn floor_is_the_global_maximum_across_every_scope() { let (_dir, cat) = open(); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "zzz_last_in_key_order"), TenantId::new(1), - "zzz_last_in_key_order", b"a", Surrogate::new(3), ) .unwrap(); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "aaa_first_in_key_order"), TenantId::new(1), - "aaa_first_in_key_order", b"b", Surrogate::new(9_000), ) .unwrap(); cat.put_surrogate( - DatabaseId::new(7), + nodedb_types::CollectionKey::from_bare(DatabaseId::new(7), "other_db"), TenantId::new(2), - "other_db", b"c", Surrogate::new(41), ) @@ -478,9 +476,8 @@ mod max_bound_surrogate_tests { fn floor_outranks_a_stale_hwm_singleton() { let (_dir, cat) = open(); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), TenantId::new(1), - "users", b"alice", Surrogate::new(500), ) @@ -519,21 +516,28 @@ mod tests { fn put_then_get_roundtrip() { let (_dir, cat) = open_catalog(); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), T0, - "users", b"alice", Surrogate::new(7), ) .unwrap(); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, T0, "users", b"alice") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"alice" + ) + .unwrap(), Some(Surrogate::new(7)) ); assert_eq!( - cat.get_pk_for_surrogate(DatabaseId::DEFAULT, T0, "users", Surrogate::new(7)) - .unwrap(), + cat.get_pk_for_surrogate( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + Surrogate::new(7) + ) + .unwrap(), Some(b"alice".to_vec()) ); } @@ -544,29 +548,35 @@ mod tests { let t1 = TenantId::new(1); let t2 = TenantId::new(2); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), t1, - "users", b"alice", Surrogate::new(10), ) .unwrap(); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), t2, - "users", b"alice", Surrogate::new(20), ) .unwrap(); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, t1, "users", b"alice") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + t1, + b"alice" + ) + .unwrap(), Some(Surrogate::new(10)) ); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, t2, "users", b"alice") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + t2, + b"alice" + ) + .unwrap(), Some(Surrogate::new(20)) ); } @@ -575,8 +585,12 @@ mod tests { fn missing_returns_none() { let (_dir, cat) = open_catalog(); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, T0, "users", b"nobody") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"nobody" + ) + .unwrap(), None ); } @@ -585,56 +599,72 @@ mod tests { fn delete_is_idempotent_and_removes_both_directions() { let (_dir, cat) = open_catalog(); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), T0, - "users", b"alice", Surrogate::new(7), ) .unwrap(); - cat.delete_surrogate(DatabaseId::DEFAULT, T0, "users", b"alice") - .unwrap(); + cat.delete_surrogate( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"alice", + ) + .unwrap(); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, T0, "users", b"alice") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"alice" + ) + .unwrap(), None ); - cat.delete_surrogate(DatabaseId::DEFAULT, T0, "users", b"alice") - .unwrap(); + cat.delete_surrogate( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"alice", + ) + .unwrap(); } #[test] fn scan_returns_only_named_collection() { let (_dir, cat) = open_catalog(); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), T0, - "users", b"alice", Surrogate::new(1), ) .unwrap(); - cat.put_surrogate(DatabaseId::DEFAULT, T0, "users", b"bob", Surrogate::new(2)) - .unwrap(); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"bob", + Surrogate::new(2), + ) + .unwrap(); + cat.put_surrogate( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "orders"), T0, - "orders", b"alice", Surrogate::new(3), ) .unwrap(); // A different tenant's same-named collection must not leak into the scan. cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), TenantId::new(9), - "users", b"carol", Surrogate::new(4), ) .unwrap(); let mut got = cat - .scan_surrogates_for_collection(DatabaseId::DEFAULT, T0, "users") + .scan_surrogates_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + ) .unwrap(); got.sort(); assert_eq!( @@ -650,30 +680,47 @@ mod tests { fn delete_all_wipes_collection_and_leaves_others_intact() { let (_dir, cat) = open_catalog(); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), T0, - "users", b"alice", Surrogate::new(1), ) .unwrap(); - cat.put_surrogate(DatabaseId::DEFAULT, T0, "orders", b"o1", Surrogate::new(2)) - .unwrap(); - cat.delete_all_surrogates_for_collection(DatabaseId::DEFAULT, T0, "users") - .unwrap(); + cat.put_surrogate( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "orders"), + T0, + b"o1", + Surrogate::new(2), + ) + .unwrap(); + cat.delete_all_surrogates_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + ) + .unwrap(); assert!( - cat.scan_surrogates_for_collection(DatabaseId::DEFAULT, T0, "users") - .unwrap() - .is_empty() + cat.scan_surrogates_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0 + ) + .unwrap() + .is_empty() ); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, T0, "orders", b"o1") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "orders"), + T0, + b"o1" + ) + .unwrap(), Some(Surrogate::new(2)) ); // double-delete is a no-op - cat.delete_all_surrogates_for_collection(DatabaseId::DEFAULT, T0, "users") - .unwrap(); + cat.delete_all_surrogates_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + ) + .unwrap(); } // ── Migration tests ─────────────────────────────────────────────────── @@ -702,9 +749,12 @@ mod tests { cat.migrate_surrogate_pk().unwrap(); cat.migrate_surrogate_pk_v3().unwrap(); assert!( - cat.scan_surrogates_for_collection(DatabaseId::DEFAULT, T0, "users") - .unwrap() - .is_empty() + cat.scan_surrogates_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0 + ) + .unwrap() + .is_empty() ); } @@ -724,9 +774,8 @@ mod tests { let (_dir, cat) = open_catalog(); // v2 row already exists, written under the default tenant in v3 … cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), T0, - "users", b"alice", Surrogate::new(7), ) @@ -781,15 +830,18 @@ mod tests { // default identity that wrote them pre-upgrade. let default_tenant = TenantId::new(1); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, default_tenant, "users", b"alice") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + default_tenant, + b"alice" + ) + .unwrap(), Some(Surrogate::new(7)) ); assert_eq!( cat.get_pk_for_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), default_tenant, - "users", Surrogate::new(7) ) .unwrap(), @@ -802,9 +854,8 @@ mod tests { let (_dir, cat) = open_catalog(); // v3 already populated … cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), T0, - "users", b"alice", Surrogate::new(7), ) @@ -824,8 +875,12 @@ mod tests { } cat.migrate_surrogate_pk_v3().unwrap(); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, T0, "users", b"alice") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"alice" + ) + .unwrap(), Some(Surrogate::new(7)) ); } diff --git a/nodedb/src/control/security/catalog/system_catalog.rs b/nodedb/src/control/security/catalog/system_catalog.rs index 45525a49b..efa13ba3a 100644 --- a/nodedb/src/control/security/catalog/system_catalog.rs +++ b/nodedb/src/control/security/catalog/system_catalog.rs @@ -21,6 +21,9 @@ use super::types::*; pub struct SystemCatalog { pub(super) db: Arc, pub(super) crdt_signing_root: Arc>>, + /// Committed DEFINE EVENT definitions, for readers that must not read + /// redb. + pub(super) event_defs: Arc, #[cfg(test)] pub(super) fail_next_user_counter_write: Arc, #[cfg(test)] @@ -47,6 +50,7 @@ impl SystemCatalog { let catalog = Self { db: Arc::new(db), crdt_signing_root: Arc::new(std::sync::RwLock::new(None)), + event_defs: Arc::new(super::event_defs_index::EventDefsIndex::new()), #[cfg(test)] fail_next_user_counter_write: Arc::new(std::sync::atomic::AtomicBool::new(false)), #[cfg(test)] @@ -55,6 +59,7 @@ impl SystemCatalog { fail_next_collection_write: Arc::new(std::sync::atomic::AtomicBool::new(false)), }; catalog.bootstrap_default_database()?; + catalog.reload_event_definitions()?; Ok(catalog) } @@ -69,6 +74,7 @@ impl SystemCatalog { let catalog = Self { db: Arc::new(db), crdt_signing_root: Arc::new(std::sync::RwLock::new(None)), + event_defs: Arc::new(super::event_defs_index::EventDefsIndex::new()), #[cfg(test)] fail_next_user_counter_write: Arc::new(std::sync::atomic::AtomicBool::new(false)), #[cfg(test)] @@ -77,6 +83,7 @@ impl SystemCatalog { fail_next_collection_write: Arc::new(std::sync::atomic::AtomicBool::new(false)), }; catalog.bootstrap_default_database()?; + catalog.reload_event_definitions()?; Ok(catalog) } diff --git a/nodedb/src/control/security/catalog/tenant_group_marks.rs b/nodedb/src/control/security/catalog/tenant_group_marks.rs new file mode 100644 index 000000000..e47ad2c8d --- /dev/null +++ b/nodedb/src/control/security/catalog/tenant_group_marks.rs @@ -0,0 +1,95 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Persistent per-group tenant write marks backing +//! `_system.tenant_group_marks`. +//! +//! For each data group this node replicates, the newest commit HLC of any +//! write of each tenant the group applied. The apply loop writes a group's +//! marks before it saves the applied floor that covers them, so every +//! committed entry is either covered by a persisted mark or above the floor, +//! where Raft delivers it again after a restart and the loop derives its mark +//! again. A Calvin commit writes its mark before its install is acknowledged. + +use redb::{ReadableDatabase, ReadableTable, TableDefinition}; + +use super::types::{SystemCatalog, catalog_err}; + +/// Table: `(group_id, tenant_id)` -> `(commit_hlc, site_code, collection)`. +pub(super) const TENANT_GROUP_MARKS: TableDefinition<(u64, u64), (u64, u8, &str)> = + TableDefinition::new("_system.tenant_group_marks"); + +/// One persisted mark. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct StoredGroupMark { + pub group_id: u64, + pub tenant_id: u64, + /// HLC wall time, in nanoseconds, of the newest write. + pub hlc: u64, + /// Which apply path recorded the write. + pub site: u8, + /// The collection the write named, empty when it named none. + pub collection: String, +} + +impl SystemCatalog { + /// Every persisted mark. + pub fn load_tenant_group_marks(&self) -> crate::Result> { + let read_txn = self + .db + .begin_read() + .map_err(|e| catalog_err("load_tenant_group_marks read txn", e))?; + let table = read_txn + .open_table(TENANT_GROUP_MARKS) + .map_err(|e| catalog_err("open tenant_group_marks", e))?; + let mut marks = Vec::new(); + for entry in table + .iter() + .map_err(|e| catalog_err("iterate tenant_group_marks", e))? + { + let (key, value) = entry.map_err(|e| catalog_err("read tenant_group_mark", e))?; + let (group_id, tenant_id) = key.value(); + let (hlc, site, collection) = value.value(); + marks.push(StoredGroupMark { + group_id, + tenant_id, + hlc, + site, + collection: collection.to_owned(), + }); + } + Ok(marks) + } + + /// Raise every mark in `marks` in one transaction. A persisted mark at or + /// above the new one stays. + pub fn raise_tenant_group_marks(&self, marks: &[StoredGroupMark]) -> crate::Result<()> { + if marks.is_empty() { + return Ok(()); + } + let write_txn = self + .db + .begin_write() + .map_err(|e| catalog_err("raise_tenant_group_marks txn", e))?; + { + let mut table = write_txn + .open_table(TENANT_GROUP_MARKS) + .map_err(|e| catalog_err("open tenant_group_marks", e))?; + for mark in marks { + let key = (mark.group_id, mark.tenant_id); + let current = table + .get(key) + .map_err(|e| catalog_err("get tenant_group_mark", e))? + .map(|guard| guard.value().0); + if current.is_some_and(|hlc| hlc >= mark.hlc) { + continue; + } + table + .insert(key, (mark.hlc, mark.site, mark.collection.as_str())) + .map_err(|e| catalog_err("insert tenant_group_mark", e))?; + } + } + write_txn + .commit() + .map_err(|e| catalog_err("commit tenant_group_marks", e)) + } +} diff --git a/nodedb/src/control/security/catalog/vector_index_params.rs b/nodedb/src/control/security/catalog/vector_index_params.rs index 54cca2f89..0e9c34145 100644 --- a/nodedb/src/control/security/catalog/vector_index_params.rs +++ b/nodedb/src/control/security/catalog/vector_index_params.rs @@ -37,13 +37,35 @@ impl SystemCatalog { write_txn.commit().map_err(|e| catalog_err("commit", e)) } - /// Load vector index parameters for a specific collection/field. + /// Load vector index parameters for a specific collection/field, with + /// this connection's uncommitted transactional DDL merged in. pub fn get_vector_index_params( &self, database_id: u64, tenant_id: u64, collection: &str, field_name: &str, + ) -> crate::Result> { + let committed = + self.get_committed_vector_index_params(database_id, tenant_id, collection, field_name)?; + Ok( + crate::control::catalog_overlay::resolve_vector_index_params( + database_id, + tenant_id, + collection, + field_name, + committed, + ), + ) + } + + /// Committed-only read, bypassing the transaction DDL overlay. + pub fn get_committed_vector_index_params( + &self, + database_id: u64, + tenant_id: u64, + collection: &str, + field_name: &str, ) -> crate::Result> { let key = vector_index_params_key(database_id, tenant_id, collection, field_name); let read_txn = self diff --git a/nodedb/src/control/security/catalog/wal_tombstones.rs b/nodedb/src/control/security/catalog/wal_tombstones.rs index 3605451cf..20cfd1d67 100644 --- a/nodedb/src/control/security/catalog/wal_tombstones.rs +++ b/nodedb/src/control/security/catalog/wal_tombstones.rs @@ -115,10 +115,13 @@ pub(super) fn load_wal_tombstones_in( { let (key, value) = entry.map_err(|e| catalog_err("read wal_tombstone", e))?; let (database_id, tenant_id, collection) = key.value(); + // Rows name the collection by its bare catalog name. set.insert( - database_id, + nodedb_types::CollectionKey::from_bare( + nodedb_types::DatabaseId::new(database_id), + collection, + ), tenant_id, - collection.to_string(), value.value(), ); } @@ -127,9 +130,15 @@ pub(super) fn load_wal_tombstones_in( #[cfg(test)] mod tests { - use super::*; + use nodedb_types::{CollectionKey, DatabaseId}; use tempfile::TempDir; + use super::*; + + fn key(database: u64, name: &str) -> CollectionKey<'_> { + CollectionKey::from_bare(DatabaseId::new(database), name) + } + fn catalog() -> (SystemCatalog, TempDir) { let tmp = TempDir::new().unwrap(); let path = tmp.path().join("system.redb"); @@ -146,9 +155,9 @@ mod tests { let set = cat.load_wal_tombstones().unwrap(); assert_eq!(set.len(), 3); - assert_eq!(set.purge_lsn(7, 1, "users"), Some(100)); - assert_eq!(set.purge_lsn(7, 1, "orders"), Some(150)); - assert_eq!(set.purge_lsn(8, 1, "users"), Some(200)); + assert_eq!(set.purge_lsn(key(7, "users"), 1), Some(100)); + assert_eq!(set.purge_lsn(key(7, "orders"), 1), Some(150)); + assert_eq!(set.purge_lsn(key(8, "users"), 1), Some(200)); } #[test] @@ -157,12 +166,16 @@ mod tests { cat.record_wal_tombstone(7, 1, "users", 100).unwrap(); cat.record_wal_tombstone(7, 1, "users", 50).unwrap(); assert_eq!( - cat.load_wal_tombstones().unwrap().purge_lsn(7, 1, "users"), + cat.load_wal_tombstones() + .unwrap() + .purge_lsn(key(7, "users"), 1), Some(100) ); cat.record_wal_tombstone(7, 1, "users", 200).unwrap(); assert_eq!( - cat.load_wal_tombstones().unwrap().purge_lsn(7, 1, "users"), + cat.load_wal_tombstones() + .unwrap() + .purge_lsn(key(7, "users"), 1), Some(200) ); } @@ -179,7 +192,7 @@ mod tests { let set = cat.load_wal_tombstones().unwrap(); assert_eq!(set.len(), 1); - assert_eq!(set.purge_lsn(7, 1, "c"), Some(1000)); + assert_eq!(set.purge_lsn(key(7, "c"), 1), Some(1000)); } #[test] @@ -190,4 +203,15 @@ mod tests { assert_eq!(cat.delete_wal_tombstones_before_lsn(100).unwrap(), 0); assert_eq!(cat.load_wal_tombstones().unwrap().len(), 1); } + + /// A persisted row holds the bare name. Loaded, it must shadow the + /// qualified storage name a named database's data records carry. + #[test] + fn loaded_tombstone_shadows_qualified_records() { + let (cat, _tmp) = catalog(); + cat.record_wal_tombstone(1024, 1, "orders", 100).unwrap(); + let set = cat.load_wal_tombstones().unwrap(); + assert!(set.is_tombstoned(1024, 1, "1024/orders", 99)); + assert!(!set.is_tombstoned(1024, 1, "1024/orders", 100)); + } } diff --git a/nodedb/src/control/security/credential/store/crud.rs b/nodedb/src/control/security/credential/store/crud.rs index 69ce91aa9..34b795cf1 100644 --- a/nodedb/src/control/security/credential/store/crud.rs +++ b/nodedb/src/control/security/credential/store/crud.rs @@ -293,31 +293,6 @@ impl CredentialStore { Ok(()) } - /// Replace the `accessible_databases` list on a service account. - /// - /// Requires the caller to have already verified superuser authority. - /// For non-service-account users, returns an error. - pub fn set_service_account_databases( - &self, - name: &str, - databases: Vec, - ) -> crate::Result<()> { - let mut users = write_lock(&self.users); - let record = users - .get_mut(name) - .ok_or_else(|| crate::Error::BadRequest { - detail: format!("service account '{name}' not found"), - })?; - if !record.is_service_account { - return Err(crate::Error::BadRequest { - detail: format!("'{name}' is a user, not a service account"), - }); - } - record.accessible_databases = databases; - self.commit_user_mutation(record, Some(SessionInvalidationReason::RoleAltered))?; - Ok(()) - } - /// Remove a role from a user. Triggers `RoleRevoked` soft-revoke on /// open sessions. pub fn remove_role(&self, username: &str, role: &Role) -> crate::Result<()> { diff --git a/nodedb/src/control/security/credential/store/mod.rs b/nodedb/src/control/security/credential/store/mod.rs index c94a29b60..a330c7116 100644 --- a/nodedb/src/control/security/credential/store/mod.rs +++ b/nodedb/src/control/security/credential/store/mod.rs @@ -6,6 +6,7 @@ pub mod core; pub mod crud; pub mod list; pub mod replication; +pub mod user_builders; pub use auth::{AuthRejection, PasswordVerification, ScramCredentials, ScramLookup}; pub use core::CredentialStore; diff --git a/nodedb/src/control/security/credential/store/replication.rs b/nodedb/src/control/security/credential/store/replication.rs index a72525c00..069529063 100644 --- a/nodedb/src/control/security/credential/store/replication.rs +++ b/nodedb/src/control/security/credential/store/replication.rs @@ -27,14 +27,8 @@ use crate::types::TenantId; use super::super::super::catalog::StoredUser; use super::super::super::identity::Role; -use super::super::super::time::now_secs; -use super::super::hash::{ - compute_scram_salted_password, generate_scram_salt, hash_password_argon2, -}; use super::super::record::UserRecord; -use super::core::{ - CredentialStore, PasswordPrincipal, read_lock, validate_password_assignment, write_lock, -}; +use super::core::{CredentialStore, read_lock, write_lock}; impl CredentialStore { /// Build a `StoredUser` ready for replication via @@ -49,116 +43,62 @@ impl CredentialStore { tenant_id: TenantId, roles: Vec, ) -> crate::Result { - { - let users = read_lock(&self.users); - if users.contains_key(username) { - return Err(crate::Error::BadRequest { - detail: format!("user '{username}' already exists"), - }); - } + if read_lock(&self.users).contains_key(username) { + return Err(crate::Error::BadRequest { + detail: format!("user '{username}' already exists"), + }); } - validate_password_assignment(password, PasswordPrincipal::New)?; + self.prepare_new_user(username, password, tenant_id, roles) + } - let salt = generate_scram_salt(); - let scram_salted_password = compute_scram_salted_password(password, &salt); - let password_hash = hash_password_argon2(password, &self.argon2_config)?; - let user_id = self.alloc_user_id()?; - let is_superuser = roles.contains(&Role::Superuser); - let now = now_secs(); + /// Build a service-account `StoredUser` ready for replication via + /// `CatalogEntry::PutUser`. + pub fn prepare_service_account( + &self, + name: &str, + tenant_id: TenantId, + roles: Vec, + accessible_databases: Vec, + ) -> crate::Result { + if read_lock(&self.users).contains_key(name) { + return Err(crate::Error::BadRequest { + detail: format!("user or service account '{name}' already exists"), + }); + } + self.prepare_new_service_account(name, tenant_id, roles, accessible_databases) + } - Ok(StoredUser { - user_id, - username: username.to_string(), - tenant_id: tenant_id.as_u64(), - password_hash, - scram_salt: salt, - scram_salted_password, - roles: roles.iter().map(|r| r.to_string()).collect(), - is_superuser, - is_active: true, - is_service_account: false, - created_at: now, - updated_at: now, - password_expires_at: self.compute_expiry(), - must_change_password: false, - password_changed_at: now, - default_database_id: 0, - accessible_databases: vec![], - }) + /// Build the updated `StoredUser` of committed service account `name` + /// restricted to `databases`. + pub fn prepare_service_account_databases( + &self, + name: &str, + databases: Vec, + ) -> crate::Result { + let base = self.existing_active(name)?; + self.prepare_service_account_databases_from(base, databases) } - /// Build an updated `StoredUser` from an existing user with - /// specific fields replaced. Used by `ALTER USER SET PASSWORD` - /// and `ALTER USER SET ROLE`. Returns the updated record - /// ready for propose. + /// Build an updated `StoredUser` from committed user `username` with a + /// new password and/or role set. pub fn prepare_user_update( &self, username: &str, new_password: Option<&str>, new_roles: Option>, ) -> crate::Result { - let users = read_lock(&self.users); - let existing = users - .get(username) - .ok_or_else(|| crate::Error::BadRequest { - detail: format!("user '{username}' not found"), - })?; - if !existing.is_active { - return Err(crate::Error::BadRequest { - detail: format!("user '{username}' is inactive"), - }); - } - if let Some(password) = new_password { - validate_password_assignment( - password, - PasswordPrincipal::Existing { - is_service_account: existing.is_service_account, - }, - )?; - } - let mut stored = existing.to_stored(); - drop(users); - - if let Some(pw) = new_password { - let salt = generate_scram_salt(); - stored.scram_salted_password = compute_scram_salted_password(pw, &salt); - stored.scram_salt = salt; - stored.password_hash = hash_password_argon2(pw, &self.argon2_config)?; - stored.password_expires_at = self.compute_expiry(); - stored.must_change_password = false; - stored.password_changed_at = now_secs(); - } - if let Some(roles) = new_roles { - stored.is_superuser = roles.contains(&Role::Superuser); - stored.roles = roles.iter().map(|r| r.to_string()).collect(); - } - stored.updated_at = now_secs(); - Ok(stored) + let base = self.existing_active(username)?; + self.prepare_user_update_from(base, new_password, new_roles) } /// Build an updated `StoredUser` that sets `must_change_password`. - /// Used by `ALTER USER MUST CHANGE PASSWORD`. pub fn prepare_set_must_change_password( &self, username: &str, required: bool, ) -> crate::Result { - let users = read_lock(&self.users); - let existing = users - .get(username) - .ok_or_else(|| crate::Error::BadRequest { - detail: format!("user '{username}' not found"), - })?; - if !existing.is_active { - return Err(crate::Error::BadRequest { - detail: format!("user '{username}' is inactive"), - }); - } - let mut stored = existing.to_stored(); - drop(users); - stored.must_change_password = required; - stored.updated_at = now_secs(); - Ok(stored) + let base = self.existing_active(username)?; + Ok(self.prepare_set_must_change_password_from(base, required)) } /// Build an updated `StoredUser` that sets `password_expires_at`. @@ -168,47 +108,18 @@ impl CredentialStore { username: &str, expires_at: u64, ) -> crate::Result { - let users = read_lock(&self.users); - let existing = users - .get(username) - .ok_or_else(|| crate::Error::BadRequest { - detail: format!("user '{username}' not found"), - })?; - if !existing.is_active { - return Err(crate::Error::BadRequest { - detail: format!("user '{username}' is inactive"), - }); - } - let mut stored = existing.to_stored(); - drop(users); - stored.password_expires_at = expires_at; - stored.updated_at = now_secs(); - Ok(stored) + let base = self.existing_active(username)?; + Ok(self.prepare_set_password_expires_at_from(base, expires_at)) } /// Build an updated `StoredUser` that sets `default_database_id`. - /// Used by `ALTER USER SET DEFAULT DATABASE `. pub fn prepare_set_default_database( &self, username: &str, database_id: u64, ) -> crate::Result { - let users = read_lock(&self.users); - let existing = users - .get(username) - .ok_or_else(|| crate::Error::BadRequest { - detail: format!("user '{username}' not found"), - })?; - if !existing.is_active { - return Err(crate::Error::BadRequest { - detail: format!("user '{username}' is inactive"), - }); - } - let mut stored = existing.to_stored(); - drop(users); - stored.default_database_id = database_id; - stored.updated_at = now_secs(); - Ok(stored) + let base = self.existing_active(username)?; + Ok(self.prepare_set_default_database_from(base, database_id)) } /// Install a replicated `StoredUser` into the in-memory cache and diff --git a/nodedb/src/control/security/credential/store/user_builders.rs b/nodedb/src/control/security/credential/store/user_builders.rs new file mode 100644 index 000000000..74a17867a --- /dev/null +++ b/nodedb/src/control/security/credential/store/user_builders.rs @@ -0,0 +1,206 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Build the `StoredUser` a user-mutating statement proposes. +//! +//! Each builder takes the record it starts from rather than looking it up. +//! A statement inside a transaction starts from the user as the transaction +//! sees it: a user it created earlier is visible, and a user it dropped is +//! not. Outside a transaction that is the committed record, which +//! [`CredentialStore::stored_user`] returns. The builders touch neither the +//! in-memory map nor redb: the applier installs the record after commit. + +use crate::types::TenantId; + +use super::super::super::catalog::StoredUser; +use super::super::super::identity::Role; +use super::super::super::time::now_secs; +use super::super::hash::{ + compute_scram_salted_password, generate_scram_salt, hash_password_argon2, +}; +use super::core::{CredentialStore, PasswordPrincipal, read_lock, validate_password_assignment}; + +impl CredentialStore { + /// The committed record of active user `username`. + pub fn stored_user(&self, username: &str) -> Option { + let users = read_lock(&self.users); + users + .get(username) + .filter(|user| user.is_active) + .map(|user| user.to_stored()) + } + + /// Build a new password user. Allocates a user_id and hashes the password + /// (Argon2 + SCRAM salt). The caller has checked the name is free. + pub fn prepare_new_user( + &self, + username: &str, + password: &str, + tenant_id: TenantId, + roles: Vec, + ) -> crate::Result { + validate_password_assignment(password, PasswordPrincipal::New)?; + + let salt = generate_scram_salt(); + let scram_salted_password = compute_scram_salted_password(password, &salt); + let password_hash = hash_password_argon2(password, &self.argon2_config)?; + let user_id = self.alloc_user_id()?; + let is_superuser = roles.contains(&Role::Superuser); + let now = now_secs(); + + Ok(StoredUser { + user_id, + username: username.to_string(), + tenant_id: tenant_id.as_u64(), + password_hash, + scram_salt: salt, + scram_salted_password, + roles: roles.iter().map(|r| r.to_string()).collect(), + is_superuser, + is_active: true, + is_service_account: false, + created_at: now, + updated_at: now, + password_expires_at: self.compute_expiry(), + must_change_password: false, + password_changed_at: now, + default_database_id: 0, + accessible_databases: vec![], + }) + } + + /// Build a new service account. It authenticates by API key only, so it + /// carries no password hash. The caller has checked the name is free. + pub fn prepare_new_service_account( + &self, + name: &str, + tenant_id: TenantId, + roles: Vec, + accessible_databases: Vec, + ) -> crate::Result { + let user_id = self.alloc_user_id()?; + let now = now_secs(); + Ok(StoredUser { + user_id, + username: name.to_string(), + tenant_id: tenant_id.as_u64(), + password_hash: String::new(), + scram_salt: Vec::new(), + scram_salted_password: Vec::new(), + is_superuser: roles.contains(&Role::Superuser), + roles: roles.iter().map(|r| r.to_string()).collect(), + is_active: true, + is_service_account: true, + created_at: now, + updated_at: now, + password_expires_at: 0, + must_change_password: false, + password_changed_at: now, + default_database_id: 0, + accessible_databases: accessible_databases + .iter() + .map(|database_id| database_id.as_u64()) + .collect(), + }) + } + + /// `base` with a new password and/or role set. Used by `ALTER USER SET + /// PASSWORD`, `ALTER USER SET ROLE`, and `GRANT` / `REVOKE` of roles. + pub fn prepare_user_update_from( + &self, + mut base: StoredUser, + new_password: Option<&str>, + new_roles: Option>, + ) -> crate::Result { + if let Some(password) = new_password { + validate_password_assignment( + password, + PasswordPrincipal::Existing { + is_service_account: base.is_service_account, + }, + )?; + let salt = generate_scram_salt(); + base.scram_salted_password = compute_scram_salted_password(password, &salt); + base.scram_salt = salt; + base.password_hash = hash_password_argon2(password, &self.argon2_config)?; + base.password_expires_at = self.compute_expiry(); + base.must_change_password = false; + base.password_changed_at = now_secs(); + } + if let Some(roles) = new_roles { + base.is_superuser = roles.contains(&Role::Superuser); + base.roles = roles.iter().map(|r| r.to_string()).collect(); + } + base.updated_at = now_secs(); + Ok(base) + } + + /// `base` with `must_change_password` set to `required`. + pub fn prepare_set_must_change_password_from( + &self, + mut base: StoredUser, + required: bool, + ) -> StoredUser { + base.must_change_password = required; + base.updated_at = now_secs(); + base + } + + /// `base` with `password_expires_at`. `0` means "NEVER EXPIRES". + pub fn prepare_set_password_expires_at_from( + &self, + mut base: StoredUser, + expires_at: u64, + ) -> StoredUser { + base.password_expires_at = expires_at; + base.updated_at = now_secs(); + base + } + + /// `base` with `default_database_id`. + pub fn prepare_set_default_database_from( + &self, + mut base: StoredUser, + database_id: u64, + ) -> StoredUser { + base.default_database_id = database_id; + base.updated_at = now_secs(); + base + } + + /// Service account `base` restricted to `databases`. Refuses a password + /// user. + pub fn prepare_service_account_databases_from( + &self, + mut base: StoredUser, + databases: Vec, + ) -> crate::Result { + if !base.is_service_account { + return Err(crate::Error::BadRequest { + detail: format!("'{}' is a user, not a service account", base.username), + }); + } + base.accessible_databases = databases + .iter() + .map(|database_id| database_id.as_u64()) + .collect(); + base.updated_at = now_secs(); + Ok(base) + } + + /// The committed active record of `username`, or the error a + /// username-addressed builder reports. + pub(super) fn existing_active(&self, username: &str) -> crate::Result { + let users = read_lock(&self.users); + let existing = users + .get(username) + .ok_or_else(|| crate::Error::BadRequest { + detail: format!("user '{username}' not found"), + })?; + if !existing.is_active { + return Err(crate::Error::BadRequest { + detail: format!("user '{username}' is inactive"), + }); + } + Ok(existing.to_stored()) + } +} diff --git a/nodedb/src/control/security/identity/plan_permission.rs b/nodedb/src/control/security/identity/plan_permission.rs index 03f5c4c6d..9f43425b5 100644 --- a/nodedb/src/control/security/identity/plan_permission.rs +++ b/nodedb/src/control/security/identity/plan_permission.rs @@ -272,6 +272,9 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm // Mirrors `ResolveTxn`: Calvin scheduler's commit path, treated as Write though it doesn't mutate base state. PhysicalPlan::Meta(MetaOp::CalvinResolve { .. }) => Permission::Write, + // Installs a committed transaction's post-images into base state. + PhysicalPlan::Meta(MetaOp::ApplyTransactionRedo { .. }) => Permission::Write, + // KV engine: read operations. PhysicalPlan::Kv( KvOp::Get { .. } @@ -285,6 +288,7 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm | KvOp::SortedIndexRange { .. } | KvOp::SortedIndexCount { .. } | KvOp::SortedIndexScore { .. } + | KvOp::SortedIndexTxnRead { .. } // Read-only: reports what a governed write would apply; that write is authorized separately. | KvOp::ResolveWrite(_), ) => Permission::Read, diff --git a/nodedb/src/control/security/jwt_policy/gate.rs b/nodedb/src/control/security/jwt_policy/gate.rs index 192a29a79..259937194 100644 --- a/nodedb/src/control/security/jwt_policy/gate.rs +++ b/nodedb/src/control/security/jwt_policy/gate.rs @@ -10,28 +10,33 @@ //! [`SharedState`] carries, so it hangs off the two call sites that own one: //! the HTTP bearer path and the native/OIDC bearer path. +use crate::control::security::identity::AuthenticatedIdentity; use crate::control::security::jwt::JwtClaims; use crate::control::state::SharedState; -use crate::types::TenantId; -use super::{provisioning, scopes}; +use super::{provisioning, roles, scopes}; /// Apply the state-dependent half of the JWT policy to a verified token. /// -/// `tenant_id` is the tenant the *identity* was bound to by its provider — -/// never a tenant asserted by the token's claims. +/// `identity` is the identity the token bound to: its tenant is the one its +/// provider is bound to — never a tenant asserted by the token's claims — +/// and its roles are the resolved ones the session will hold. /// -/// A deployment with no JWKS registry has no JWT authentication at all, so -/// there is no policy to apply and the gate is a no-op. +/// An identity holding a custom role not defined in its tenant is refused +/// first, before any record is provisioned. A deployment with no JWKS +/// registry has no JWT authentication at all, so there is no other policy to +/// apply. pub fn enforce_stateful_jwt_policy( state: &SharedState, claims: &JwtClaims, - tenant_id: TenantId, + identity: &AuthenticatedIdentity, ) -> crate::Result<()> { + roles::refuse_undefined_roles(state, identity)?; let Some(registry) = state.jwks_registry.as_ref() else { return Ok(()); }; let config = registry.jwt_config(); + let tenant_id = identity.tenant_id; scopes::enforce_declared_scopes(config.enforce_scopes, &state.scope_defs, claims, tenant_id)?; provisioning::provision_and_check_status( diff --git a/nodedb/src/control/security/jwt_policy/mod.rs b/nodedb/src/control/security/jwt_policy/mod.rs index 4c0c56b5b..4ab535fe8 100644 --- a/nodedb/src/control/security/jwt_policy/mod.rs +++ b/nodedb/src/control/security/jwt_policy/mod.rs @@ -4,6 +4,7 @@ pub mod claim_path; pub mod gate; pub mod provisioning; pub mod remap; +pub mod roles; pub mod scopes; pub mod status; @@ -11,5 +12,6 @@ pub use claim_path::{resolve_claim, string_list}; pub use gate::enforce_stateful_jwt_policy; pub use provisioning::provision_and_check_status; pub use remap::{REMAPPABLE_FIELDS, remap_claims, validate_claim_remap}; +pub use roles::refuse_undefined_roles; pub use scopes::enforce_declared_scopes; pub use status::check_blocked_status; diff --git a/nodedb/src/control/security/jwt_policy/roles.rs b/nodedb/src/control/security/jwt_policy/roles.rs new file mode 100644 index 000000000..abf93e64c --- /dev/null +++ b/nodedb/src/control/security/jwt_policy/roles.rs @@ -0,0 +1,149 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Refuse an external identity that resolved to an undefined custom role. +//! +//! A JWT or OIDC login takes its roles from the identity provider: the +//! token's claims, remapped by `[auth.jwt]`, or the provider's claim-mapping +//! rules. A resolved name that is not built in parses to a custom role, and a +//! custom role grants something only while it is defined in the tenant the +//! provider is bound to. An identity holding an undefined role would hold +//! nothing, and a just-in-time provisioned record would store it. So the +//! login is refused with a typed error naming the role, and the refusal is +//! recorded in the audit log, before anything is provisioned. + +use crate::control::security::audit::AuditEvent; +use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::security::role_assignment::{self, RoleRefusal}; +use crate::control::state::SharedState; + +/// Refuse `identity` when any of its roles is neither built in nor defined +/// in its tenant. +pub fn refuse_undefined_roles( + state: &SharedState, + identity: &AuthenticatedIdentity, +) -> crate::Result<()> { + let tenant_id = identity.tenant_id; + let checked = role_assignment::check_assignable(&identity.roles, tenant_id.as_u64(), |name| { + state + .roles + .get_role(name) + .map(|role| role.tenant_id.as_u64()) + }); + let Err(refusal) = checked else { + return Ok(()); + }; + let role = match refusal { + RoleRefusal::Undefined { name } => name, + other => other.to_string(), + }; + state.audit_record( + AuditEvent::AuthFailure, + Some(tenant_id), + &identity.username, + &format!("external login refused: role \"{role}\" is not defined in tenant {tenant_id}"), + ); + Err(crate::Error::ExternalRoleUndefined { + subject: identity.username.clone(), + role, + tenant_id: tenant_id.as_u64(), + }) +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::control::security::identity::{AuthMethod, Role}; + use crate::types::TenantId; + use crate::wal::WalManager; + + fn state(dir: &tempfile::TempDir) -> Arc { + let wal = Arc::new( + WalManager::open_for_testing(&dir.path().join("roles.wal")).expect("open WAL"), + ); + let (dispatcher, _data_sides) = Dispatcher::new(1, 64); + SharedState::new(dispatcher, wal).expect("shared state") + } + + fn external(tenant: u64, roles: Vec) -> AuthenticatedIdentity { + AuthenticatedIdentity::new_regular( + 7, + "idp-alice", + TenantId::new(tenant), + AuthMethod::OidcBearer, + roles, + None, + AuthenticatedIdentity::default_database_set(false), + ) + } + + fn auth_failures(state: &SharedState) -> Vec { + state + .audit + .lock() + .unwrap_or_else(|p| p.into_inner()) + .query_by_event(&AuditEvent::AuthFailure) + .into_iter() + .map(|entry| entry.detail.clone()) + .collect() + } + + #[test] + fn an_undefined_claimed_role_refuses_the_login_and_is_audited() { + let dir = tempfile::tempdir().expect("tempdir"); + let state = state(&dir); + let identity = external(5, vec![Role::ReadOnly, Role::Custom("ghost".into())]); + + match refuse_undefined_roles(&state, &identity) { + Err(crate::Error::ExternalRoleUndefined { + subject, + role, + tenant_id, + }) => { + assert_eq!(subject, "idp-alice"); + assert_eq!(role, "ghost"); + assert_eq!(tenant_id, 5); + } + other => panic!("expected ExternalRoleUndefined, got {other:?}"), + } + let failures = auth_failures(&state); + assert!( + failures.iter().any(|detail| detail.contains("\"ghost\"")), + "the refusal must be audited: {failures:?}" + ); + } + + #[test] + fn a_defined_role_of_the_bound_tenant_is_accepted() { + let dir = tempfile::tempdir().expect("tempdir"); + let state = state(&dir); + state + .roles + .create_role("analyst", TenantId::new(5), None, None) + .expect("create role"); + + refuse_undefined_roles( + &state, + &external(5, vec![Role::ReadWrite, Role::Custom("analyst".into())]), + ) + .expect("a defined role is accepted"); + assert!(auth_failures(&state).is_empty()); + } + + #[test] + fn a_role_defined_only_in_another_tenant_is_refused() { + let dir = tempfile::tempdir().expect("tempdir"); + let state = state(&dir); + state + .roles + .create_role("analyst", TenantId::new(5), None, None) + .expect("create role"); + + assert!(matches!( + refuse_undefined_roles(&state, &external(6, vec![Role::Custom("analyst".into())])), + Err(crate::Error::ExternalRoleUndefined { .. }) + )); + } +} diff --git a/nodedb/src/control/security/mod.rs b/nodedb/src/control/security/mod.rs index 8f73759b8..87c455765 100644 --- a/nodedb/src/control/security/mod.rs +++ b/nodedb/src/control/security/mod.rs @@ -4,6 +4,8 @@ pub mod apikey; pub mod audit; pub mod auth_apikey; pub mod auth_context; +pub mod auth_fence; +pub mod auth_lease; pub mod blacklist; pub mod buses; pub mod catalog; @@ -41,6 +43,7 @@ pub mod request_scope; pub mod risk; pub mod rls; pub mod role; +pub mod role_assignment; pub mod scope; pub mod session_handle; pub mod session_registry; diff --git a/nodedb/src/control/security/oidc/verify.rs b/nodedb/src/control/security/oidc/verify.rs index f953b2aa0..5188cd385 100644 --- a/nodedb/src/control/security/oidc/verify.rs +++ b/nodedb/src/control/security/oidc/verify.rs @@ -157,7 +157,7 @@ pub async fn verify_bearer_token( crate::control::security::jwt_policy::enforce_stateful_jwt_policy( state, verified_claims, - identity.tenant_id, + &identity, )?; Ok((identity, verified)) diff --git a/nodedb/src/control/security/permission_tree/cache.rs b/nodedb/src/control/security/permission_tree/cache.rs index f092da0ab..cb798f133 100644 --- a/nodedb/src/control/security/permission_tree/cache.rs +++ b/nodedb/src/control/security/permission_tree/cache.rs @@ -2,14 +2,18 @@ //! In-memory permission cache: parent hierarchy + grant lookups. //! -//! Loaded from the resource graph and permission collection on startup. -//! Maintained via CDC events for real-time invalidation. -//! Lives entirely in the Control Plane (Send + Sync). +//! Loaded from the governed collections and permission tables by a reload, +//! and kept current by the Event Plane's permission step. `progress` records +//! how far the cache reflects each core's writes. Lives entirely in the +//! Control Plane (Send + Sync). use std::collections::{HashMap, HashSet}; +use std::sync::Arc; use tracing::{debug, info}; +use super::sources::SourceIndex; +use super::sync_state::ApplyProgress; use super::types::{PermissionGrant, PermissionTreeDef}; /// Per-tenant permission state: resource hierarchy + permission grants. @@ -37,7 +41,7 @@ struct TenantPermissions { /// Central permission cache shared across all sessions. /// /// Thread-safe: wrapped in `Arc>` by SharedState. -#[derive(Default, Debug)] +#[derive(Debug)] pub struct PermissionCache { /// Per-tenant permission state. tenants: HashMap, @@ -45,15 +49,111 @@ pub struct PermissionCache { /// Per-collection permission tree definitions. /// Key: `(tenant_id, collection_name)`. tree_defs: HashMap<(u64, String), PermissionTreeDef>, + + /// How far the cache reflects each core's writes. + progress: ApplyProgress, + + /// The source collections of `tree_defs`, readable without this cache's + /// lock. Rebuilt on every tree-definition change. + sources: Arc, +} + +impl Default for PermissionCache { + fn default() -> Self { + Self::new() + } +} + +/// One collection a reload scans, and what its rows carry. +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +pub struct TreeSource { + pub tenant_id: u64, + pub collection: String, + pub kind: TreeSourceKind, +} + +/// What a source collection's rows carry. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum TreeSourceKind { + /// A governed collection: each row's `id` and `parent_id` form an edge. + Hierarchy, + /// A permission table: each row is a grant. + Grants, } impl PermissionCache { + /// An empty cache. It needs a reload before planning may use it. pub fn new() -> Self { - Self::default() + Self { + tenants: HashMap::new(), + tree_defs: HashMap::new(), + progress: ApplyProgress::new(), + sources: Arc::new(SourceIndex::default()), + } + } + + /// The lock-free index of this cache's source collections. + pub fn sources(&self) -> Arc { + Arc::clone(&self.sources) + } + + /// How far the cache reflects each core's writes. + pub fn progress(&self) -> &ApplyProgress { + &self.progress } - /// Register a permission tree definition for a collection. + /// Mutable access to the apply progress, for the permission step and a + /// reload. + pub fn progress_mut(&mut self) -> &mut ApplyProgress { + &mut self.progress + } + + /// Every collection a reload scans, deduplicated. + pub fn tree_sources(&self) -> Vec { + let mut sources: HashSet = HashSet::new(); + for ((tenant_id, collection), def) in &self.tree_defs { + sources.insert(TreeSource { + tenant_id: *tenant_id, + collection: collection.clone(), + kind: TreeSourceKind::Hierarchy, + }); + sources.insert(TreeSource { + tenant_id: *tenant_id, + collection: def.permission_table.clone(), + kind: TreeSourceKind::Grants, + }); + } + sources.into_iter().collect() + } + + /// Replace a tenant's hierarchy and grants with a reload's result, and + /// bump its version so a cached plan built from the old state goes stale. + pub fn replace_tenant_state( + &mut self, + tenant_id: u64, + edges: &[(String, String)], + grants: &[PermissionGrant], + ) { + let version = self.tenant_version(tenant_id); + self.tenants.insert( + tenant_id, + TenantPermissions { + version, + ..TenantPermissions::default() + }, + ); + self.load_edges(tenant_id, edges); + self.load_grants(tenant_id, grants); + self.bump_tenant_version(tenant_id); + } + + /// Register a permission tree definition for a collection. Registering + /// the definition already held changes nothing, so the DDL node and the + /// metadata applier can both apply one change. pub fn register_tree_def(&mut self, tenant_id: u64, collection: &str, def: PermissionTreeDef) { + if self.get_tree_def(tenant_id, collection) == Some(&def) { + return; + } info!( tenant_id, collection, @@ -62,11 +162,25 @@ impl PermissionCache { ); self.tree_defs .insert((tenant_id, collection.to_owned()), def); + self.sources.rebuild(&self.tree_defs); + // The new sources may already hold rows no reload has read. + self.progress.mark_reload_needed(); + self.bump_tenant_version(tenant_id); } - /// Remove a permission tree definition for a collection. + /// Remove a permission tree definition for a collection. Removing an + /// absent definition changes nothing. pub fn unregister_tree_def(&mut self, tenant_id: u64, collection: &str) { - self.tree_defs.remove(&(tenant_id, collection.to_owned())); + if self + .tree_defs + .remove(&(tenant_id, collection.to_owned())) + .is_none() + { + return; + } + self.sources.rebuild(&self.tree_defs); + self.progress.mark_reload_needed(); + self.bump_tenant_version(tenant_id); info!(tenant_id, collection, "permission_tree: unregistered"); } @@ -386,7 +500,66 @@ mod tests { assert!(cache.get_tree_def(1, "other").is_none()); assert!(cache.get_tree_def(2, "documents").is_none()); + // The DDL node and the metadata applier both apply one change. + let version = cache.tenant_version(1); + cache.register_tree_def(1, "documents", def); + assert_eq!(cache.tenant_version(1), version); + cache.unregister_tree_def(1, "documents"); assert!(cache.get_tree_def(1, "documents").is_none()); + let version = cache.tenant_version(1); + cache.unregister_tree_def(1, "documents"); + assert_eq!(cache.tenant_version(1), version); + } + + #[test] + fn a_reload_replaces_the_tenant_state_and_bumps_its_version() { + let mut cache = PermissionCache::new(); + cache.put_edge(1, "doc-1", "folder-1"); + cache.put_grant( + 1, + &PermissionGrant { + resource_id: "doc-1".into(), + grantee: "user-1".into(), + level: "viewer".into(), + inherited: false, + }, + ); + let before = cache.bump_tenant_version(1); + + cache.replace_tenant_state(1, &[("doc-2".into(), "folder-2".into())], &[]); + + assert_eq!(cache.get_parent(1, "doc-1"), None); + assert_eq!(cache.get_parent(1, "doc-2"), Some("folder-2")); + assert!(cache.get_grant(1, "doc-1", "user-1").is_none()); + assert!(cache.tenant_version(1) > before); + } + + #[test] + fn tree_sources_name_the_governed_collection_and_the_permission_table() { + let mut cache = PermissionCache::new(); + let def: PermissionTreeDef = sonic_rs::from_str( + r#"{"resource_column":"id","graph_index":"tree","permission_table":"grants"}"#, + ) + .expect("tree def"); + cache.register_tree_def(1, "docs", def); + let mut sources = cache.tree_sources(); + sources.sort_by(|a, b| a.collection.cmp(&b.collection)); + assert_eq!( + sources, + vec![ + TreeSource { + tenant_id: 1, + collection: "docs".into(), + kind: TreeSourceKind::Hierarchy, + }, + TreeSource { + tenant_id: 1, + collection: "grants".into(), + kind: TreeSourceKind::Grants, + }, + ] + ); + assert!(cache.progress().needs_reload_for(&[])); } } diff --git a/nodedb/src/control/security/permission_tree/event_handler.rs b/nodedb/src/control/security/permission_tree/event_handler.rs index bc7f97769..828a27b4f 100644 --- a/nodedb/src/control/security/permission_tree/event_handler.rs +++ b/nodedb/src/control/security/permission_tree/event_handler.rs @@ -1,9 +1,16 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Event Plane integration: process WriteEvents that affect permission trees. +//! Event Plane integration: the permission step. //! -//! When a row is written to a collection that serves as a permission table -//! or resource hierarchy for some permission tree, update the in-memory cache. +//! The Event Plane passes every event it takes off a core's ring through +//! [`apply_ring_events`], in ring order. An event for a permission table or a +//! governed collection updates the cache. Every event advances the core's +//! apply progress, so lease coverage can tell how far the cache reflects the +//! core's writes (see [`super::sync_state`]). +//! +//! Events rebuilt from the WAL during catch-up never pass through here. Their +//! ring copies do, or, when the ring dropped them, a reload covers their +//! writes. use std::sync::Arc; @@ -15,26 +22,47 @@ use super::cache::PermissionCache; use super::invalidation; use super::types::PermissionGrant; -/// Process a WriteEvent and update the permission cache if relevant. +/// Apply the permission effect of `events`, taken off core `core_id`'s ring +/// in ring order, and advance that core's apply progress. /// -/// Called from the Event Plane consumer after CDC routing. Checks if the -/// event's collection is a permission table or resource graph for any -/// registered permission tree, and updates the cache accordingly. -pub fn handle_permission_event( - event: &WriteEvent, +/// Holds the write lock for the whole batch and across no await. `notify` +/// wakes the coverage waits for this core. +pub async fn apply_ring_events( + core_id: usize, + events: &[WriteEvent], cache: &Arc>, + notify: &tokio::sync::Notify, ) { + if events.is_empty() { + return; + } + // A test parks the permission step here to prove a permission-row write + // is not acknowledged before the step applies it. The gate sits before + // the lock, so a reload can still run. + #[cfg(feature = "failpoints")] + crate::control::fail_gate::wait("permission_tree::before_apply").await; + { + let mut guard = cache.write().await; + for event in events { + if event.op.is_data_event() + && !guard.progress().covered_by_reload(core_id, event.sequence) + { + apply_event(&mut guard, event); + } + guard.progress_mut().note_consumed(core_id, event.sequence); + } + } + notify.notify_waiters(); +} + +/// Apply one data event to the cache when its collection is a permission +/// table or a governed collection of a registered tree. +fn apply_event(cache: &mut PermissionCache, event: &WriteEvent) { let collection = event.collection.as_ref(); let tenant_id = event.tenant_id.as_u64(); - // Non-blocking lock — skip if contended; next event will catch up. - let mut guard = match cache.try_write() { - Ok(g) => g, - Err(_) => return, - }; - - let is_permission_table = guard.tree_defs_using_permission_table(tenant_id, collection); - let is_resource_graph = guard.tree_defs_using_graph(tenant_id, collection); + let is_permission_table = cache.tree_defs_using_permission_table(tenant_id, collection); + let is_resource_graph = cache.tree_defs_using_graph(tenant_id, collection); if !is_permission_table && !is_resource_graph { return; @@ -51,16 +79,13 @@ pub fn handle_permission_event( && let Some(ref val) = new_val && let Some(grant) = extract_grant(val) { - invalidation::on_grant_upsert(&mut guard, tenant_id, &grant); + invalidation::on_grant_upsert(cache, tenant_id, &grant); } if is_resource_graph && let Some(ref val) = new_val - && let (Some(child_id), Some(parent_id)) = ( - val.get("id").and_then(|v| v.as_str()), - val.get("parent_id").and_then(|v| v.as_str()), - ) + && let Some((child_id, parent_id)) = extract_edge(val) { - invalidation::on_edge_upsert(&mut guard, tenant_id, child_id, parent_id); + invalidation::on_edge_upsert(cache, tenant_id, child_id, parent_id); } } crate::event::types::WriteOp::Delete => { @@ -76,26 +101,33 @@ pub fn handle_permission_event( val.get("grantee").and_then(|v| v.as_str()), ) { - invalidation::on_grant_delete(&mut guard, tenant_id, resource_id, grantee); + invalidation::on_grant_delete(cache, tenant_id, resource_id, grantee); } if is_resource_graph && let Some(ref val) = old_val && let Some(child_id) = val.get("id").and_then(|v| v.as_str()) { - invalidation::on_edge_delete(&mut guard, tenant_id, child_id); + invalidation::on_edge_delete(cache, tenant_id, child_id); } } - _ => {} + crate::event::types::WriteOp::BulkInsert { .. } + | crate::event::types::WriteOp::BulkDelete { .. } + | crate::event::types::WriteOp::Heartbeat => {} } debug!( tenant_id, - collection, "permission_tree: cache updated from CDC event" + collection, "permission_tree: cache updated from a write event" ); } +/// Extract a `(child_id, parent_id)` edge from a governed collection's row. +pub(super) fn extract_edge(val: &serde_json::Value) -> Option<(&str, &str)> { + Some((val.get("id")?.as_str()?, val.get("parent_id")?.as_str()?)) +} + /// Extract a PermissionGrant from a JSON value (permission table row). -fn extract_grant(val: &serde_json::Value) -> Option { +pub(super) fn extract_grant(val: &serde_json::Value) -> Option { Some(PermissionGrant { resource_id: val.get("resource_id")?.as_str()?.to_owned(), grantee: val.get("grantee")?.as_str()?.to_owned(), @@ -106,3 +138,132 @@ fn extract_grant(val: &serde_json::Value) -> Option { .unwrap_or(true), }) } + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + use crate::event::types::{EventSource, RowId, WriteOp}; + use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; + + const TENANT: u64 = 1; + + fn cache_with_tree() -> Arc> { + let def = sonic_rs::from_str( + r#"{"resource_column":"id","graph_index":"docs_tree","permission_table":"grants"}"#, + ) + .expect("tree def"); + let mut cache = PermissionCache::new(); + cache.register_tree_def(TENANT, "docs", def); + cache.progress_mut().install_reload(&[0]); + Arc::new(tokio::sync::RwLock::new(cache)) + } + + fn grant_event(sequence: u64, op: WriteOp) -> WriteEvent { + let row = serde_json::json!({ + "resource_id": "d1", + "grantee": "role_a", + "level": "viewer", + "inherited": false, + }); + let body: Option> = Some(Arc::from( + nodedb_types::json_to_msgpack(&row).expect("encode grant row"), + )); + let (new_value, old_value) = match op { + WriteOp::Delete => (None, body), + _ => (body, None), + }; + WriteEvent { + sequence, + collection: Arc::from("grants"), + op, + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("g1")), + lsn: Lsn::new(sequence), + record: None, + database_id: DatabaseId::DEFAULT, + tenant_id: TenantId::new(TENANT), + vshard_id: VShardId::new(0), + source: EventSource::User, + new_value, + old_value, + system_time_ms: None, + valid_time_ms: None, + user_id: None, + statement_digest: None, + } + } + + #[tokio::test] + async fn grant_event_applies_after_a_held_read_lock_is_released() { + let cache = cache_with_tree(); + let notify = Arc::new(tokio::sync::Notify::new()); + let version_before = cache.read().await.tenant_version(TENANT); + let reader = cache.read().await; + + let apply = { + let cache = Arc::clone(&cache); + let notify = Arc::clone(¬ify); + tokio::spawn(async move { + apply_ring_events(0, &[grant_event(1, WriteOp::Insert)], &cache, ¬ify).await + }) + }; + tokio::task::yield_now().await; + assert!(!apply.is_finished(), "the update waits for the reader"); + drop(reader); + apply.await.expect("apply task"); + + let guard = cache.read().await; + assert_eq!( + guard.get_grant(TENANT, "d1", "role_a"), + Some(("viewer", false)) + ); + assert_eq!(guard.tenant_version(TENANT), version_before + 1); + assert!(guard.progress().caught_up(&[1])); + } + + /// An event a reload already covers is not applied again: the ring may + /// have dropped a later write to the same row. + #[tokio::test] + async fn an_event_covered_by_a_reload_is_not_applied_again() { + let cache = cache_with_tree(); + let notify = tokio::sync::Notify::new(); + cache.write().await.progress_mut().install_reload(&[2]); + + apply_ring_events(0, &[grant_event(2, WriteOp::Insert)], &cache, ¬ify).await; + assert!( + cache + .read() + .await + .get_grant(TENANT, "d1", "role_a") + .is_none() + ); + + apply_ring_events(0, &[grant_event(3, WriteOp::Insert)], &cache, ¬ify).await; + let guard = cache.read().await; + assert!(guard.get_grant(TENANT, "d1", "role_a").is_some()); + assert!(guard.progress().caught_up(&[3])); + } + + /// A gap in the ring leaves the core behind even though later events + /// apply. + #[tokio::test] + async fn a_dropped_event_leaves_the_core_behind() { + let cache = cache_with_tree(); + let notify = tokio::sync::Notify::new(); + apply_ring_events( + 0, + &[ + grant_event(1, WriteOp::Insert), + grant_event(3, WriteOp::Delete), + ], + &cache, + ¬ify, + ) + .await; + let guard = cache.read().await; + assert!(guard.get_grant(TENANT, "d1", "role_a").is_none()); + assert!(!guard.progress().caught_up(&[3])); + assert!(guard.progress().needs_reload_for(&[3])); + } +} diff --git a/nodedb/src/control/security/permission_tree/mod.rs b/nodedb/src/control/security/permission_tree/mod.rs index a9d4eefef..8baa47db6 100644 --- a/nodedb/src/control/security/permission_tree/mod.rs +++ b/nodedb/src/control/security/permission_tree/mod.rs @@ -3,8 +3,12 @@ pub mod cache; pub mod event_handler; pub mod invalidation; +pub mod reload; pub mod resolver; +pub mod sources; +pub mod sync_state; pub mod types; -pub use cache::PermissionCache; +pub use cache::{PermissionCache, TreeSource, TreeSourceKind}; +pub use sources::SourceIndex; pub use types::PermissionTreeDef; diff --git a/nodedb/src/control/security/permission_tree/reload.rs b/nodedb/src/control/security/permission_tree/reload.rs new file mode 100644 index 000000000..8cbf152a1 --- /dev/null +++ b/nodedb/src/control/security/permission_tree/reload.rs @@ -0,0 +1,141 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Reload the permission cache from its source collections. +//! +//! A reload reads the emitted-event counter of every core, then scans every +//! governed collection and permission table on this node's own cores. A scan +//! runs on its core after every write whose event the counter already +//! counted, so the reloaded state covers each of those events. The reload +//! holds the cache's write lock throughout: the permission step waits, so no +//! event it applies falls between the counter read and the install. +//! +//! The scans read local state, never a leader's. The cache must match this +//! node's own apply position. The scans dispatch through the read-only local +//! path, never the write funnel. + +use std::collections::HashMap; + +use nodedb_types::{QualifiedCollection, TenantId}; + +use crate::control::local_dispatch::{LocalRead, dispatch_local_read, reject_data_plane_error}; +use crate::control::state::SharedState; +use crate::types::DatabaseId; + +use super::cache::{PermissionCache, TreeSource, TreeSourceKind}; +use super::event_handler::{extract_edge, extract_grant}; +use super::types::PermissionGrant; + +/// Edges and grants a reload read for one tenant. +#[derive(Default)] +struct TenantRows { + edges: Vec<(String, String)>, + grants: Vec, +} + +/// Reload every registered tree's sources, unless the cache already reflects +/// every event at or below `targets` (another reload got there first). +/// +/// `targets` of `None` reloads unconditionally: no permission step runs, so +/// only a reload reflects writes. +pub async fn reload_all(state: &SharedState, targets: Option<&[u64]>) -> crate::Result<()> { + let mut cache = state.permission_cache.write().await; + if let Some(targets) = targets + && cache.progress().caught_up(targets) + { + return Ok(()); + } + reload_locked(state, &mut cache).await +} + +/// Reload when the cache is stale: no reload covers the registered trees yet +/// (startup, or a tree definition changed), or a core lost an event. +/// +/// Planning and the lease renewal call this. A writer waiting for its +/// acknowledgement never does: a stale cache is reloaded before the next +/// statement plans, and that reload reads the write. +pub async fn reload_if_stale(state: &SharedState) -> crate::Result<()> { + // Checked under the read lock first: planning calls this for every + // statement, and the write lock is needed only when the cache is stale. + if !state.permission_cache.read().await.progress().is_stale() { + return Ok(()); + } + let mut cache = state.permission_cache.write().await; + if !cache.progress().is_stale() { + return Ok(()); + } + reload_locked(state, &mut cache).await +} + +/// Reload every registered tree's sources into `cache`, whose write lock the +/// caller holds. +async fn reload_locked(state: &SharedState, cache: &mut PermissionCache) -> crate::Result<()> { + let fence = &state.authorization_fence; + // Read under the write lock: every event the permission step applied is + // counted, and none it applies later can be. + let emitted = fence.emitted_snapshot().unwrap_or_default(); + + let mut rows: HashMap = HashMap::new(); + for source in cache.tree_sources() { + let docs = scan_source(state, &source).await?; + let tenant = rows.entry(source.tenant_id).or_default(); + match source.kind { + TreeSourceKind::Hierarchy => tenant.edges.extend(docs.iter().filter_map(|doc| { + extract_edge(doc).map(|(child, parent)| (child.to_owned(), parent.to_owned())) + })), + TreeSourceKind::Grants => tenant.grants.extend(docs.iter().filter_map(extract_grant)), + } + } + for (tenant_id, tenant) in rows { + cache.replace_tenant_state(tenant_id, &tenant.edges, &tenant.grants); + } + cache.progress_mut().install_reload(&emitted); + fence.permission_applied().notify_waiters(); + Ok(()) +} + +/// Every row of one source collection, read on this node's own core. +/// +/// A source that no longer exists holds no rows. A source that is not a +/// document collection cannot carry edges or grants, and refuses the reload: +/// planning with it would silently grant or deny nothing. +async fn scan_source( + state: &SharedState, + source: &TreeSource, +) -> crate::Result> { + let database_id = DatabaseId::DEFAULT; + let catalog = state.credentials.catalog(); + let stored = catalog.get_collection(database_id, source.tenant_id, &source.collection)?; + let Some(stored) = stored.filter(|collection| collection.is_active) else { + return Ok(Vec::new()); + }; + if !stored.collection_type.is_document() { + return Err(crate::Error::FeatureNotSupported { + detail: format!( + "permission tree source '{}' is a {} collection; the hierarchy and the \ + permission table must be document collections", + source.collection, stored.collection_type + ), + }); + } + + // A read-only dispatch: a reload can run while a write waits for its + // acknowledgement, and it must never issue a request through the write + // path. + let response = dispatch_local_read( + state, + TenantId::new(source.tenant_id), + database_id, + nodedb_types::CollectionKey::from_bare(database_id, &source.collection).vshard(), + LocalRead::DocumentScan { + collection: QualifiedCollection::new(database_id, &source.collection), + }, + ) + .await?; + reject_data_plane_error(&response)?; + Ok( + crate::data::executor::response_codec::decode_raw_scan_to_docs(response.payload.as_bytes()) + .into_iter() + .filter_map(|(_, body)| nodedb_types::json_from_msgpack(&body).ok()) + .collect(), + ) +} diff --git a/nodedb/src/control/security/permission_tree/sources.rs b/nodedb/src/control/security/permission_tree/sources.rs new file mode 100644 index 000000000..ef27af7f1 --- /dev/null +++ b/nodedb/src/control/security/permission_tree/sources.rs @@ -0,0 +1,178 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Which collections feed a permission tree, readable without the cache lock. +//! +//! A write path decides on every write whether the write changes +//! authorization state: it does when it writes a governed collection or a +//! permission table of a registered tree. That check runs on hot write paths, +//! so it reads this index under a plain read lock rather than the cache's +//! async lock. +//! +//! Two sets feed the answer, and a collection in either counts: +//! +//! - **Cached:** rebuilt by the cache whenever a tree definition changes. +//! - **Committed:** updated by the metadata applier as it commits a change, +//! before the cache takes it from the queue. +//! +//! A write therefore counts as an authorization change from the moment its +//! tree definition committed on this node. A removed tree may count a little +//! longer, until both sets drop it, which only adds a barrier. +//! +//! Tree sources live in the default database, as the permission-tree DDL +//! writes them, so each source collection homes on one vShard. + +use std::collections::{HashMap, HashSet}; +use std::sync::RwLock; + +use crate::types::DatabaseId; + +use super::types::PermissionTreeDef; + +#[derive(Debug, Default)] +struct SourceSet { + collections: HashSet, + vshards: HashSet, +} + +impl SourceSet { + fn from_defs<'a>( + defs: impl IntoIterator, + ) -> Self { + let mut set = Self::default(); + for ((_, governed), def) in defs { + for collection in [governed.as_str(), def.permission_table.as_str()] { + set.collections.insert(collection.to_owned()); + set.vshards.insert( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection) + .vshard() + .as_u32(), + ); + } + } + set + } +} + +#[derive(Debug, Default)] +struct Committed { + defs: HashMap<(u64, String), PermissionTreeDef>, + set: SourceSet, +} + +/// The source collections of every registered or committed tree. +#[derive(Debug, Default)] +pub struct SourceIndex { + cached: RwLock, + committed: RwLock, +} + +impl SourceIndex { + /// Rebuild the cached set from the cache's tree definitions. + pub(super) fn rebuild(&self, tree_defs: &HashMap<(u64, String), PermissionTreeDef>) { + *self.cached.write().unwrap_or_else(|p| p.into_inner()) = SourceSet::from_defs(tree_defs); + } + + /// Record a tree definition the metadata applier committed. `None` + /// removes the tree of `(tenant_id, collection)`. + pub fn note_committed( + &self, + tenant_id: u64, + collection: &str, + def: Option<&PermissionTreeDef>, + ) { + let mut committed = self.committed.write().unwrap_or_else(|p| p.into_inner()); + let key = (tenant_id, collection.to_owned()); + match def { + Some(def) => { + committed.defs.insert(key, def.clone()); + } + None => { + committed.defs.remove(&key); + } + } + committed.set = SourceSet::from_defs(&committed.defs); + } + + fn any(&self, test: impl Fn(&SourceSet) -> bool) -> bool { + test(&self.cached.read().unwrap_or_else(|p| p.into_inner())) + || test(&self.committed.read().unwrap_or_else(|p| p.into_inner()).set) + } + + /// Whether no tree is registered or committed. + pub fn is_empty(&self) -> bool { + !self.any(|set| !set.collections.is_empty()) + } + + /// Whether `collection` feeds a tree. + pub fn is_source_collection(&self, collection: &str) -> bool { + self.any(|set| set.collections.contains(collection)) + } + + /// Whether `vshard_id` homes a collection that feeds a tree. + pub fn is_source_vshard(&self, vshard_id: u32) -> bool { + self.any(|set| set.vshards.contains(&vshard_id)) + } + + /// Every vShard that homes a source collection. + pub fn source_vshards(&self) -> Vec { + let mut vshards: HashSet = self + .cached + .read() + .unwrap_or_else(|p| p.into_inner()) + .vshards + .clone(); + vshards.extend( + self.committed + .read() + .unwrap_or_else(|p| p.into_inner()) + .set + .vshards + .iter() + .copied(), + ); + vshards.into_iter().collect() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn the_index_names_governed_collections_and_permission_tables() { + let def: PermissionTreeDef = sonic_rs::from_str( + r#"{"resource_column":"id","graph_index":"tree","permission_table":"grants"}"#, + ) + .expect("tree def"); + let mut defs = HashMap::new(); + defs.insert((1, "docs".to_owned()), def); + let index = SourceIndex::default(); + assert!(index.is_empty()); + index.rebuild(&defs); + assert!(index.is_source_collection("docs")); + assert!(index.is_source_collection("grants")); + assert!(!index.is_source_collection("other")); + let grants_vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "grants") + .vshard() + .as_u32(); + assert!(index.is_source_vshard(grants_vshard)); + defs.clear(); + index.rebuild(&defs); + assert!(index.is_empty()); + assert!(!index.is_source_vshard(grants_vshard)); + } + + #[test] + fn a_committed_tree_counts_before_the_cache_takes_it() { + let def: PermissionTreeDef = sonic_rs::from_str( + r#"{"resource_column":"id","graph_index":"tree","permission_table":"grants"}"#, + ) + .expect("tree def"); + let index = SourceIndex::default(); + index.note_committed(1, "docs", Some(&def)); + assert!(index.is_source_collection("grants")); + assert!(index.is_source_collection("docs")); + index.note_committed(1, "docs", None); + assert!(index.is_empty()); + } +} diff --git a/nodedb/src/control/security/permission_tree/sync_state.rs b/nodedb/src/control/security/permission_tree/sync_state.rs new file mode 100644 index 000000000..8fc34fdda --- /dev/null +++ b/nodedb/src/control/security/permission_tree/sync_state.rs @@ -0,0 +1,198 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! How far the permission cache reflects each core's writes. +//! +//! Every Data Plane core numbers the events it emits with a contiguous +//! sequence. The Event Plane passes every event it takes off a core's ring +//! through the permission step, in ring order. For each core the cache keeps +//! `applied`: every event numbered at or below it was either applied to the +//! cache, or its write is covered by a reload. +//! +//! `applied` only advances by one. A dropped event leaves a gap the step can +//! never close, so the core stays behind until a reload covers it. A reload +//! reads the emitted counters first, then scans the source collections. Each +//! scan runs on its core after every write whose event was counted, so the +//! reload covers every event numbered at or below the counter it read. +//! +//! A reload also names, per core, the highest event it covers. An event at or +//! below that number reaching the step later is not applied again. Its write +//! is in the reload, and a later write the ring dropped may have overwritten +//! it. + +/// Apply progress of one core. +#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)] +struct CoreApply { + /// Every event at or below this number is reflected in the cache. + applied: u64, + /// Every event at or below this number is covered by the last reload. + reloaded_through: u64, + /// An event above `applied + 1` passed the step: one was dropped. + gap: bool, +} + +/// Apply progress of the permission cache, per core. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct ApplyProgress { + cores: Vec, + /// No reload has covered the registered trees yet. Set at startup, before + /// the durable grants are loaded, and whenever a tree definition changes. + needs_reload: bool, +} + +impl Default for ApplyProgress { + fn default() -> Self { + Self::new() + } +} + +impl ApplyProgress { + /// Progress of a cache that holds no source data yet. + pub fn new() -> Self { + Self { + cores: Vec::new(), + needs_reload: true, + } + } + + fn core_mut(&mut self, core_id: usize) -> &mut CoreApply { + if self.cores.len() <= core_id { + self.cores.resize(core_id + 1, CoreApply::default()); + } + &mut self.cores[core_id] + } + + fn core(&self, core_id: usize) -> CoreApply { + self.cores.get(core_id).copied().unwrap_or_default() + } + + /// Whether the last reload covers event `sequence` of `core_id`. Such an + /// event is never applied again. + pub fn covered_by_reload(&self, core_id: usize, sequence: u64) -> bool { + sequence <= self.core(core_id).reloaded_through + } + + /// Record that event `sequence` of `core_id` passed the permission step. + pub fn note_consumed(&mut self, core_id: usize, sequence: u64) { + let core = self.core_mut(core_id); + if sequence <= core.applied { + return; + } + if sequence == core.applied + 1 { + core.applied = sequence; + } else { + core.gap = true; + } + } + + /// Whether planning must reload before it reads the cache: no reload + /// covers the registered trees, or a core lost an event the step can + /// never apply. + pub fn is_stale(&self) -> bool { + self.needs_reload || self.cores.iter().any(|core| core.gap) + } + + /// Record that the source set changed: a tree definition was registered + /// or removed. + pub fn mark_reload_needed(&mut self) { + self.needs_reload = true; + } + + /// Whether the cache reflects every event numbered at or below + /// `targets[core]` on each core. + pub fn caught_up(&self, targets: &[u64]) -> bool { + !self.needs_reload + && targets + .iter() + .enumerate() + .all(|(core_id, target)| self.core(core_id).applied >= *target) + } + + /// Whether waiting cannot reach `targets`: no reload covered the sources + /// yet, or a core that is behind its target lost an event. + pub fn needs_reload_for(&self, targets: &[u64]) -> bool { + self.needs_reload + || targets.iter().enumerate().any(|(core_id, target)| { + let core = self.core(core_id); + core.gap && core.applied < *target + }) + } + + /// Record a reload that covers every event at or below `emitted[core]`. + pub fn install_reload(&mut self, emitted: &[u64]) { + for (core_id, through) in emitted.iter().enumerate() { + let core = self.core_mut(core_id); + core.applied = core.applied.max(*through); + core.reloaded_through = core.reloaded_through.max(*through); + // Every event the step saw was emitted before the counter was + // read, so the reload covers any gap among them. + core.gap = false; + } + self.needs_reload = false; + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn reloaded(emitted: &[u64]) -> ApplyProgress { + let mut progress = ApplyProgress::new(); + progress.install_reload(emitted); + progress + } + + #[test] + fn a_fresh_cache_needs_a_reload_before_it_is_caught_up() { + let progress = ApplyProgress::new(); + assert!(!progress.caught_up(&[])); + assert!(progress.needs_reload_for(&[])); + assert!(reloaded(&[0]).caught_up(&[0])); + } + + #[test] + fn contiguous_events_advance_the_core() { + let mut progress = reloaded(&[0, 0]); + progress.note_consumed(1, 1); + progress.note_consumed(1, 2); + assert!(progress.caught_up(&[0, 2])); + assert!(!progress.caught_up(&[0, 3])); + assert!(!progress.needs_reload_for(&[0, 3])); + } + + #[test] + fn a_dropped_event_holds_the_core_until_a_reload() { + let mut progress = reloaded(&[0]); + progress.note_consumed(0, 1); + progress.note_consumed(0, 3); + progress.note_consumed(0, 4); + assert!(!progress.caught_up(&[4])); + assert!(progress.needs_reload_for(&[4])); + // A target the core reached before the gap needs nothing. + assert!(!progress.needs_reload_for(&[1])); + + progress.install_reload(&[4]); + assert!(progress.caught_up(&[4])); + assert!(progress.covered_by_reload(0, 3)); + assert!(!progress.covered_by_reload(0, 5)); + progress.note_consumed(0, 5); + assert!(progress.caught_up(&[5])); + } + + #[test] + fn a_tree_change_requires_a_new_reload() { + let mut progress = reloaded(&[2]); + progress.mark_reload_needed(); + assert!(!progress.caught_up(&[2])); + assert!(progress.needs_reload_for(&[2])); + } + + #[test] + fn a_lost_event_makes_the_cache_stale_until_a_reload() { + let mut progress = reloaded(&[0]); + assert!(!progress.is_stale()); + progress.note_consumed(0, 2); + assert!(progress.is_stale()); + progress.install_reload(&[2]); + assert!(!progress.is_stale()); + } +} diff --git a/nodedb/src/control/security/permission_tree/types.rs b/nodedb/src/control/security/permission_tree/types.rs index c51c76dc8..8766807bf 100644 --- a/nodedb/src/control/security/permission_tree/types.rs +++ b/nodedb/src/control/security/permission_tree/types.rs @@ -25,7 +25,7 @@ pub const DEFAULT_DELETE_LEVEL: &str = "owner"; /// /// Stored as JSON in `StoredCollection.permission_tree_def`. /// Binds the collection to a resource hierarchy graph and a permission table. -#[derive(Debug, Clone, Serialize, Deserialize)] +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] pub struct PermissionTreeDef { /// Column in this collection that serves as the resource identifier. /// Used to look up the resource in the permission graph. diff --git a/nodedb/src/control/security/role.rs b/nodedb/src/control/security/role.rs index 11d59feec..0a21b09fe 100644 --- a/nodedb/src/control/security/role.rs +++ b/nodedb/src/control/security/role.rs @@ -126,30 +126,7 @@ impl RoleStore { tenant_id: TenantId, parent: Option<&str>, ) -> crate::Result { - if is_builtin(name) { - return Err(crate::Error::BadRequest { - detail: format!("'{name}' is a built-in role and cannot be created"), - }); - } - let roles = self.roles.read(); - if roles.contains_key(name) { - return Err(crate::Error::BadRequest { - detail: format!("role '{name}' already exists"), - }); - } - if let Some(parent_name) = parent { - validate_parent(name, parent_name, &roles)?; - } - let now = std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap_or_default() - .as_secs(); - Ok(StoredRole { - name: name.to_string(), - tenant_id: tenant_id.as_u64(), - parent: parent.unwrap_or("").to_string(), - created_at: now, - }) + prepare_role_against(name, tenant_id, parent, &self.roles.read()) } /// Create a custom role. Returns error if it already exists or would create a cycle. @@ -216,10 +193,16 @@ impl RoleStore { let mut roles = self.roles.write(); // Check no other role inherits from this one. - let has_children = roles.values().any(|r| r.parent.as_deref() == Some(name)); - if has_children { - return Err(crate::Error::BadRequest { - detail: format!("cannot drop role '{name}': other roles inherit from it"), + let mut children: Vec = roles + .values() + .filter(|r| r.parent.as_deref() == Some(name)) + .map(|r| r.name.clone()) + .collect(); + if !children.is_empty() { + children.sort(); + return Err(crate::Error::RoleInUse { + role: name.to_string(), + dependents: super::role_assignment::RoleDependents::ChildRoles(children), }); } @@ -309,6 +292,52 @@ impl RoleStore { } } +/// Build a `StoredRole` ready for replication via `CatalogEntry::PutRole`, +/// validated against `roles`: the custom roles the creating statement sees. +/// Inside a transaction those include the roles it created earlier. Rejects a +/// built-in name, a duplicate, an undefined parent, and a parent that would +/// close a cycle or exceed [`MAX_ROLE_INHERITANCE_DEPTH`]. +pub fn prepare_role_against( + name: &str, + tenant_id: TenantId, + parent: Option<&str>, + roles: &HashMap, +) -> crate::Result { + if is_builtin(name) { + return Err(crate::Error::BadRequest { + detail: format!("'{name}' is a built-in role and cannot be created"), + }); + } + if roles.contains_key(name) { + return Err(crate::Error::BadRequest { + detail: format!("role '{name}' already exists"), + }); + } + if let Some(parent_name) = parent { + validate_parent(name, parent_name, roles)?; + } + let now = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap_or_default() + .as_secs(); + Ok(StoredRole { + name: name.to_string(), + tenant_id: tenant_id.as_u64(), + parent: parent.unwrap_or("").to_string(), + created_at: now, + }) +} + +/// [`RoleStore::check_inheritance_cycle`] against `roles`, the custom roles +/// the altering statement sees. +pub fn check_inheritance_cycle_against( + role_name: &str, + parent: &str, + roles: &HashMap, +) -> crate::Result<()> { + check_inheritance_chain(role_name, parent, roles) +} + /// Walk the inheritance chain starting from `start_name` upward through the /// given `roles` map. Returns the chain length (number of hops including /// `start_name` itself). If the chain is a cycle or exceeds @@ -375,11 +404,11 @@ fn validate_parent( check_inheritance_chain(child_name, parent_name, roles) } +/// Every name the role parser maps to a built-in role, so a custom role can +/// never shadow one: `cluster_admin` and the `database_*:{id}` forms as well +/// as the five tenant roles. fn is_builtin(name: &str) -> bool { - matches!( - name, - "superuser" | "tenant_admin" | "readwrite" | "readonly" | "monitor" - ) + super::role_assignment::is_builtin_role_name(name) } #[cfg(test)] diff --git a/nodedb/src/control/security/role_assignment.rs b/nodedb/src/control/security/role_assignment.rs new file mode 100644 index 000000000..7b758f21f --- /dev/null +++ b/nodedb/src/control/security/role_assignment.rs @@ -0,0 +1,317 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Which role names a user can hold, and when a custom role can be dropped. +//! +//! A role name parses to a built-in [`Role`] or to [`Role::Custom`]. A custom +//! name is only a name: it grants something only once a custom role of that +//! name is defined in the user's tenant. So a user may hold a custom role only +//! while it is defined there, and a custom role may be dropped only while no +//! user holds it and no role inherits from it, as PostgreSQL refuses to drop +//! a role that objects still depend on. +//! +//! The same rules run in two places, each against its own view of the +//! committed catalog: +//! +//! - **Proposal:** the DDL handler checks the in-memory role and user stores, +//! and refuses the statement before anything is proposed. +//! - **Apply:** the metadata applier checks the redb catalog at the entry's +//! log position. Every node applies the same log in the same order, so a +//! `PutUser` or `DeleteRole` that raced another change is skipped on every +//! node alike, and a replayed log never produces a user holding an +//! undefined role. + +use std::fmt; + +use crate::control::security::catalog::{StoredRole, StoredUser}; + +use super::identity::Role; + +/// Why a role assignment or a role drop is refused. +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +pub enum RoleRefusal { + /// The role is neither built in nor defined in the user's tenant. + #[error("role \"{name}\" does not exist")] + Undefined { name: String }, + /// Users still hold the role. + #[error( + "role \"{name}\" cannot be dropped because users still hold it: {}", + .users.join(", ") + )] + HeldByUsers { name: String, users: Vec }, + /// Other roles inherit from the role. + #[error( + "role \"{name}\" cannot be dropped because other roles inherit from it: {}", + .children.join(", ") + )] + InheritedBy { name: String, children: Vec }, +} + +impl RoleRefusal { + /// The SQLSTATE PostgreSQL gives for the same refusal. + pub fn sqlstate(&self) -> &'static str { + match self { + Self::Undefined { .. } => "42704", + Self::HeldByUsers { .. } | Self::InheritedBy { .. } => "2BP01", + } + } +} + +/// What still depends on a custom role, so a DROP ROLE of it is refused. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum RoleDependents { + /// Users that hold the role. + Users(Vec), + /// Roles that inherit from the role. + ChildRoles(Vec), +} + +impl fmt::Display for RoleDependents { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::Users(users) => write!(f, "users still hold it: {}", users.join(", ")), + Self::ChildRoles(children) => { + write!(f, "other roles inherit from it: {}", children.join(", ")) + } + } + } +} + +impl From for crate::Error { + fn from(refusal: RoleRefusal) -> Self { + match refusal { + RoleRefusal::Undefined { name } => crate::Error::UndefinedObject { kind: "role", name }, + RoleRefusal::HeldByUsers { name, users } => crate::Error::RoleInUse { + role: name, + dependents: RoleDependents::Users(users), + }, + RoleRefusal::InheritedBy { name, children } => crate::Error::RoleInUse { + role: name, + dependents: RoleDependents::ChildRoles(children), + }, + } + } +} + +/// Parse a role name. Unknown names become [`Role::Custom`]. +pub fn parse_role_name(name: &str) -> Role { + match name.parse() { + Ok(role) => role, + Err(e) => match e {}, + } +} + +/// Whether `name` names a built-in role, so it can never be a custom role. +pub fn is_builtin_role_name(name: &str) -> bool { + !matches!(parse_role_name(name), Role::Custom(_)) +} + +/// Check that a user of `tenant_id` can hold every role in `roles`. +/// +/// `custom_tenant` returns the tenant a custom role of the given name is +/// defined in, or `None` when no such role is defined. +pub fn check_assignable<'a>( + roles: impl IntoIterator, + tenant_id: u64, + custom_tenant: impl Fn(&str) -> Option, +) -> Result<(), RoleRefusal> { + for role in roles { + if let Role::Custom(name) = role + && custom_tenant(name) != Some(tenant_id) + { + return Err(RoleRefusal::Undefined { name: name.clone() }); + } + } + Ok(()) +} + +/// Check that the custom role `name` can be dropped: no user in `users` +/// holds it, and no role in `roles` inherits from it. +pub fn check_droppable<'a>( + name: &str, + users: impl IntoIterator, + roles: impl IntoIterator, +) -> Result<(), RoleRefusal> { + let mut holders: Vec = users + .into_iter() + .filter(|(_, held)| held.iter().any(|role| role == name)) + .map(|(user, _)| user.to_string()) + .collect(); + if !holders.is_empty() { + holders.sort(); + return Err(RoleRefusal::HeldByUsers { + name: name.to_string(), + users: holders, + }); + } + let mut children: Vec = roles + .into_iter() + .filter(|(_, parent)| *parent == name) + .map(|(child, _)| child.to_string()) + .collect(); + if !children.is_empty() { + children.sort(); + return Err(RoleRefusal::InheritedBy { + name: name.to_string(), + children, + }); + } + Ok(()) +} + +/// [`check_assignable`] for a stored user against the stored roles. +pub fn check_stored_user(user: &StoredUser, roles: &[StoredRole]) -> Result<(), RoleRefusal> { + let parsed: Vec = user + .roles + .iter() + .map(String::as_str) + .map(parse_role_name) + .collect(); + check_assignable(&parsed, user.tenant_id, |name| { + roles + .iter() + .find(|role| role.name == name) + .map(|role| role.tenant_id) + }) +} + +/// [`check_droppable`] for the stored role `name` against the stored users +/// and roles. Inactive users are dropped users and hold nothing. +pub fn check_stored_drop( + name: &str, + users: &[StoredUser], + roles: &[StoredRole], +) -> Result<(), RoleRefusal> { + check_droppable( + name, + users + .iter() + .filter(|user| user.is_active) + .map(|user| (user.username.as_str(), user.roles.as_slice())), + roles + .iter() + .map(|role| (role.name.as_str(), role.parent.as_str())), + ) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn custom(name: &str) -> Role { + Role::Custom(name.to_string()) + } + + #[test] + fn built_in_roles_are_always_assignable() { + let roles = [ + Role::Superuser, + Role::TenantAdmin, + Role::ReadWrite, + Role::ReadOnly, + Role::Monitor, + ]; + assert_eq!(check_assignable(&roles, 1, |_| None), Ok(())); + } + + #[test] + fn an_undefined_custom_role_is_refused_by_name() { + let roles = [Role::ReadWrite, custom("read_write")]; + assert_eq!( + check_assignable(&roles, 1, |_| None), + Err(RoleRefusal::Undefined { + name: "read_write".into() + }) + ); + assert_eq!( + RoleRefusal::Undefined { name: "x".into() }.sqlstate(), + "42704" + ); + } + + #[test] + fn a_custom_role_is_assignable_only_in_its_own_tenant() { + let lookup = |name: &str| (name == "analyst").then_some(7); + assert_eq!(check_assignable(&[custom("analyst")], 7, lookup), Ok(())); + assert!(check_assignable(&[custom("analyst")], 8, lookup).is_err()); + } + + #[test] + fn a_held_or_inherited_role_cannot_be_dropped() { + let held = vec!["analyst".to_string()]; + let none: Vec = Vec::new(); + assert_eq!( + check_droppable( + "analyst", + [("bob", held.as_slice()), ("amy", none.as_slice())], + [] + ), + Err(RoleRefusal::HeldByUsers { + name: "analyst".into(), + users: vec!["bob".into()] + }) + ); + assert_eq!( + check_droppable( + "analyst", + [("amy", none.as_slice())], + [("junior", "analyst")] + ), + Err(RoleRefusal::InheritedBy { + name: "analyst".into(), + children: vec!["junior".into()] + }) + ); + assert_eq!( + check_droppable("analyst", [("amy", none.as_slice())], [("junior", "")]), + Ok(()) + ); + } + + /// A role still in use keeps the `2BP01` class as a `crate::Error`, and + /// its message names the role and its dependents with no tenant and no + /// CASCADE hint. + #[test] + fn a_role_in_use_is_a_dependent_objects_error() { + use crate::control::server::pgwire::types::error_to_sqlstate; + + for refusal in [ + RoleRefusal::HeldByUsers { + name: "analyst".into(), + users: vec!["bob".into()], + }, + RoleRefusal::InheritedBy { + name: "analyst".into(), + children: vec!["junior".into()], + }, + ] { + let message = refusal.to_string(); + let err = crate::Error::from(refusal); + let (_, state, rendered) = error_to_sqlstate(&err); + assert_eq!(state, "2BP01"); + assert_eq!(rendered, message); + assert!(!rendered.contains("tenant"), "{rendered}"); + assert!(!rendered.contains("CASCADE"), "{rendered}"); + let public = crate::error_classify::classify(&err); + assert_eq!( + public.code(), + nodedb_types::error::ErrorCode::DEPENDENT_OBJECTS_EXIST + ); + assert_eq!(public.message(), message); + } + } + + #[test] + fn every_parsed_built_in_name_is_built_in() { + for name in [ + "superuser", + "cluster_admin", + "tenant_admin", + "readwrite", + "readonly", + "monitor", + ] { + assert!(is_builtin_role_name(name), "{name}"); + } + assert!(!is_builtin_role_name("read_write")); + } +} diff --git a/nodedb/src/control/sequence/error_map.rs b/nodedb/src/control/sequence/error_map.rs index 5fd09090f..74cf7a332 100644 --- a/nodedb/src/control/sequence/error_map.rs +++ b/nodedb/src/control/sequence/error_map.rs @@ -21,7 +21,13 @@ pub(crate) fn undefined_sequence(name: &str) -> SqlError { pub(crate) fn map_sequence_error(name: &str, error: SequenceError) -> SqlError { match error { SequenceError::NotFound { .. } => undefined_sequence(name), - other => SqlError::ObjectNotInPrerequisiteState { + other @ (SequenceError::Exhausted { .. } + | SequenceError::NotYetCalled { .. } + | SequenceError::OutOfRange { .. } + | SequenceError::AlreadyExists { .. } + | SequenceError::InvalidDefinition { .. } + | SequenceError::FormatParse { .. } + | SequenceError::InvalidResetScope { .. }) => SqlError::ObjectNotInPrerequisiteState { object: name.to_string(), detail: other.to_string(), }, @@ -40,7 +46,13 @@ pub(crate) fn sequence_error_to_error(name: &str, error: SequenceError) -> crate kind: "sequence", name: name.to_string(), }, - other => crate::Error::ObjectNotInPrerequisiteState { + other @ (SequenceError::Exhausted { .. } + | SequenceError::NotYetCalled { .. } + | SequenceError::OutOfRange { .. } + | SequenceError::AlreadyExists { .. } + | SequenceError::InvalidDefinition { .. } + | SequenceError::FormatParse { .. } + | SequenceError::InvalidResetScope { .. }) => crate::Error::ObjectNotInPrerequisiteState { object: name.to_string(), detail: other.to_string(), }, diff --git a/nodedb/src/control/server/broadcast.rs b/nodedb/src/control/server/broadcast.rs index 82034dc81..a6603f2fd 100644 --- a/nodedb/src/control/server/broadcast.rs +++ b/nodedb/src/control/server/broadcast.rs @@ -44,7 +44,7 @@ pub(crate) fn broadcast_call_count_increment() { /// a constraint refusal keeps its own SQLSTATE. A `{code:?}` dump into a /// generic dispatch failure made every one of them read as internal. fn typed_core_error(resp: &Response) -> crate::Error { - match crate::control::server::dispatch_utils::reject_data_plane_error(resp) { + match crate::control::local_dispatch::reject_data_plane_error(resp) { Err(error) => error, // `NotFound` is an empty observation elsewhere. A barrier takes any // error status as a core that did not acknowledge. diff --git a/nodedb/src/control/server/calvin_submit/hook.rs b/nodedb/src/control/server/calvin_submit/hook.rs index 9e2334189..57a061d0a 100644 --- a/nodedb/src/control/server/calvin_submit/hook.rs +++ b/nodedb/src/control/server/calvin_submit/hook.rs @@ -76,7 +76,15 @@ impl nodedb_cluster::CalvinSubmit for RegistryCalvinSubmit { }; // Re-derive the participating-vshard set skipped during serialization // (the wire bytes carry only the read/write sets). - tx_class.restore_derived(); + if let Err(e) = tx_class.restore_derived() { + return SubmitCalvinTxnResponse { + error: Some(TypedClusterError::Internal { + code: 0, + message: format!("calvin-submit: TxClass participants underivable: {e}"), + }), + payload_bytes: None, + }; + } let timeout = Duration::from_millis(req.deadline_remaining_ms.max(1)); match submit_and_await_calvin_with_timeout(&self.state, tx_class, timeout).await { diff --git a/nodedb/src/control/server/calvin_submit/inbox_hook.rs b/nodedb/src/control/server/calvin_submit/inbox_hook.rs index 161ca1385..b585f12e4 100644 --- a/nodedb/src/control/server/calvin_submit/inbox_hook.rs +++ b/nodedb/src/control/server/calvin_submit/inbox_hook.rs @@ -86,7 +86,18 @@ impl nodedb_cluster::CalvinSubmitInbox for RegistryCalvinSubmitInbox { }; // Re-derive the participating-vshard set skipped during serialization // (the wire bytes carry only the read/write sets). - tx_class.restore_derived(); + if let Err(e) = tx_class.restore_derived() { + return SubmitCalvinInboxResponse { + inbox_seq: 0, + epoch: 0, + position: 0, + participants: 0, + error: Some(TypedClusterError::Internal { + code: 0, + message: format!("calvin-inbox: TxClass participants underivable: {e}"), + }), + }; + } let timeout = Duration::from_millis(req.deadline_remaining_ms.max(1)); match submit_local_assign(&self.state, tx_class, timeout).await { diff --git a/nodedb/src/control/server/dispatch_utils/dispatch.rs b/nodedb/src/control/server/dispatch_utils/dispatch.rs index 1100ec601..7a69a081c 100644 --- a/nodedb/src/control/server/dispatch_utils/dispatch.rs +++ b/nodedb/src/control/server/dispatch_utils/dispatch.rs @@ -9,12 +9,17 @@ use crate::control::server::shared::clone_write::CloneCheckedTask; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; +use super::minted::{MintedRecords, RecordOwner}; use super::submit_write::{ ChangeFeedOwner, SubmitWrite, WalDurability, WriteOrdering, submit_write, }; use super::types::{AutocommitWrite, DataPlaneDispatch, WriteDispatch}; /// Dispatch a clone-checked, capability-bearing external task to the Data Plane. +/// +/// The read route: it appends no WAL record. A write whose caller owns no +/// record for it goes through `dispatch_authorized_durable_write`, or +/// `dispatch_authorized_task_by_class` where one call site carries both. pub async fn dispatch_authorized_to_data_plane( shared: &SharedState, checked: CloneCheckedTask, @@ -34,6 +39,38 @@ pub async fn dispatch_authorized_to_data_plane( durability: WalDurability::CallerSupplied { wal_lsn: None, resolved_now_ms: None, + minted: None, + }, + }, + ) + .await +} + +/// Dispatch a clone-checked task whose records the caller already appended +/// under `minted`. The request carries the highest of their LSNs, so the +/// write is durable before it is acknowledged. The funnel closes their +/// outcome-floor window from the task's outcome. +pub(crate) async fn dispatch_authorized_minted_to_data_plane( + shared: &SharedState, + checked: CloneCheckedTask, + trace_id: TraceId, + minted: MintedRecords, +) -> crate::Result { + let task = checked.into_authorized().into_physical_task(); + dispatch_to_data_plane_inner( + shared, + DataPlaneDispatch { + tenant_id: task.tenant_id, + database_id: task.database_id, + vshard_id: task.vshard_id, + plan: task.plan, + trace_id, + event_source: crate::event::EventSource::User, + txn_id: task.txn_id, + durability: WalDurability::CallerSupplied { + wal_lsn: minted.highest(), + resolved_now_ms: None, + minted: Some(minted), }, }, ) @@ -57,7 +94,11 @@ pub async fn dispatch_authorized_autocommit_write( trace_id, event_source: crate::event::EventSource::User, txn_id: task.txn_id, - durability: WalDurability::AppendHere { now_override: None }, + durability: WalDurability::AppendHere { + now_override: None, + apply_key: 0, + commit_hlc: None, + }, }, ) .await @@ -87,7 +128,11 @@ pub(crate) async fn dispatch_authorized_autocommit_write_with_source( trace_id, event_source, txn_id: task.txn_id, - durability: WalDurability::AppendHere { now_override: None }, + durability: WalDurability::AppendHere { + now_override: None, + apply_key: 0, + commit_hlc: None, + }, }, ) .await @@ -146,6 +191,7 @@ pub(crate) async fn dispatch_to_data_plane_with_source( durability: WalDurability::CallerSupplied { wal_lsn: None, resolved_now_ms: None, + minted: None, }, }, ) @@ -176,6 +222,7 @@ pub(crate) async fn dispatch_trusted_internal_write_to_data_plane( txn_id, wal_lsn, resolved_now_ms, + minted, } = write; dispatch_to_data_plane_inner( shared, @@ -187,13 +234,12 @@ pub(crate) async fn dispatch_trusted_internal_write_to_data_plane( trace_id, event_source, txn_id, - // Caller pre-appended and supplied `wal_lsn` (e.g. the procedural - // batch-flush path whose dispatched plan is a `TransactionBatch` - // whose per-task records were appended upstream): the funnel must not + // Caller pre-appended and supplied `wal_lsn`: the funnel must not // append again. durability: WalDurability::CallerSupplied { wal_lsn, resolved_now_ms, + minted, }, }, ) @@ -236,7 +282,11 @@ pub(crate) async fn dispatch_autocommit_write( txn_id, // The funnel appends the WAL record under the admission guard just // before enqueue and stamps the minted LSN onto the `Request`. - durability: WalDurability::AppendHere { now_override: None }, + durability: WalDurability::AppendHere { + now_override: None, + apply_key: 0, + commit_hlc: None, + }, }, ) .await @@ -246,6 +296,9 @@ pub(crate) async fn dispatch_autocommit_write( /// id so the Data Plane can resolve this transaction's staging overlay /// (read-your-own-writes) and route `StageWrite`. Used by the native endpoint, /// whose in-transaction tasks flow through this shared path. +/// +/// It appends no WAL record, so it refuses a write that only the funnel's +/// `AppendHere` route logs. A staged write is not such a write: COMMIT logs it. pub(crate) async fn dispatch_to_data_plane_with_txn( shared: &SharedState, tenant_id: TenantId, @@ -255,6 +308,7 @@ pub(crate) async fn dispatch_to_data_plane_with_txn( trace_id: TraceId, txn_id: Option, ) -> crate::Result { + super::durability_barrier::refuse_unlogged_write(&plan)?; dispatch_to_data_plane_inner( shared, DataPlaneDispatch { @@ -271,6 +325,7 @@ pub(crate) async fn dispatch_to_data_plane_with_txn( durability: WalDurability::CallerSupplied { wal_lsn: None, resolved_now_ms: None, + minted: None, }, }, ) @@ -289,39 +344,58 @@ async fn dispatch_to_data_plane_inner( trace_id, event_source, txn_id, - durability, + mut durability, } = params; - // Resolve any Exchange data-movement nodes before dispatch: a root-level - // Gather fans the child to all cores and returns the merged response here; - // a Broadcast join child is gathered and embedded so the plan reaching a - // core is self-contained. Safe no-op for the many non-Exchange callers - // (writes, metrics, triggers). Catalog materialization is identity-scoped - // and already done upstream on the pgwire/native paths. - // Internal funnel (COPY, cursors, materialized-view refresh, constraint - // subqueries): not session-transaction-scoped, so `None`. - let plan = match crate::control::server::exchange::resolve_exchange_in_plan( - shared, - database_id, + let owner = RecordOwner { tenant_id, - plan, - trace_id, - None, - ) - .await? - { - crate::control::server::exchange::Resolved::Gathered( - resp, - _shard_watermarks, - _shuffle_reads, - ) => { - return Ok(resp); + database_id, + vshard_id, + }; + // A write that carries its own records is never a query. Only a query + // plan holds Exchange nodes, and resolving one fans it out to the cores, + // so a record-carrying query would reach the cores before any close. + let plan = if durability.has_minted() { + if matches!(plan, PhysicalPlan::Query(_)) { + if let Some(minted) = durability.take_minted() { + minted.cancel(&shared.wal, owner, 0).await?; + } + return Err(crate::Error::Internal { + detail: "a write carrying WAL records reached the funnel as a query plan; \ + nothing was dispatched" + .into(), + }); } - crate::control::server::exchange::Resolved::Plan(p) => *p, - // Internal funnel callers want a fully-collected Response, not a lazy - // stream: materialize the stream into one merged-array Response, - // preserving the prior gather-then-return behaviour on this path. - crate::control::server::exchange::Resolved::Stream(s) => { - return crate::control::server::exchange::gather::stream_to_response(s).await; + plan + } else { + // Resolve any Exchange data-movement nodes before dispatch: a + // root-level Gather fans the child to all cores and returns the merged + // response here. A Broadcast join child is gathered and embedded so + // the plan reaching a core is self-contained. Plans with no Exchange + // node pass through unchanged. Catalog materialization is + // identity-scoped and already done upstream on the pgwire and native + // paths. The internal funnel is not session-transaction-scoped, so the + // transaction id is `None`. + let resolved = crate::control::server::exchange::resolve_exchange_in_plan( + shared, + database_id, + tenant_id, + plan, + trace_id, + None, + ) + .await?; + match resolved { + crate::control::server::exchange::Resolved::Plan(p) => *p, + crate::control::server::exchange::Resolved::Gathered( + resp, + _shard_watermarks, + _shuffle_reads, + ) => return Ok(resp), + // Internal funnel callers want one merged `Response`, not a lazy + // stream. + crate::control::server::exchange::Resolved::Stream(s) => { + return crate::control::server::exchange::gather::stream_to_response(s).await; + } } }; @@ -387,7 +461,7 @@ mod tests { PhysicalTask { tenant_id, database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::from_collection_in_database(DatabaseId::DEFAULT, ARRAY), + vshard_id: nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, ARRAY).vshard(), plan: crate::bridge::envelope::PhysicalPlan::Array(ArrayOp::Put { array_id: ArrayId::in_database(tenant_id, DatabaseId::DEFAULT, ARRAY), cells_msgpack: zerompk::to_msgpack_vec(&cells).expect("encode cells"), @@ -510,4 +584,362 @@ mod tests { "the minted redo must be fsync-durable before the write is acknowledged" ); } + + // --- Caller records under the outcome floor --- + + /// A read plan: the funnel admits it without a gate, so these tests reach + /// the dispatch and response paths with the caller's records attached. + fn point_get_plan() -> crate::bridge::envelope::PhysicalPlan { + crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::PointGet { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "users"), + document_id: "u1".into(), + surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: Vec::new(), + rls_filters: Vec::new(), + system_time: nodedb_types::SystemTimeScope::Current, + valid_at_ms: None, + }, + ) + } + + fn minted_record(state: &SharedState) -> (super::MintedRecords, Lsn) { + let minted = super::MintedRecords::open(&state.outcome_floor); + let lsn = minted + .appender(&state.wal, crate::wal::manager::NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"row", + ) + .expect("append"); + (minted, lsn) + } + + fn write_with(minted: super::MintedRecords, lsn: Lsn) -> super::WriteDispatch { + super::WriteDispatch { + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), + plan: point_get_plan(), + trace_id: crate::types::TraceId::ZERO, + event_source: crate::event::EventSource::User, + txn_id: None, + wal_lsn: Some(lsn), + resolved_now_ms: None, + minted: Some(minted), + } + } + + fn replayed(state: &SharedState) -> Vec { + state.wal.sync().expect("sync"); + state + .wal + .replay() + .expect("replay") + .iter() + .map(|record| record.header.lsn) + .collect() + } + + /// Answer one request with `status` and `code`. + async fn respond_once_with( + state: Arc, + mut side: CoreChannelDataSide, + status: Status, + code: Option, + ) { + let deadline = Instant::now() + Duration::from_secs(5); + let mut handled = false; + while !handled && Instant::now() < deadline { + if let Ok(request) = side.request_rx.try_pop() { + side.response_tx + .try_push(BridgeResponse { + inner: crate::bridge::envelope::Response { + request_id: request.inner.request_id, + status, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: code.clone().map(Box::new), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + }, + }) + .expect("fake data-plane response queue has capacity"); + handled = true; + } + state.poll_and_route_responses(); + tokio::task::yield_now().await; + } + assert!(handled, "fake data plane received the dispatched request"); + state.poll_and_route_responses(); + } + + #[tokio::test] + async fn a_refused_dispatch_cancels_the_callers_records() { + let (state, _side, _directory) = fixture(); + let (minted, lsn) = minted_record(&state); + state + .dispatcher + .lock() + .expect("dispatcher") + .begin_data_plane_drain(); + + let result = + super::dispatch_trusted_internal_write_to_data_plane(&state, write_with(minted, lsn)) + .await; + + assert!(result.is_err(), "a draining dispatcher refuses the request"); + assert!(!replayed(&state).contains(&lsn.as_u64())); + assert!(state.outcome_floor.floor() >= lsn); + assert_eq!(state.outcome_floor.leaked_windows(), 0); + } + + #[tokio::test] + async fn a_refusal_that_applied_nothing_cancels_the_callers_records() { + let (state, side, _directory) = fixture(); + let (minted, lsn) = minted_record(&state); + let responder = tokio::spawn(respond_once_with( + Arc::clone(&state), + side, + Status::Error, + Some(crate::bridge::envelope::ErrorCode::RejectedConstraint { + constraint: "unique".into(), + detail: "duplicate key".into(), + }), + )); + + let response = + super::dispatch_trusted_internal_write_to_data_plane(&state, write_with(minted, lsn)) + .await + .expect("the refusal is a response"); + responder.await.expect("responder completes"); + + assert_eq!(response.status, Status::Error); + assert!(!replayed(&state).contains(&lsn.as_u64())); + assert!(state.outcome_floor.floor() >= lsn); + } + + #[tokio::test] + async fn an_applied_write_settles_the_callers_records() { + let (state, side, _directory) = fixture(); + let (minted, lsn) = minted_record(&state); + let responder = tokio::spawn(respond_once_with( + Arc::clone(&state), + side, + Status::Ok, + None, + )); + + let response = + super::dispatch_trusted_internal_write_to_data_plane(&state, write_with(minted, lsn)) + .await + .expect("the write applies"); + responder.await.expect("responder completes"); + + assert_eq!(response.status, Status::Ok); + assert!(replayed(&state).contains(&lsn.as_u64())); + assert!(state.outcome_floor.floor() >= lsn); + assert_eq!(state.outcome_floor.leaked_windows(), 0); + } + + /// A point write the admission gate serializes on its key. + fn incr_plan() -> crate::bridge::envelope::PhysicalPlan { + crate::bridge::envelope::PhysicalPlan::Kv(nodedb_physical::physical_plan::KvOp::Incr { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "counters"), + key: b"k1".to_vec(), + delta: 1, + ttl_ms: 0, + surrogate: nodedb_types::Surrogate::new(1), + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, + }) + } + + /// Wait until the floor passes `lsn`, routing responses meanwhile. + async fn floor_passes(state: &SharedState, lsn: Lsn) -> bool { + let deadline = Instant::now() + Duration::from_secs(5); + while state.outcome_floor.floor() < lsn && Instant::now() < deadline { + state.poll_and_route_responses(); + tokio::task::yield_now().await; + } + state.outcome_floor.floor() >= lsn + } + + #[tokio::test] + async fn a_caller_dropped_while_waiting_for_admission_cancels_its_records() { + let (state, _side, _directory) = fixture(); + let (minted, lsn) = minted_record(&state); + let plan = incr_plan(); + let (_, keys) = + crate::control::server::shared::write_admission::lock_keys::plan_lock_keys(&plan) + .expect("a point write has a lock key"); + let key = keys.into_iter().next().expect("one key"); + let held = state.write_order_locks.lock_owned(key).await; + let mut write = write_with(minted, lsn); + write.plan = plan; + + let waited = tokio::time::timeout( + Duration::from_millis(50), + super::dispatch_trusted_internal_write_to_data_plane(&state, write), + ) + .await; + drop(held); + + assert!(waited.is_err(), "the write waits behind the held key"); + assert!( + !replayed(&state).contains(&lsn.as_u64()), + "a marker names it" + ); + assert!(state.outcome_floor.floor() >= lsn); + assert_eq!(state.outcome_floor.leaked_windows(), 0); + assert_eq!(state.outcome_floor.held_windows(), 0); + } + + /// Drop the caller once its write is enqueued, then answer the write. + async fn drop_after_dispatch_then_answer( + status: Status, + code: Option, + ) -> (Arc, Lsn, tempfile::TempDir) { + let (state, side, directory) = fixture(); + let (minted, lsn) = minted_record(&state); + let waited = tokio::time::timeout( + Duration::from_millis(50), + super::dispatch_trusted_internal_write_to_data_plane(&state, write_with(minted, lsn)), + ) + .await; + assert!(waited.is_err(), "no response arrived before the drop"); + assert!( + state.outcome_floor.floor() < lsn, + "the core holds the records" + ); + respond_once_with(Arc::clone(&state), side, status, code).await; + assert!( + floor_passes(&state, lsn).await, + "the final response closed the window" + ); + assert_eq!(state.outcome_floor.leaked_windows(), 0); + (state, lsn, directory) + } + + #[tokio::test] + async fn a_caller_dropped_after_dispatch_settles_its_records_from_the_answer() { + let (state, lsn, _directory) = drop_after_dispatch_then_answer(Status::Ok, None).await; + assert!(replayed(&state).contains(&lsn.as_u64())); + } + + #[tokio::test] + async fn a_caller_dropped_after_dispatch_cancels_its_records_on_a_refusal() { + let (state, lsn, _directory) = drop_after_dispatch_then_answer( + Status::Error, + Some(crate::bridge::envelope::ErrorCode::RejectedConstraint { + constraint: "unique".into(), + detail: "duplicate key".into(), + }), + ) + .await; + assert!(!replayed(&state).contains(&lsn.as_u64())); + } + + /// A record-carrying write never resolves an Exchange, so nothing of it + /// reaches a core before its records close. + #[tokio::test] + async fn a_record_carrying_query_is_refused_before_any_fan_out() { + let (state, mut side, _directory) = fixture(); + let (minted, lsn) = minted_record(&state); + let mut write = write_with(minted, lsn); + write.plan = crate::bridge::envelope::PhysicalPlan::Query( + nodedb_physical::physical_plan::QueryOp::Exchange( + nodedb_physical::physical_plan::ExchangeOp { + child: Box::new(point_get_plan()), + mode: nodedb_physical::physical_plan::ExchangeMode::Gather { + as_aggregate: false, + }, + }, + ), + ); + + let result = super::dispatch_trusted_internal_write_to_data_plane(&state, write).await; + + assert!(result.is_err(), "a query plan cannot carry records"); + assert!( + side.request_rx.try_pop().is_err(), + "no request reached a core" + ); + assert!( + !replayed(&state).contains(&lsn.as_u64()), + "a marker names it" + ); + assert!(state.outcome_floor.floor() >= lsn); + assert_eq!(state.outcome_floor.leaked_windows(), 0); + } + + /// A committed proposal refused for good is refused on every replica, so + /// its abort marker carries the proposal key and a redelivered copy finds + /// the refusal in the ledger rebuilt after a restart. + #[tokio::test] + async fn a_final_refusal_of_a_keyed_proposal_is_its_ledger_outcome() { + const KEY: u64 = 0xC0FF_EE01; + let (state, side, _directory) = fixture(); + let plan = + crate::bridge::envelope::PhysicalPlan::Kv(nodedb_physical::physical_plan::KvOp::Put { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "cache"), + key: b"k1".to_vec(), + value: b"v1".to_vec(), + ttl_ms: 0, + surrogate: nodedb_types::Surrogate::new(1), + returning: None, + rls_filters: Vec::new(), + provenance: None, + }); + let responder = tokio::spawn(respond_once_with( + Arc::clone(&state), + side, + Status::Error, + Some(crate::bridge::envelope::ErrorCode::RejectedConstraint { + constraint: "unique".into(), + detail: "duplicate key".into(), + }), + )); + + let outcome = super::submit_write( + &state, + super::SubmitWrite { + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), + plan, + trace_id: crate::types::TraceId::ZERO, + event_source: crate::event::EventSource::User, + txn_id: None, + user_id: None, + durability: super::WalDurability::AppendHere { + now_override: None, + apply_key: KEY, + commit_hlc: None, + }, + ordering: super::WriteOrdering::AlreadyOrdered, + change_feed: super::ChangeFeedOwner::Unowned, + }, + ) + .await + .expect("the refusal is a response"); + responder.await.expect("responder completes"); + assert_eq!(outcome.response.status, Status::Error); + + state.wal.sync().expect("sync"); + let ledger = crate::control::distributed_applier::ProposalLedger::from_records( + &state.wal.replay().expect("replay"), + 8, + ); + assert!( + ledger.prior(KEY).is_some(), + "the refusal's marker names the proposal" + ); + } } diff --git a/nodedb/src/control/server/dispatch_utils/durability_barrier.rs b/nodedb/src/control/server/dispatch_utils/durability_barrier.rs index 8a8de71b1..fc6d66b3e 100644 --- a/nodedb/src/control/server/dispatch_utils/durability_barrier.rs +++ b/nodedb/src/control/server/dispatch_utils/durability_barrier.rs @@ -26,7 +26,7 @@ use std::sync::atomic::{AtomicU64, Ordering}; use crate::bridge::envelope::PhysicalPlan; use crate::control::server::shared::write_admission::plan_is_write; -use nodedb_physical::physical_plan::GraphOp; +use nodedb_physical::physical_plan::{GraphOp, MetaOp}; /// Count of writes acknowledged with no durable redo record despite belonging /// to an engine whose every write-class op mints one on this path. @@ -56,9 +56,10 @@ pub fn writes_acked_without_durability() -> u64 { /// * `Crdt` — constraint installs are Raft-log-replay durable and /// `RestoreToVersion` only computes a forward delta that a follow-up /// `Apply` logs; -/// * `Meta` — COMMIT's single transaction redo, the procedural batch flush and -/// the Calvin ops each own durability on their own path and arrive with the -/// LSN they minted (or none, by design); +/// * `Meta` — the procedural batch flush and the Calvin ops each own +/// durability on their own path and arrive with the LSN they minted (or +/// none, by design). `ApplyTransactionRedo` is the one `Meta` write this +/// funnel appends a record for, and it is held to the barrier; /// * `ClusterArray` — a coordinator-side routing wrapper: each owning shard's /// apply mints the redo for the cells it actually holds; /// * an empty `EdgePutBatch` / `EdgeDeleteBatch`, which has no edge to make @@ -86,6 +87,8 @@ pub(super) fn funnel_minted_redo_engine(plan: &PhysicalPlan) -> Option<&'static PhysicalPlan::Graph(GraphOp::EdgePutBatch { edges }) if edges.is_empty() => None, PhysicalPlan::Graph(GraphOp::EdgeDeleteBatch { edges }) if edges.is_empty() => None, PhysicalPlan::Graph(_) => Some("graph"), + // The committed-redo apply appends its `TransactionRedo` record here. + PhysicalPlan::Meta(MetaOp::ApplyTransactionRedo { .. }) => Some("transaction"), PhysicalPlan::Document(_) | PhysicalPlan::Crdt(_) | PhysicalPlan::Meta(_) @@ -95,6 +98,24 @@ pub(super) fn funnel_minted_redo_engine(plan: &PhysicalPlan) -> Option<&'static } } +/// Refuse a write whose redo record only the funnel's `AppendHere` route +/// mints, on a dispatch route that appends nothing. +/// +/// The read route and the staged-write route supply no LSN and no minted +/// records. A write that reaches one of them applies with no WAL record, so +/// the refusal fires before the write is enqueued, never after it applied. +pub(super) fn refuse_unlogged_write(plan: &PhysicalPlan) -> crate::Result<()> { + match funnel_minted_redo_engine(plan) { + None => Ok(()), + Some(engine) => Err(crate::Error::Internal { + detail: format!( + "a {engine} write reached a dispatch route that appends no WAL record; an \ + autocommit write must dispatch through the durable write route" + ), + }), + } +} + /// Called at the durable-at-ack barrier when there is no LSN to wait on. /// /// `missing_redo_engine` is [`funnel_minted_redo_engine`]'s verdict for the @@ -128,6 +149,7 @@ mod tests { surrogate: Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }) } @@ -138,6 +160,19 @@ mod tests { assert_eq!(funnel_minted_redo_engine(&kv_put()), Some("kv")); } + /// A committed transaction's redo apply mints its record in the funnel, + /// so an acknowledgement without it must trip the barrier. + #[test] + fn committed_redo_apply_requires_a_funnel_minted_redo() { + let plan = PhysicalPlan::Meta(MetaOp::ApplyTransactionRedo { + redo: vec![1], + collections: vec!["c".into()], + sum_targets: Vec::new(), + origin: nodedb_physical::physical_plan::RedoOrigin::Commit, + }); + assert_eq!(funnel_minted_redo_engine(&plan), Some("transaction")); + } + /// A read carries no durability obligation at all. #[test] fn read_requires_no_redo() { @@ -173,6 +208,24 @@ mod tests { assert_eq!(funnel_minted_redo_engine(&plan), None); } + /// The read route refuses a KV write before it is enqueued, and passes a + /// read and a staged write. + #[test] + fn the_read_route_refuses_a_write_it_cannot_log() { + assert!(refuse_unlogged_write(&kv_put()).is_err()); + let staged = PhysicalPlan::Meta(MetaOp::StageWrite { + plan: Box::new(kv_put()), + }); + assert!(refuse_unlogged_write(&staged).is_ok()); + let read = PhysicalPlan::Kv(KvOp::Get { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), + key: b"k".to_vec(), + rls_filters: Vec::new(), + surrogate_ceiling: None, + }); + assert!(refuse_unlogged_write(&read).is_ok()); + } + /// A zero-edge batch appends nothing on purpose, so it must not be held to /// an invariant its non-empty sibling satisfies. #[test] diff --git a/nodedb/src/control/server/dispatch_utils/durable_write.rs b/nodedb/src/control/server/dispatch_utils/durable_write.rs new file mode 100644 index 000000000..4b4390d0a --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/durable_write.rs @@ -0,0 +1,247 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The durable route for an autocommit write. +//! +//! A planned autocommit write reaches its engine one of two ways: +//! +//! - Cluster mode, replicable write: the write is proposed through Raft. Every +//! replica applies the committed entry through the funnel, which appends the +//! redo record. The proposing node publishes the change event. +//! - Otherwise: the write enters the funnel with `WalDurability::AppendHere`. +//! The funnel appends the redo record under the write-admission guard, inside +//! the write's outcome-floor window. +//! +//! A write dispatched any other way applies with no WAL record. A crash loses +//! it, and no replica sees it. Every caller that holds an autocommit write +//! dispatches it through this module. + +use std::sync::atomic::Ordering; + +use crate::bridge::envelope::{PhysicalPlan, Response, Status}; +use crate::control::server::shared::clone_write::CloneCheckedTask; +use crate::control::server::shared::write_admission::plan_is_write; +use crate::control::state::SharedState; +use crate::control::wal_replication::{ + ReplicableWrite, propose_replicated_entry, to_replicated_entry, +}; +use crate::types::{DatabaseId, Lsn, RequestId, TenantId, TraceId, VShardId}; + +use super::change_events::{extract_write_change_set, publish_change_set_with_lsn}; +use super::dispatch::{ + dispatch_authorized_autocommit_write, dispatch_authorized_to_data_plane, + dispatch_autocommit_write, +}; +use super::types::AutocommitWrite; + +/// Dispatch a trusted autocommit write on the durable route. +/// +/// A write that carries a transaction id applies at the statement inside an +/// open block: a write the transaction cannot buffer. It is never proposed +/// on its own, so it takes the local `AppendHere` route. +pub(crate) async fn dispatch_durable_autocommit_write( + shared: &SharedState, + write: AutocommitWrite, +) -> crate::Result { + if write.txn_id.is_none() + && let Some(response) = propose_if_replicable( + shared, + WriteTarget { + tenant_id: write.tenant_id, + database_id: write.database_id, + vshard_id: write.vshard_id, + }, + &write.plan, + write.event_source, + ) + .await? + { + return Ok(response); + } + dispatch_autocommit_write(shared, write).await +} + +/// Dispatch an authorized autocommit write on the durable route. +/// +/// Same routing as [`dispatch_durable_autocommit_write`], for a task that +/// passed the clone-write gate and authorization. +/// +/// In cluster mode a write whose RLS write policy is decided per row cannot +/// be proposed bare: a follower has no writing identity to decide the policy +/// against. It resolves to a concrete row set first, on this node, and the +/// resolved write is proposed (`control::write_resolve`), as the planned +/// pgwire and native writes do. +pub(crate) async fn dispatch_authorized_durable_write( + shared: &SharedState, + checked: CloneCheckedTask, + trace_id: TraceId, +) -> crate::Result { + if checked.txn_id().is_none() + && shared.async_raft_proposer().is_some() + && let Some(resolver) = crate::control::write_resolve::resolver_for_plan(checked.plan()) + { + return crate::control::write_resolve::run_authorized_write_resolve( + shared, + checked.into_authorized(), + resolver, + ) + .await; + } + if checked.txn_id().is_none() + && let Some(response) = propose_if_replicable( + shared, + WriteTarget { + tenant_id: checked.tenant_id(), + database_id: checked.database_id(), + vshard_id: checked.vshard_id(), + }, + checked.plan(), + crate::event::EventSource::User, + ) + .await? + { + return Ok(response); + } + dispatch_authorized_autocommit_write(shared, checked, trace_id).await +} + +/// Dispatch an authorized autocommit write on the durable route, tagged with +/// `event_source`. +/// +/// Same routing as [`dispatch_authorized_durable_write`], for a write that +/// runs under a source other than a client's, such as a Lite sync push. Every +/// replica stamps the source on the write's events, so a synced write does +/// not re-fire AFTER triggers. +/// +/// A clustered write whose RLS write policy must resolve against current rows +/// before it is proposed is refused: the resolved write carries none of the +/// plan's sync provenance, so the idempotency gate could not run. +pub(crate) async fn dispatch_authorized_durable_write_with_source( + shared: &SharedState, + checked: CloneCheckedTask, + trace_id: TraceId, + event_source: crate::event::EventSource, +) -> crate::Result { + if checked.txn_id().is_none() + && shared.async_raft_proposer().is_some() + && crate::control::write_resolve::resolver_for_plan(checked.plan()).is_some() + { + return Err(crate::Error::PlanError { + detail: format!( + "a {event_source:?} write to '{}' needs its row-level-security policy resolved \ + before it is proposed, and the resolved write cannot carry its sync provenance", + checked.plan().collection().unwrap_or("") + ), + }); + } + if checked.txn_id().is_none() + && let Some(response) = propose_if_replicable( + shared, + WriteTarget { + tenant_id: checked.tenant_id(), + database_id: checked.database_id(), + vshard_id: checked.vshard_id(), + }, + checked.plan(), + event_source, + ) + .await? + { + return Ok(response); + } + super::dispatch::dispatch_authorized_autocommit_write_with_source( + shared, + checked, + trace_id, + event_source, + ) + .await +} + +/// Dispatch one authorized task by its class: a write on the durable route, +/// anything else on the read route. +/// +/// For a transport fallback that dispatches reads and writes through one call +/// site with no gateway installed. +pub(crate) async fn dispatch_authorized_task_by_class( + shared: &SharedState, + checked: CloneCheckedTask, + trace_id: TraceId, +) -> crate::Result { + if plan_is_write(checked.plan()) { + dispatch_authorized_durable_write(shared, checked, trace_id).await + } else { + dispatch_authorized_to_data_plane(shared, checked, trace_id).await + } +} + +/// The coordinates a write is proposed under. +struct WriteTarget { + tenant_id: TenantId, + database_id: DatabaseId, + vshard_id: VShardId, +} + +/// Propose `plan` through Raft when this node runs a proposer and the plan +/// encodes to a replicated entry. `None` means the write takes the local route. +/// +/// A Data-Plane verdict comes back as `Err(Error::DataPlane(code))`, the shape +/// the pgwire replicated path returns. +async fn propose_if_replicable( + shared: &SharedState, + target: WriteTarget, + plan: &PhysicalPlan, + event_source: crate::event::EventSource, +) -> crate::Result> { + let Some(proposer) = shared.async_raft_proposer() else { + return Ok(None); + }; + let WriteTarget { + tenant_id, + database_id, + vshard_id, + } = target; + let replicable = ReplicableWrite::decide_for_replication(plan)?; + let Some(entry) = to_replicated_entry(tenant_id, database_id, vshard_id, &replicable)? else { + return Ok(None); + }; + let entry = entry.with_event_source(event_source); + let (payload, write_version) = propose_replicated_entry(shared, proposer, entry).await?; + // Replicas apply with `ChangeFeedOwner::Unowned`. The proposing node + // handled the write once, so it publishes the change event. + publish_change_set_with_lsn( + shared, + tenant_id, + database_id, + extract_write_change_set(plan, tenant_id), + write_version, + ); + Ok(Some(replicated_write_response( + shared, + payload, + write_version, + ))) +} + +/// The response a committed and applied replicated write answers with. +/// +/// `write_version` is the written collection's `coll_write_lsn` after the +/// write. It is the watermark and the read version, so a session can floor a +/// later read at it. +fn replicated_write_response( + shared: &SharedState, + payload: Vec, + write_version: Lsn, +) -> Response { + Response { + request_id: RequestId::new(shared.request_id_counter.fetch_add(1, Ordering::Relaxed)), + status: Status::Ok, + attempt: 1, + partial: false, + payload: payload.into(), + watermark_lsn: write_version, + error_code: None, + read_set_valid: None, + read_version_lsn: write_version, + write_set: Vec::new(), + } +} diff --git a/nodedb/src/control/server/dispatch_utils/minted/mod.rs b/nodedb/src/control/server/dispatch_utils/minted/mod.rs new file mode 100644 index 000000000..847da8f72 --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/minted/mod.rs @@ -0,0 +1,11 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The records a write appends for one Data-Plane dispatch, and how their +//! outcome-floor window closes. + +mod owned; +mod records; +mod resolve; + +pub(crate) use owned::{Collect, OwnedResponse, OwnedWait, await_response_owned}; +pub(crate) use records::{MintedRecords, RecordOwner}; diff --git a/nodedb/src/control/server/dispatch_utils/minted/owned.rs b/nodedb/src/control/server/dispatch_utils/minted/owned.rs new file mode 100644 index 000000000..df10d0e91 --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/minted/owned.rs @@ -0,0 +1,261 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Waiting for a dispatched write's response in a task the caller does not +//! own. +//! +//! Once a write is enqueued, its records close from the core's final +//! response. A caller future dropped mid-wait would drop the records with it +//! and leak their window. The wait and the close run in a spawned task +//! instead. The caller awaits what the task reports, and a dropped caller +//! leaves the task running until the records close. + +use std::sync::Arc; +use std::time::Instant; + +use tokio::sync::oneshot; + +use crate::bridge::envelope::Response; +use crate::control::ResponseReceiver; +use crate::control::local_dispatch::{DispatchCollectError, collect_bounded_response}; +use crate::wal::WalManager; + +use super::records::{MintedRecords, RecordOwner}; +use super::resolve::{resolve_at_final, resolve_on_response}; + +/// How the owned task reads the response it reports. +#[derive(Debug, Clone, Copy)] +pub(crate) enum Collect { + /// Every frame up to the final one, merged, under a byte budget. + Merged { max_result_bytes: usize }, + /// The first frame. A partial first frame leaves the task closing the + /// records from the final one. + First, +} + +/// Where the owned task waits, and how the records close. +pub(crate) struct OwnedWait { + pub wal: Arc, + pub owner: RecordOwner, + /// The key a final refusal's abort marker carries, `0` when this write's + /// refusals are never final. + pub final_refusal_key: u64, + /// The instant the caller stops waiting. + pub deadline: Instant, + pub collect: Collect, +} + +/// What the owned task reports to the caller. +pub(crate) enum OwnedResponse { + /// A response arrived by the deadline. `closed` is the result of closing + /// the records from it: a failed cancel returns its error and holds the + /// window. A partial response reports `Ok` here, and the task closes the + /// records from the final one. + Answered { + response: Response, + closed: crate::Result<()>, + }, + /// The deadline passed first. The task closes the records once the + /// final response arrives. + DeadlineExceeded, + /// The merged response outgrew its byte budget. The task closes the + /// records once the final response arrives. + OverBudget { bytes: usize }, + /// The channel closed without a final response. The records are held. + ChannelClosed, +} + +/// Wait for `rx`'s response in a spawned task that owns `minted`, and +/// return what it reports. +pub(crate) async fn await_response_owned( + wait: OwnedWait, + rx: ResponseReceiver, + minted: MintedRecords, +) -> crate::Result { + let (report_tx, report_rx) = oneshot::channel(); + tokio::spawn(wait_and_close(wait, rx, minted, report_tx)); + report_rx.await.map_err(|_| crate::Error::Internal { + detail: "the task waiting for a dispatched write's response ended without \ + reporting" + .into(), + }) +} + +async fn wait_and_close( + wait: OwnedWait, + mut rx: ResponseReceiver, + minted: MintedRecords, + report: oneshot::Sender, +) { + let OwnedWait { + wal, + owner, + final_refusal_key, + deadline, + collect, + } = wait; + let until = tokio::time::Instant::from_std(deadline); + let collected = match collect { + Collect::Merged { max_result_bytes } => { + tokio::time::timeout_at(until, collect_bounded_response(&mut rx, max_result_bytes)) + .await + } + Collect::First => { + tokio::time::timeout_at(until, async { + rx.recv().await.ok_or(DispatchCollectError::ChannelClosed) + }) + .await + } + }; + match collected { + Ok(Ok(response)) if response.partial => { + let _ = report.send(OwnedResponse::Answered { + response, + closed: Ok(()), + }); + resolve_at_final(&wal, owner, final_refusal_key, rx, minted).await; + } + Ok(Ok(response)) => { + let closed = + resolve_on_response(&wal, owner, final_refusal_key, &response, minted).await; + let _ = report.send(OwnedResponse::Answered { response, closed }); + } + Ok(Err(DispatchCollectError::OverBudget { bytes })) => { + let _ = report.send(OwnedResponse::OverBudget { bytes }); + resolve_at_final(&wal, owner, final_refusal_key, rx, minted).await; + } + Ok(Err(DispatchCollectError::ChannelClosed)) => { + minted.hold(); + let _ = report.send(OwnedResponse::ChannelClosed); + } + Err(_) => { + let _ = report.send(OwnedResponse::DeadlineExceeded); + resolve_at_final(&wal, owner, final_refusal_key, rx, minted).await; + } + } +} + +#[cfg(test)] +mod tests { + use std::time::Duration; + + use super::*; + use crate::bridge::dispatch::OutcomeFloor; + use crate::bridge::envelope::{ErrorCode, Payload, Status}; + use crate::control::RequestTracker; + use crate::types::{DatabaseId, Lsn, RequestId, TenantId, VShardId}; + use crate::wal::manager::NO_APPLY_KEY; + + fn owner() -> RecordOwner { + RecordOwner { + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), + } + } + + fn minted_record(wal: &Arc, floor: &Arc) -> (MintedRecords, Lsn) { + let minted = MintedRecords::open(floor); + let lsn = minted + .appender(wal, NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"x", + ) + .expect("append"); + (minted, lsn) + } + + fn refusal(id: u64) -> Response { + Response { + request_id: RequestId::new(id), + status: Status::Error, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: Some(Box::new(ErrorCode::RejectedConstraint { + constraint: "unique".into(), + detail: "duplicate key".into(), + })), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + fn wait(wal: &Arc, deadline: Instant) -> OwnedWait { + OwnedWait { + wal: Arc::clone(wal), + owner: owner(), + final_refusal_key: 0, + deadline, + collect: Collect::Merged { + max_result_bytes: 1 << 20, + }, + } + } + + /// The caller future is dropped while it waits. The task still closes + /// the records from the refusal that arrives afterwards. + #[tokio::test] + async fn a_dropped_caller_still_closes_its_records() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (minted, lsn) = minted_record(&wal, &floor); + let tracker = RequestTracker::new(); + let rx = tracker.register(RequestId::new(1)); + let deadline = Instant::now() + Duration::from_secs(30); + + let caller = tokio::time::timeout( + Duration::from_millis(10), + await_response_owned(wait(&wal, deadline), rx, minted), + ) + .await; + assert!( + caller.is_err(), + "the caller gave up before the core answered" + ); + assert!(floor.floor() < lsn, "the window holds while the core works"); + + assert!(tracker.complete(refusal(1))); + for _ in 0..200 { + if floor.floor() >= lsn { + break; + } + tokio::time::sleep(Duration::from_millis(5)).await; + } + assert!(floor.floor() >= lsn, "the refusal closed the records"); + assert_eq!(floor.leaked_windows(), 0); + } + + /// The deadline passes first. The caller hears it at once, and the task + /// closes the records from the final response that follows. + #[tokio::test] + async fn a_deadline_reports_at_once_and_the_records_close_later() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (minted, lsn) = minted_record(&wal, &floor); + let tracker = RequestTracker::new(); + let rx = tracker.register(RequestId::new(2)); + + let outcome = await_response_owned(wait(&wal, Instant::now()), rx, minted) + .await + .expect("report"); + assert!(matches!(outcome, OwnedResponse::DeadlineExceeded)); + assert!(floor.floor() < lsn); + + assert!(tracker.complete(refusal(2))); + for _ in 0..200 { + if floor.floor() >= lsn { + break; + } + tokio::time::sleep(Duration::from_millis(5)).await; + } + assert!(floor.floor() >= lsn); + } +} diff --git a/nodedb/src/control/server/dispatch_utils/minted/records.rs b/nodedb/src/control/server/dispatch_utils/minted/records.rs new file mode 100644 index 000000000..d38746087 --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/minted/records.rs @@ -0,0 +1,610 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The records a write appended to the WAL for one Data-Plane dispatch, held +//! under the outcome-floor window opened before the first append. +//! +//! The window holds the outcome floor below every record until the write's +//! outcome is final. Each path that ends the write picks one close: +//! +//! - [`MintedRecords::settle`]: the records applied, or their outcome at the +//! core is final. +//! - [`MintedRecords::cancel`]: nothing applied. A `WriteAborted` marker names +//! each record, and the window settles once the markers are durable. +//! - [`MintedRecords::hold`]: the records have no final outcome in this +//! process. Restart replay must reach them, so the floor stays below them +//! until the process exits. +//! +//! Appends go through [`MintedRecords::appender`], which records the LSN of +//! every record it writes. A plan that appends several records is cancelled +//! whole. +//! +//! Records dropped without a close still close. Records never sent to a core +//! are cancelled in place: a `WriteAborted` marker names each one and the +//! window settles. Records a core can hold have no known outcome, so their +//! window leaks and files its report. A caller dropped at any await before +//! the dispatch therefore leaves nothing open. After the dispatch the +//! records belong to the task that waits for the final response. + +use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::{Arc, Mutex, MutexGuard, OnceLock}; + +use crate::bridge::dispatch::{OutcomeFloor, ResendRefusal, WriteWindow}; +use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::wal_dispatch::{WalAppendOutcome, WalAppendRequest, wal_append}; +use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; +use crate::wal::WalManager; +use crate::wal::manager::{AppendSink, NO_APPLY_KEY, RecordedAppend, WalAppender}; + +/// Where a write's records live. The abort markers that cancel them carry it. +#[derive(Debug, Clone, Copy)] +pub(crate) struct RecordOwner { + pub tenant_id: TenantId, + pub database_id: DatabaseId, + pub vshard_id: VShardId, +} + +/// The records one write appended, and the window that holds the outcome +/// floor below them. +#[must_use = "minted records hold the outcome floor until they settle, cancel, or hold"] +pub(crate) struct MintedRecords { + /// `None` once a close took it. + window: Option, + appended: Mutex>, + /// The existing record this set resends. A resent record belongs to the + /// write that appended it, and only that write can cancel it. + resent: Option, + /// The WAL the records went to. The first append stores it. + wal: OnceLock>, + /// Whether a core can hold the records. + sent: AtomicBool, +} + +impl std::fmt::Debug for MintedRecords { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("MintedRecords") + .field("open", &self.window.is_some()) + .field("appended", &*self.recorded()) + .field("resent", &self.resent) + .field("sent", &self.sent.load(Ordering::Acquire)) + .finish() + } +} + +/// A closed set's parts: the window, every appended record, and the +/// resent LSN. +type Parts = (WriteWindow, Vec, Option); + +impl MintedRecords { + /// Open the window. Call it before the first record is appended. + pub(crate) fn open(floor: &Arc) -> Self { + Self::with_window(floor.open_write(), None) + } + + /// Hold an existing record at `lsn` that is sent to a core again. + /// Refused, with the reason, when its outcome is final or a live or held + /// window carries it to one: a second apply would land below the floor, + /// or apply the record twice. + pub(crate) fn resend(floor: &Arc, lsn: Lsn) -> Result { + let window = floor.open_existing(lsn)?; + Ok(Self::with_window(window, Some(lsn))) + } + + fn with_window(window: WriteWindow, resent: Option) -> Self { + Self { + window: Some(window), + appended: Mutex::new(Vec::new()), + resent, + wal: OnceLock::new(), + sent: AtomicBool::new(false), + } + } + + fn recorded(&self) -> MutexGuard<'_, Vec> { + self.appended.lock().unwrap_or_else(|p| p.into_inner()) + } + + /// An appender whose records carry `apply_key` and join this set. + pub(crate) fn appender<'a>( + &'a self, + wal: &'a Arc, + apply_key: u64, + ) -> WalAppender<'a> { + self.wal.get_or_init(|| Arc::clone(wal)); + wal.recording_appender(apply_key, self) + } + + /// Append `plan`'s redo records under this window. Row-write records + /// carry `event_source`, the source the write is dispatched with. + pub(crate) fn append_plan( + &self, + wal: &Arc, + owner: RecordOwner, + plan: &PhysicalPlan, + event_source: crate::event::EventSource, + ) -> crate::Result { + wal_append(WalAppendRequest { + wal: self.appender(wal, NO_APPLY_KEY), + event_source, + tenant_id: owner.tenant_id, + vshard_id: owner.vshard_id, + database_id: owner.database_id, + plan, + credentials: None, + now_override: None, + }) + } + + /// Mark the records as held by a core. Call it once the request carrying + /// them is enqueued, or once they are committed to a path that carries + /// them to their outcome. Dropped records are never cancelled after it. + pub(crate) fn mark_sent(&self) { + self.sent.store(true, Ordering::Release); + } + + /// The highest appended or resent LSN, or `None` when there is none. + pub(crate) fn highest(&self) -> Option { + self.recorded() + .iter() + .map(|record| record.lsn) + .chain(self.resent) + .max() + } + + /// Every appended LSN, in append order. + #[cfg(test)] + pub(crate) fn lsns(&self) -> Vec { + self.recorded().iter().map(|record| record.lsn).collect() + } + + /// Take the window and the records out. `None` when a close already + /// took them. + fn take_parts(&mut self) -> Option { + let window = self.window.take()?; + let appended = std::mem::take(&mut *self.recorded()); + Some((window, appended, self.resent)) + } + + /// The outcome of every record is final. + pub(crate) fn settle(mut self) { + if let Some((window, _, _)) = self.take_parts() { + window.settle(); + } + } + + /// The records have no final outcome in this process. + #[track_caller] + pub(crate) fn hold(mut self) { + if let Some((window, _, _)) = self.take_parts() { + window.hold(); + } + } + + /// Cancel every record with a `WriteAborted` marker that carries + /// `marker_key`, wait until the markers are durable, then settle. + /// + /// A failed append or fsync holds the window and returns the error: the + /// records stay replayable, so the floor must not pass them. + /// + /// A crash before the markers are durable still leaves the records + /// replayable. The markers make the refusal durable once it is reported. + /// + /// A resent record is not cancelled here: the write that appended it + /// cancels it from its own outcome. The window settles. + /// + /// The cancel runs in a task the caller does not own, so a caller + /// dropped mid-cancel leaves the task to close the window. + pub(crate) async fn cancel( + self, + wal: &Arc, + owner: RecordOwner, + marker_key: u64, + ) -> crate::Result<()> { + let wal = Arc::clone(wal); + tokio::spawn(async move { self.cancel_in_place(&wal, owner, marker_key).await }) + .await + .map_err(|error| crate::Error::Internal { + detail: format!("the task cancelling a write's records failed: {error}"), + })? + } + + async fn cancel_in_place( + mut self, + wal: &WalManager, + owner: RecordOwner, + marker_key: u64, + ) -> crate::Result<()> { + let Some((window, appended, resent)) = self.take_parts() else { + return Ok(()); + }; + if resent.is_some() { + window.settle(); + return Ok(()); + } + let mut last_marker = None; + for record in &appended { + match wal.appender(marker_key).append_write_aborted( + owner.tenant_id, + owner.vshard_id, + owner.database_id, + record.lsn, + ) { + Ok(marker) => last_marker = Some(marker), + Err(error) => { + window.hold(); + return Err(error); + } + } + } + if let Some(marker) = last_marker + && let Err(error) = wal.wait_durable(marker).await + { + window.hold(); + return Err(error); + } + tracing::debug!( + cancelled = appended.len(), + "refused write records cancelled in the WAL" + ); + window.settle(); + Ok(()) + } + + /// Cancel records whose write another path carries to its outcome, such + /// as a Calvin route or a Raft proposal, in a task the caller does not + /// own. That path's result stands. A cancel error holds the window, + /// which files its report, and is logged with `path` naming the route. + /// + /// The caller awaits [`Superseded::finish`] once the other path returns. + /// A caller dropped before then leaves the task to finish the cancel. + pub(crate) fn supersede( + self, + wal: Arc, + owner: RecordOwner, + path: &'static str, + ) -> Superseded { + Superseded { + task: tokio::spawn(async move { + if let Err(error) = self.cancel_in_place(&wal, owner, 0).await { + tracing::error!( + path, + %error, + "records of a write carried by another path could not be \ + cancelled; their window is held until restart" + ); + } + }), + } + } + + /// Close records dropped without a close. Runs in the dropping thread and + /// spawns nothing. + /// + /// Records no core holds are cancelled in place, and the window settles + /// once each marker is appended. The markers are not awaited: nothing + /// reported an outcome for these records, so a crash that loses a marker + /// leaves a write whose caller never learned its outcome. A marker that + /// fails to append holds the window. + /// + /// A resent record, or a set with nothing appended, settles. Records a + /// core can hold leak their window, which files its report. + fn close_dropped(&mut self) { + let sent = self.sent.load(Ordering::Acquire); + let Some((window, appended, resent)) = self.take_parts() else { + return; + }; + if sent { + drop(window); + return; + } + if resent.is_some() || appended.is_empty() { + window.settle(); + return; + } + let Some(wal) = self.wal.get() else { + // Only `appender` adds records, and it stores the WAL first. + window.hold(); + return; + }; + for record in &appended { + if let Err(error) = wal.appender(NO_APPLY_KEY).append_write_aborted( + record.tenant_id, + record.vshard_id, + record.database_id, + record.lsn, + ) { + tracing::error!( + %error, + lsn = record.lsn.as_u64(), + "records dropped before dispatch could not be cancelled; \ + their window is held until restart" + ); + window.hold(); + return; + } + } + tracing::debug!( + cancelled = appended.len(), + "records dropped before dispatch cancelled in the WAL" + ); + window.settle(); + } +} + +/// Each append joins the set, and the window owns its LSN from the moment +/// the record exists. A resend of the record is refused while it is owned. +impl AppendSink for MintedRecords { + fn record(&self, append: RecordedAppend) { + if let Some(window) = &self.window { + window.own(append.lsn); + } + self.recorded().push(append); + } +} + +impl Drop for MintedRecords { + fn drop(&mut self) { + self.close_dropped(); + } +} + +/// The task cancelling records another path superseded. +pub(crate) struct Superseded { + task: tokio::task::JoinHandle<()>, +} + +impl Superseded { + /// Wait until the cancel ended. + pub(crate) async fn finish(self) { + if let Err(error) = self.task.await { + tracing::error!(%error, "the task cancelling superseded records failed"); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::wal::manager::NO_APPLY_KEY; + + fn owner() -> RecordOwner { + RecordOwner { + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), + } + } + + fn append(wal: &Arc, minted: &MintedRecords, body: &[u8]) -> Lsn { + minted + .appender(wal, NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + body, + ) + .expect("append") + } + + fn open_wal(dir: &tempfile::TempDir) -> Arc { + Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")) + } + + fn replayed(wal: &WalManager) -> Vec { + wal.sync().expect("sync"); + wal.replay() + .expect("replay") + .iter() + .map(|record| record.header.lsn) + .collect() + } + + #[test] + fn settled_records_release_the_floor() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let floor = OutcomeFloor::new(); + let minted = MintedRecords::open(&floor); + let lsn = append(&wal, &minted, b"a"); + assert!(floor.floor() < lsn); + minted.settle(); + assert_eq!(floor.floor(), lsn); + } + + #[test] + fn held_records_keep_the_floor_below_them() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let floor = OutcomeFloor::new(); + let minted = MintedRecords::open(&floor); + let lsn = append(&wal, &minted, b"a"); + minted.hold(); + assert!(floor.floor() < lsn); + assert_eq!(floor.leaked_windows(), 0, "a hold is not a leak"); + } + + #[tokio::test] + async fn a_resent_record_is_never_cancelled() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let lsn = wal + .appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"a", + ) + .expect("append"); + let resent = MintedRecords::resend(&floor, lsn).expect("the floor is below the record"); + assert!(floor.floor() < lsn); + resent.cancel(&wal, owner(), 0).await.expect("cancel"); + wal.sync().expect("sync"); + let replayed: Vec = wal + .replay() + .expect("replay") + .iter() + .map(|record| record.header.lsn) + .collect(); + assert!(replayed.contains(&lsn.as_u64()), "no marker names it"); + assert_eq!(floor.floor(), lsn); + assert!( + MintedRecords::resend(&floor, lsn).is_err(), + "the floor passed the record" + ); + } + + #[tokio::test] + async fn cancelled_records_are_dropped_from_replay_and_release_the_floor() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let minted = MintedRecords::open(&floor); + let first = append(&wal, &minted, b"a"); + let second = append(&wal, &minted, b"b"); + assert_eq!(minted.lsns(), vec![first, second]); + assert_eq!(minted.highest(), Some(second)); + minted.cancel(&wal, owner(), 0).await.expect("cancel"); + assert!( + wal.durable_through() > second.as_u64(), + "markers are durable" + ); + let replayed: Vec = wal + .replay() + .expect("replay") + .iter() + .map(|record| record.header.lsn) + .collect(); + assert!(!replayed.contains(&first.as_u64())); + assert!(!replayed.contains(&second.as_u64())); + assert!(floor.floor() >= second, "the window settled"); + } + + #[test] + fn records_dropped_before_dispatch_are_cancelled_and_release_the_floor() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let floor = OutcomeFloor::new(); + let minted = MintedRecords::open(&floor); + let first = append(&wal, &minted, b"a"); + let second = append(&wal, &minted, b"b"); + drop(minted); + let replayed = replayed(&wal); + assert!(!replayed.contains(&first.as_u64())); + assert!(!replayed.contains(&second.as_u64())); + assert!(floor.floor() >= second, "the window settled"); + assert_eq!(floor.leaked_windows(), 0); + assert_eq!(floor.held_windows(), 0); + } + + #[test] + fn records_dropped_after_dispatch_keep_the_floor_below_them() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let floor = OutcomeFloor::new(); + let minted = MintedRecords::open(&floor); + let lsn = append(&wal, &minted, b"a"); + minted.mark_sent(); + drop(minted); + assert!(replayed(&wal).contains(&lsn.as_u64()), "no marker names it"); + assert!(floor.floor() < lsn); + assert_eq!( + floor.leaked_windows(), + 1, + "the dropped window files its leak" + ); + } + + #[test] + fn a_dropped_set_with_nothing_appended_settles() { + let floor = OutcomeFloor::new(); + drop(MintedRecords::open(&floor)); + assert_eq!(floor.leaked_windows(), 0); + assert_eq!(floor.held_windows(), 0); + } + + #[test] + fn a_dropped_resend_settles_without_a_marker() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let floor = OutcomeFloor::new(); + let lsn = wal + .appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"a", + ) + .expect("append"); + drop(MintedRecords::resend(&floor, lsn).expect("the floor is below the record")); + assert!(replayed(&wal).contains(&lsn.as_u64()), "no marker names it"); + assert_eq!(floor.floor(), lsn); + assert_eq!(floor.leaked_windows(), 0); + } + + #[cfg(feature = "failpoints")] + #[test] + fn a_failed_drop_cancel_holds_the_window() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let floor = OutcomeFloor::new(); + let minted = MintedRecords::open(&floor); + let lsn = append(&wal, &minted, b"a"); + { + let _fail = + crate::fail_point::FailGuard::fail("wal::append_write_aborted", "disk full"); + drop(minted); + } + assert!(replayed(&wal).contains(&lsn.as_u64()), "no marker names it"); + assert!(floor.floor() < lsn); + assert_eq!(floor.held_windows(), 1); + assert_eq!(floor.leaked_windows(), 0); + } + + /// A writer that appended a record and has not sent it yet owns it, so a + /// resend cannot race it to a core. + #[test] + fn a_resend_is_refused_while_a_live_window_owns_the_record() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let floor = OutcomeFloor::new(); + let minted = MintedRecords::open(&floor); + let lsn = append(&wal, &minted, b"a"); + + assert!(floor.floor() < lsn); + assert!( + MintedRecords::resend(&floor, lsn).is_err(), + "the writer owns it" + ); + + minted.settle(); + } + + /// A record whose owner closed has a final outcome, even while an older + /// window keeps the floor below it. + #[test] + fn a_resend_is_refused_after_the_owner_closed_above_the_floor() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let floor = OutcomeFloor::new(); + let older = MintedRecords::open(&floor); + append(&wal, &older, b"a"); + let newer = MintedRecords::open(&floor); + let lsn = append(&wal, &newer, b"b"); + newer.settle(); + + assert!(floor.floor() < lsn, "the older window holds the floor"); + assert!( + MintedRecords::resend(&floor, lsn).is_err(), + "its outcome is final" + ); + + older.settle(); + assert!(floor.floor() >= lsn); + } +} diff --git a/nodedb/src/control/server/dispatch_utils/minted/resolve.rs b/nodedb/src/control/server/dispatch_utils/minted/resolve.rs new file mode 100644 index 000000000..7bbc7104e --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/minted/resolve.rs @@ -0,0 +1,306 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Close a write's outcome-floor window from the core's final response. +//! +//! A refusal whose code proves nothing applied cancels the records. Any other +//! final response settles them: the core's outcome is final, and restart +//! replay from the floor reproduces it. + +use std::sync::Arc; + +use crate::bridge::envelope::{ErrorCode, Response, Status}; +use crate::wal::WalManager; + +use super::super::write_abort::{refusal_is_final, write_definitely_not_applied}; +use super::records::{MintedRecords, RecordOwner}; + +/// Close `minted` from the core's final `response`. +/// +/// `final_refusal_key` is the proposal key a final refusal's marker carries, +/// `0` when no proposal carries this write. A failed cancel returns the error +/// and holds the window. +pub(crate) async fn resolve_on_response( + wal: &Arc, + owner: RecordOwner, + final_refusal_key: u64, + response: &Response, + minted: MintedRecords, +) -> crate::Result<()> { + let refusal = response + .error_code + .as_deref() + .filter(|code| response.status != Status::Ok && write_definitely_not_applied(code)); + match refusal { + Some(code) => { + let marker_key = if refusal_is_final(code) { + final_refusal_key + } else { + 0 + }; + // A refused sync frame advanced its stream's high-water mark. The + // frame's record is cancelled, so the mark gets a record of its + // own, durable with the markers. + if let ErrorCode::SyncRejected { provenance, .. } = code + && let Err(error) = wal.appender(marker_key).append_sync_seq_advance( + provenance.producer_id, + provenance.epoch, + provenance.stream_id, + provenance.seq, + ) + { + minted.hold(); + return Err(error); + } + minted.cancel(wal, owner, marker_key).await + } + None => { + minted.settle(); + Ok(()) + } + } +} + +/// Close `minted` once the core's final response arrives on `rx`, after the +/// caller stopped waiting for it. +/// +/// A refusal that arrives after the caller timed out still cancels its +/// records, so a restart cannot apply a write the caller never saw applied. A +/// channel that closes before a final response leaves the outcome unknown: +/// the window holds. +pub(crate) async fn resolve_at_final( + wal: &Arc, + owner: RecordOwner, + final_refusal_key: u64, + mut rx: crate::control::ResponseReceiver, + minted: MintedRecords, +) { + loop { + match rx.recv().await { + Some(response) if response.partial => continue, + Some(response) => { + if let Err(error) = + resolve_on_response(wal, owner, final_refusal_key, &response, minted).await + { + tracing::error!( + %error, + "a late refusal's abort marker failed; its records stay replayable" + ); + } + return; + } + None => { + minted.hold(); + return; + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::dispatch::OutcomeFloor; + use crate::bridge::envelope::{ErrorCode, Payload}; + use crate::types::{DatabaseId, Lsn, RequestId, TenantId, VShardId}; + use crate::wal::manager::NO_APPLY_KEY; + + fn owner() -> RecordOwner { + RecordOwner { + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), + } + } + + fn response(status: Status, code: Option, partial: bool) -> Response { + Response { + request_id: RequestId::new(1), + status, + attempt: 1, + partial, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: code.map(Box::new), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + fn refusal() -> Response { + response( + Status::Error, + Some(ErrorCode::RejectedConstraint { + constraint: "unique".into(), + detail: "duplicate key".into(), + }), + false, + ) + } + + fn minted_record(wal: &Arc, floor: &Arc) -> (MintedRecords, Lsn) { + let minted = MintedRecords::open(floor); + let lsn = minted + .appender(wal, NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"row", + ) + .expect("append"); + (minted, lsn) + } + + fn replayed(wal: &WalManager) -> Vec { + wal.replay() + .expect("replay") + .iter() + .map(|record| record.header.lsn) + .collect() + } + + #[tokio::test] + async fn an_applied_response_settles_and_keeps_the_record() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (minted, lsn) = minted_record(&wal, &floor); + let ok = response(Status::Ok, None, false); + resolve_on_response(&wal, owner(), 0, &ok, minted) + .await + .expect("resolve"); + wal.sync().expect("sync"); + assert!(replayed(&wal).contains(&lsn.as_u64())); + assert_eq!(floor.floor(), lsn); + } + + #[tokio::test] + async fn an_ambiguous_failure_settles_and_keeps_the_record() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (minted, lsn) = minted_record(&wal, &floor); + let internal = response( + Status::Error, + Some(ErrorCode::Internal { + detail: "io".into(), + }), + false, + ); + resolve_on_response(&wal, owner(), 0, &internal, minted) + .await + .expect("resolve"); + wal.sync().expect("sync"); + assert!(replayed(&wal).contains(&lsn.as_u64())); + assert_eq!(floor.floor(), lsn); + } + + #[tokio::test] + async fn a_definite_refusal_cancels_the_record() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (minted, lsn) = minted_record(&wal, &floor); + resolve_on_response(&wal, owner(), 0, &refusal(), minted) + .await + .expect("resolve"); + assert!(!replayed(&wal).contains(&lsn.as_u64())); + assert!(floor.floor() >= lsn); + } + + /// The caller stopped waiting before the core answered. The refusal that + /// arrives later still writes its abort marker before the window settles. + #[tokio::test] + async fn a_refusal_after_the_caller_timed_out_still_writes_its_abort_marker() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (minted, lsn) = minted_record(&wal, &floor); + let (tx, rx) = tokio::sync::mpsc::channel(4); + let rx = crate::control::ResponseReceiver::from_channel(rx); + let waiter = { + let wal = Arc::clone(&wal); + tokio::spawn(async move { resolve_at_final(&wal, owner(), 0, rx, minted).await }) + }; + assert!(floor.floor() < lsn, "the window holds while the core works"); + + tx.send(response(Status::Ok, None, true)) + .await + .expect("send partial"); + tx.send(refusal()).await.expect("send refusal"); + waiter.await.expect("waiter"); + + assert!(!replayed(&wal).contains(&lsn.as_u64())); + assert!(floor.floor() >= lsn); + } + + #[tokio::test] + async fn a_channel_closed_before_a_final_response_holds_the_window() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (minted, lsn) = minted_record(&wal, &floor); + let (tx, rx) = tokio::sync::mpsc::channel::(4); + let rx = crate::control::ResponseReceiver::from_channel(rx); + let waiter = { + let wal = Arc::clone(&wal); + tokio::spawn(async move { resolve_at_final(&wal, owner(), 0, rx, minted).await }) + }; + drop(tx); + waiter.await.expect("waiter"); + assert!(floor.floor() < lsn); + assert_eq!(floor.leaked_windows(), 0); + } + + /// A sync frame the gate refused for good is cancelled, and the + /// high-water mark it advanced is journalled on its own. Restart replay + /// then restores the mark, never applies the frame, and counts the + /// refusal as the proposal's outcome. + #[tokio::test] + async fn a_refused_sync_frame_journals_its_mark_and_cancels_its_record() { + const KEY: u64 = 0xAB; + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (minted, lsn) = minted_record(&wal, &floor); + let rejected = response( + Status::Error, + Some(ErrorCode::SyncRejected { + violation: nodedb_types::sync::violation::ViolationType::PermissionDenied, + applied_seq: 4, + provenance: nodedb_types::sync::wire::SyncProvenance { + producer_id: 9, + epoch: 2, + stream_id: 1, + seq: 4, + }, + }), + false, + ); + + resolve_on_response(&wal, owner(), KEY, &rejected, minted) + .await + .expect("resolve"); + + wal.sync().expect("sync"); + let records = wal.replay().expect("replay"); + assert!( + !records + .iter() + .any(|record| record.header.lsn == lsn.as_u64()), + "the refused frame never replays" + ); + let (maps, _) = + crate::wal::replay::replay_sync_hwm_records(&records).expect("replay marks"); + assert_eq!(maps.sync_hwm.get(&(9, 1)), Some(&4)); + assert_eq!(maps.producer_epoch_floor.get(&9), Some(&2)); + let ledger = crate::control::distributed_applier::ProposalLedger::from_records(&records, 8); + assert!( + ledger.prior(KEY).is_some(), + "the refusal is the proposal's outcome" + ); + assert!(floor.floor() >= lsn); + } +} diff --git a/nodedb/src/control/server/dispatch_utils/mod.rs b/nodedb/src/control/server/dispatch_utils/mod.rs index 7f72977c5..224638b30 100644 --- a/nodedb/src/control/server/dispatch_utils/mod.rs +++ b/nodedb/src/control/server/dispatch_utils/mod.rs @@ -3,10 +3,10 @@ //! Shared dispatch utilities used by both the pgwire and native endpoints. mod change_events; -mod collect; mod dispatch; mod durability_barrier; -mod error_status; +mod durable_write; +mod minted; mod submit_write; mod types; mod write_abort; @@ -15,18 +15,25 @@ pub(crate) use change_events::{ WriteChangeSet, extract_write_change_set, publish_change_set_with_lsn, publish_cluster_array_change_events, publish_origin_change_events, }; -pub(crate) use collect::{ - DeadlineCollect, DispatchCollectError, collect_bounded_response, collect_under_deadline, -}; pub use dispatch::{dispatch_authorized_autocommit_write, dispatch_authorized_to_data_plane}; pub(crate) use dispatch::{ - dispatch_authorized_autocommit_write_with_source, dispatch_autocommit_write, - dispatch_to_data_plane, dispatch_to_data_plane_with_txn, + dispatch_authorized_autocommit_write_with_source, dispatch_authorized_minted_to_data_plane, + dispatch_autocommit_write, dispatch_to_data_plane, dispatch_to_data_plane_with_txn, dispatch_trusted_internal_write_to_data_plane, }; pub use durability_barrier::writes_acked_without_durability; -pub(crate) use error_status::reject_data_plane_error; +pub(crate) use durable_write::{ + dispatch_authorized_durable_write, dispatch_authorized_durable_write_with_source, + dispatch_authorized_task_by_class, dispatch_durable_autocommit_write, +}; +pub(crate) use minted::{ + Collect, MintedRecords, OwnedResponse, OwnedWait, RecordOwner, await_response_owned, +}; pub(crate) use submit_write::{ - ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, submit_write, + ChangeFeedOwner, PendingWrite, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, + enqueue_write, submit_write, }; pub(crate) use types::{AutocommitWrite, WriteDispatch}; +pub(crate) use write_abort::{ + error_is_final_refusal, refusal_is_final, write_definitely_not_applied, +}; diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel.rs deleted file mode 100644 index a7e9d3796..000000000 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel.rs +++ /dev/null @@ -1,450 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! THE single Control-Plane write funnel. -//! -//! Every Control-Plane path that puts a write on the SPSC bridge routes through -//! [`submit_write`]: the autocommit / internal funnel (`dispatch.rs`), the -//! pgwire local-dispatch path (`pgwire::handler::submit`), and the Raft -//! apply loop (`distributed_applier::apply_loop`). The funnel owns write -//! admission, the WAL redo append, the enqueue, the bounded response collect, -//! the post-apply redo, the durable-at-ack barrier, and — for the caller that -//! owns it (see [`ChangeFeedOwner`]) — the CDC publish, in that order, which is -//! the correctness contract. -//! -//! A path that reimplements these steps drifts silently: it is not a compile -//! error to omit the redo append or the change-event publish, and the omission -//! only surfaces as lost data after a crash, or as a change stream that never -//! fires. Add the step here, once, and every caller gets it. -//! -//! It also owns the mirror of the redo append: when it appended the record -//! itself and the Data Plane then REFUSED the write, it cancels that record -//! before returning the error. See [`super::super::write_abort`] for which verdicts -//! qualify, the residual crash window it does not close, and the latency it -//! costs a rejection. - -use std::time::Instant; - -use crate::bridge::envelope::{Priority, Request, Status}; -use crate::control::server::shared::session::statement_deadline; -use crate::control::server::wal_dispatch::{self, WalAppendRequest}; -use crate::control::state::SharedState; -use crate::types::ReadConsistency; - -use super::super::change_events::{extract_write_change_set, publish_change_set}; -use super::super::collect::{DispatchCollectError, collect_bounded_response}; -use super::ambiguous_ddl::preserve_ambiguous_array_ddl; -use super::params::{ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering}; - -/// Admit, make durable, enqueue, collect, and publish one write. -/// -/// See [`SubmitOutcome`] for what comes back. -pub(crate) async fn submit_write( - shared: &SharedState, - params: SubmitWrite, -) -> crate::Result { - let SubmitWrite { - tenant_id, - database_id, - vshard_id, - mut plan, - trace_id, - event_source, - txn_id, - user_id, - durability, - ordering, - change_feed, - } = params; - - // The running statement's deadline, pinned once at the session boundary and - // shared by every request the statement fans out into. Used for both the - // envelope the Data Plane enforces and the Control-Plane collect below, so - // the two halves cannot disagree about when this statement expires. - let deadline = statement_deadline(shared.tuning.network.default_deadline_secs); - - // Change metadata is derived from the plan HERE, before it is moved into - // the request — the publish itself happens after apply, once the response - // (which carries the event's LSN) exists, by which point the plan is gone. - // Extraction is a pure match that clones out collection / document - // identity, so a caller whose change feed is `Unowned` skips it rather than - // allocating tuples nothing will read. - let change_set = match change_feed { - ChangeFeedOwner::Funnel => Some(extract_write_change_set(&plan, tenant_id)), - ChangeFeedOwner::Unowned => None, - }; - - // Post-apply redo classification, computed before `plan` is moved (the - // RouteToCalvin admit arm moves it). For a write whose autocommit WAL path - // mints no redo of its own but whose effect must survive a WAL-only restart - // (a document PointUpdate on a collection carrying a secondary vector - // index), the durable redo is minted AFTER apply from the surrogate + - // post-image the Data Plane returns in `Response::write_set`. - // `Some(collection)` for such a write, else `None`. - let post_apply = wal_dispatch::plan_post_apply_redo(&plan); - let appends_here = matches!(&durability, WalDurability::AppendHere { .. }); - - // Durable-at-ack obligation, also computed before `plan` moves. `Some` only - // for a write whose redo record THIS funnel is required to mint; a caller - // that appended upstream (or declared durability owned elsewhere) is not - // held to it, because the LSN it does or does not supply is its own - // contract. See `durability_barrier` for why this is narrower than - // "write-class plan with no LSN". - let funnel_redo_engine = if appends_here { - super::super::durability_barrier::funnel_minted_redo_engine(&plan) - } else { - None - }; - - // Write-admission gate: every write-class plan whose ordering is not already - // final passes here. An uncontended point write takes the fast path holding - // its per-vShard deterministic locks; a contended or bulk write is submitted - // through the deterministic scheduler and its applied response is surfaced - // here; reads / control ops are `Exempt`. - // - // Ordering (fast path): the guard is acquired FIRST, then — for a write that - // owns its durability (`AppendHere`) — the WAL append happens below, under - // the guard, minting the LSN just before the enqueue. The guard is released - // immediately after the enqueue (not across the response await). - use crate::control::server::shared::write_admission::{ - WriteAdmission, WriteTarget, admit, bare_ok_response, route_write_to_calvin, - }; - let (admission, admission_guard, order_guard) = match ordering { - WriteOrdering::AlreadyOrdered => ( - crate::bridge::envelope::Admission::Exempt( - crate::bridge::envelope::ExemptReason::AlreadyOrdered, - ), - None, - None, - ), - WriteOrdering::Gate => match admit( - shared, - &WriteTarget { - tenant_id, - database_id, - vshard_id, - plan: &plan, - }, - ) { - WriteAdmission::ExemptRead => ( - crate::bridge::envelope::Admission::Exempt( - crate::bridge::envelope::ExemptReason::Read, - ), - None, - None, - ), - WriteAdmission::FastPath { guard } => { - (crate::bridge::envelope::Admission::Admitted, guard, None) - } - WriteAdmission::FastPathBlocking { key, keyed_lock } => { - // Single-node serialization point: acquire the per-key FIFO - // order-lock FIRST, before the WAL append and enqueue below. - // `tokio::sync::Mutex` is fair, so concurrent same-key writers - // are admitted in arrival order — the WAL append + enqueue then - // happen in that order, giving WAL-LSN order == enqueue order == - // apply order per key. Distinct keys use distinct per-key mutexes - // and never contend. - let order_guard = keyed_lock.lock_owned(key).await; - ( - crate::bridge::envelope::Admission::Admitted, - None, - Some(order_guard), - ) - } - WriteAdmission::RouteToCalvin => { - // The deterministic scheduler applies the write (emitting its own - // WriteEvents) and returns the applied response; a plain write with - // no RETURNING rows yields `None`, synthesized into a bare `Ok`. - // Calvin owns durability on this route (the sequenced TxClass plus - // its own `CalvinApplied` WAL record), so no local append happens. - let routed = - route_write_to_calvin(shared, tenant_id, database_id, vshard_id, plan).await?; - return Ok(SubmitOutcome { - response: routed - .unwrap_or_else(|| bare_ok_response(crate::types::RequestId::new(0))), - wal_lsn: None, - }); - } - }, - }; - - // Array DDL conversion is intentionally read-only. Once this task has - // passed authorization and admission, install its durable catalog state - // immediately before the Data-Plane dispatch. The mirror is changed only - // after the redb transaction commits. - let ddl_transition = crate::control::array_catalog::ddl::apply_authorized_ddl( - shared, - tenant_id, - database_id, - &plan, - )?; - macro_rules! rollback_ddl { - ($result:expr) => { - match $result { - Ok(value) => value, - Err(error) => { - let _ = ddl_transition.rollback(shared); - return Err(error.into()); - } - } - }; - } - - // Durability, under the guard, immediately before the enqueue: the LSN is - // minted in the same order the request is about to be enqueued. - let (wal_lsn, resolved_now_ms) = match durability { - WalDurability::AppendHere { now_override } => { - let outcome = rollback_ddl!(wal_dispatch::wal_append(WalAppendRequest { - wal: &shared.wal, - tenant_id, - vshard_id, - database_id, - plan: &plan, - credentials: None, - now_override, - })); - (outcome.lsn, outcome.resolved_now_ms) - } - WalDurability::CallerSupplied { - wal_lsn, - resolved_now_ms, - } => (wal_lsn, resolved_now_ms), - }; - - // Write the resolved LSN back into the plan itself. The envelope's - // `wal_lsn` is where most engines read their committed version from, but the - // array engine stamps its tile versions from the LSN carried in the plan - // while replay stamps them from the record header — so the plan the Data - // Plane is about to execute must name the record that reproduces it. This is - // the only place that knows both, and it knows them for every caller: no - // upstream path may allocate an LSN of its own and hope it matches. - if let Some(lsn) = wal_lsn { - wal_dispatch::stamp_minted_lsn(&mut plan, lsn); - } - - // Per-vShard QPS + latency timer. `dispatch_started` marks the wall-clock - // moment the request enters the Control Plane dispatch site; observation - // happens on every exit path (success, budget over-run, timeout) so the - // histogram captures the true end-to-end shape of the work routed to this - // vshard. - let dispatch_started = Instant::now(); - let vshard_u32 = vshard_id.as_u32(); - let observe = |shared: &SharedState| { - let latency_us = dispatch_started.elapsed().as_micros().min(u64::MAX as u128) as u64; - shared.per_vshard_metrics.observe(vshard_u32, latency_us); - }; - - let request_id = shared.next_request_id(); - let request = Request { - request_id, - tenant_id, - database_id, - vshard_id, - plan, - deadline, - priority: Priority::Normal, - trace_id, - consistency: ReadConsistency::Strong, - idempotency_key: None, - event_source, - user_roles: Vec::new(), - user_id, - statement_digest: None, - txn_id, - wal_lsn, - resolved_now_ms, - admission, - }; - - let mut rx = shared.tracker.register(request_id); - - match shared.dispatcher.lock() { - Ok(mut d) => rollback_ddl!(d.dispatch(request)), - Err(poisoned) => rollback_ddl!(poisoned.into_inner().dispatch(request)), - }; - - // Release the write-admission guards immediately after the enqueue, before - // the Data-Plane round-trip. The per-database WFQ is strict FIFO, so once LSN - // order equals enqueue order the apply order follows from the queue alone; - // holding the guards across the response await would only serialize same-key - // throughput needlessly. - // - // EXCEPTION — a post-apply-redo write (`post_apply.is_some()`) mints its - // durable redo AFTER apply, from the write-set on the response; the guards - // MUST stay held across the response collect + that append so two concurrent - // same-surrogate writes cannot reorder their redo appends. Both guard types - // are `Send`, so holding them across the `.await` is sound. Moved into an - // `Option` so the release is a single, unconditional `drop` below regardless - // of which path took it. (`None` guard slots when no lock manager was - // registered / for the exempt-read / Calvin / already-ordered cases.) - let deferred_guards = if post_apply.is_some() { - Some((admission_guard, order_guard)) - } else { - drop(admission_guard); - drop(order_guard); - None - }; - - // Collect response(s). For non-streaming queries, exactly one arrives. - // For streaming queries, multiple partial chunks arrive before the final. - // The mpsc channel is bounded (see `RequestTracker::register`); here we - // additionally cap the *total* accumulated payload so a runaway scan - // can't pin Control-Plane RAM — any query whose combined result exceeds - // `tuning.network.max_query_result_bytes` is cancelled with a typed - // `ExecutionLimitExceeded` error. - let max_result_bytes = shared.tuning.network.max_query_result_bytes as usize; - // The same instant the envelope carries. The Data Plane normally answers - // with `DeadlineExceeded` first; this bounds the wait when it is inside a - // stage that carries no safe point yet. - let response = match tokio::time::timeout_at( - tokio::time::Instant::from_std(deadline), - collect_bounded_response(&mut rx, max_result_bytes), - ) - .await - { - Ok(response) => response, - Err(_) => { - observe(shared); - // Dispatch completed, but the Data Plane may have applied CREATE - // or ALTER before this deadline. Never roll that catalog state - // back on an ambiguous post-enqueue outcome. - preserve_ambiguous_array_ddl(shared, &ddl_transition); - if !ddl_transition.preserves_on_ambiguous_apply() { - let _ = ddl_transition.rollback(shared); - } - return Err(crate::Error::DeadlineExceeded { request_id }); - } - }; - - let response = match response { - Ok(r) => r, - Err(DispatchCollectError::OverBudget { bytes }) => { - shared.tracker.cancel(&request_id); - observe(shared); - // A partial response proves dispatch began but not whether an - // Array DDL completed; preserve CREATE/ALTER and fail-stop. - preserve_ambiguous_array_ddl(shared, &ddl_transition); - if !ddl_transition.preserves_on_ambiguous_apply() { - let _ = ddl_transition.rollback(shared); - } - return Err(crate::Error::ExecutionLimitExceeded { - detail: format!( - "query result exceeded max_query_result_bytes \ - ({bytes} > {max_result_bytes} bytes)" - ), - }); - } - Err(DispatchCollectError::ChannelClosed) => { - observe(shared); - // The producer can close after applying but before sending its - // response. CREATE/ALTER must remain catalog-finalized here. - preserve_ambiguous_array_ddl(shared, &ddl_transition); - if !ddl_transition.preserves_on_ambiguous_apply() { - let _ = ddl_transition.rollback(shared); - } - // A producer that stopped after the deadline stopped because the - // statement ran out of time. Reporting the closure would hand the - // client an internal error for its own timeout. - if std::time::Instant::now() >= deadline { - return Err(crate::Error::DeadlineExceeded { request_id }); - } - return Err(crate::Error::Dispatch { - detail: "response channel closed".into(), - }); - } - }; - - if response.status != Status::Ok { - let _ = ddl_transition.rollback(shared); - super::super::write_abort::abort_refused_write( - shared, - super::super::write_abort::AbortTarget { - tenant_id, - database_id, - vshard_id, - wal_lsn, - appends_here, - }, - &response, - ) - .await?; - } - - // Mint the post-apply redo record while the guards are still held, then - // release them. A PointUpdate whose collection carries a secondary vector - // index returns its surrogate + post-image in `write_set`; without this - // durable `Put` a WAL-only restart rebuilds the HNSW from the pre-update body - // and resurrects the old embedding. - let post_apply_lsn = if let Some(collection) = &post_apply - && appends_here - && response.status == Status::Ok - { - rollback_ddl!(wal_dispatch::append_write_set_redo( - &shared.wal, - tenant_id, - vshard_id, - database_id, - collection, - &response.write_set, - )) - } else { - None - }; - drop(deferred_guards); - - // Durable-at-ack barrier: an acknowledged write must be WAL-fsync-durable - // before this response (the client ack) returns. `WalManager::append_*` only - // buffers the record and mints its `Lsn`; without this barrier a `kill -9` - // loses the buffered bytes, which is invisible for engines whose rows are - // committed durably by redb but silently destroys every engine whose only - // durability path is WAL replay: the KV hash tables, the HNSW graphs, the - // columnar / timeseries memtables, the graph node labels, the CRDT states, - // and the FTS index. `wal_lsn` is the forward write's LSN — minted above - // under the admission guard for a write that owns its durability, or supplied - // by a caller that appended upstream (procedural batch flush, - // interactive-COMMIT transaction redo). `post_apply_lsn` covers the - // post-apply redo appended just above. Both records are already buffered in - // the shared WAL; one group-commit fsync coalesces concurrent writers (see - // `WalManager::wait_durable`), and it runs here — after the admission guards - // are released — so it never serializes same-key throughput. Reads / control - // ops / trigger / staged-write dispatch carry no LSN and skip the barrier; - // `durability_barrier` decides which of those skips are legitimate and makes - // the rest loud instead of letting them ack a write nothing can recover. - if response.status == Status::Ok { - let durable_target = match (wal_lsn, post_apply_lsn) { - (Some(a), Some(b)) => Some(a.max(b)), - (a, b) => a.or(b), - }; - match durable_target { - Some(lsn) => rollback_ddl!(shared.wal.wait_durable(lsn).await), - // Nothing to fsync. Legitimate for most plans, but if this funnel - // was the one required to mint the record, the ack below promises - // durability the engine cannot deliver — silent until a `kill -9` - // proves it, hence the check. - None => super::super::durability_barrier::assert_durable_before_ack(funnel_redo_engine), - } - } - - if response.status == Status::Ok { - ddl_transition.finalize(shared)?; - } - - // Publish change events for successful writes whose change feed this funnel - // owns. `None` is a caller whose change feed is `Unowned` — see - // [`ChangeFeedOwner`] for why the node that applies those writes is not the - // node that publishes them. - if response.status == Status::Ok { - if let Some(change_set) = change_set { - publish_change_set(shared, tenant_id, database_id, change_set, &response); - } - - // Advance the tenant's observed write-HLC high-water on any successful - // dispatch. Used by the RESTORE staleness gate. Advancing on every - // success (not just writes) is intentionally conservative — - // envelope.watermark is captured AFTER fan-out so it always dominates - // the tenant_wm of a fresh backup. - shared.advance_tenant_write_hlc(tenant_id.as_u64()); - } - - observe(shared); - Ok(SubmitOutcome { response, wal_lsn }) -} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/admission.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/admission.rs new file mode 100644 index 000000000..7315b095e --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/admission.rs @@ -0,0 +1,94 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Write-admission gate dispatch for the funnel. + +use tokio::sync::OwnedMutexGuard; + +use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::shared::write_admission::{ + WriteAdmission, WriteAdmissionGuard, WriteTarget, admit, +}; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId, VShardId}; + +use super::super::params::WriteOrdering; + +/// Outcome of the write-admission phase. +pub(super) enum AdmissionOutcome { + /// Proceed with the funnel's remaining phases under these guards. + Proceed { + admission: crate::bridge::envelope::Admission, + admission_guard: Option, + order_guard: Option>, + }, + /// The deterministic Calvin scheduler must apply the write. + RouteToCalvin, +} + +/// Write-admission gate: every write-class plan whose ordering is not already +/// final passes here. An uncontended point write takes the fast path holding +/// its per-vShard deterministic locks; a contended or bulk write is submitted +/// through the deterministic scheduler and its applied response is surfaced +/// here; reads / control ops are `Exempt`. +/// +/// Ordering (fast path): the guard is acquired FIRST, then — for a write that +/// owns its durability (`AppendHere`) — the WAL append happens after this +/// call returns, under the guard, minting the LSN just before the enqueue. +/// The guard is released immediately after the enqueue (not across the +/// response await). +pub(super) async fn admit_write( + shared: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + vshard_id: VShardId, + plan: &PhysicalPlan, + ordering: WriteOrdering, +) -> AdmissionOutcome { + match ordering { + WriteOrdering::AlreadyOrdered => AdmissionOutcome::Proceed { + admission: crate::bridge::envelope::Admission::Exempt( + crate::bridge::envelope::ExemptReason::AlreadyOrdered, + ), + admission_guard: None, + order_guard: None, + }, + WriteOrdering::Gate => match admit( + shared, + &WriteTarget { + tenant_id, + database_id, + vshard_id, + plan, + }, + ) { + WriteAdmission::ExemptRead => AdmissionOutcome::Proceed { + admission: crate::bridge::envelope::Admission::Exempt( + crate::bridge::envelope::ExemptReason::Read, + ), + admission_guard: None, + order_guard: None, + }, + WriteAdmission::FastPath { guard } => AdmissionOutcome::Proceed { + admission: crate::bridge::envelope::Admission::Admitted, + admission_guard: guard, + order_guard: None, + }, + WriteAdmission::FastPathBlocking { key, keyed_lock } => { + // Single-node serialization point: acquire the per-key FIFO + // order-lock FIRST, before the WAL append and enqueue below. + // `tokio::sync::Mutex` is fair, so concurrent same-key writers + // are admitted in arrival order — the WAL append + enqueue then + // happen in that order, giving WAL-LSN order == enqueue order == + // apply order per key. Distinct keys use distinct per-key mutexes + // and never contend. + let order_guard = keyed_lock.lock_owned(key).await; + AdmissionOutcome::Proceed { + admission: crate::bridge::envelope::Admission::Admitted, + admission_guard: None, + order_guard: Some(order_guard), + } + } + WriteAdmission::RouteToCalvin => AdmissionOutcome::RouteToCalvin, + }, + } +} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs new file mode 100644 index 000000000..70bccb67b --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs @@ -0,0 +1,173 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Build the wire `Request` and hand it to the Data-Plane dispatcher. + +use std::time::Instant; + +use tokio::sync::OwnedMutexGuard; + +use crate::bridge::envelope::{Admission, PhysicalPlan, Priority, Request}; +use crate::control::array_catalog::ddl::AuthorizedDdlTransition; +use crate::control::server::shared::write_admission::WriteAdmissionGuard; +use crate::control::state::SharedState; +use crate::types::{ + DatabaseId, Lsn, ReadConsistency, RequestId, TenantId, TraceId, TxnId, VShardId, +}; + +use super::wal_append::rollback_on_err; + +/// The write-admission guards a dispatched write must hold across the +/// response await — deferred (rather than dropped at enqueue) only when the +/// write's redo is minted post-apply. +pub(super) type DeferredGuards = Option<(Option, Option>)>; + +/// Everything [`dispatch_to_data_plane`] needs to build the wire `Request`. +pub(super) struct DispatchTarget { + pub tenant_id: TenantId, + pub database_id: DatabaseId, + pub vshard_id: VShardId, + pub plan: PhysicalPlan, + pub deadline: Instant, + pub trace_id: TraceId, + pub event_source: crate::event::EventSource, + pub user_id: Option>, + pub txn_id: Option, + pub wal_lsn: Option, + pub resolved_now_ms: Option, + pub admission: Admission, +} + +/// What the dispatch phase produced: the id the response is tracked under, +/// the receiver it arrives on, the per-vShard latency timer's start instant, +/// and the write-admission guards deferred across the response await (if +/// any). +pub(super) struct DispatchOutcome { + pub request_id: RequestId, + pub rx: crate::control::ResponseReceiver, + pub dispatch_started: Instant, + pub deferred_guards: DeferredGuards, +} + +/// Build the `Request` envelope, register its response channel, and hand it +/// to the Data-Plane dispatcher — then release the write-admission guards +/// (or defer them, for a write whose redo mints post-apply). +/// +/// The per-vShard QPS + latency timer starts here, at the wall-clock moment +/// the request enters dispatch; the caller observes it on every exit path +/// (success, budget over-run, timeout) so the histogram captures the true +/// end-to-end shape of the work routed to this vshard. +/// +/// `waits_for_capacity` makes a capacity refusal wait for freed capacity and +/// retry, up to the request's deadline. Every other refusal returns at once. +pub(super) async fn dispatch_to_data_plane( + shared: &SharedState, + ddl_transition: &AuthorizedDdlTransition, + target: DispatchTarget, + admission_guard: Option, + order_guard: Option>, + post_apply_pending: bool, + waits_for_capacity: bool, +) -> crate::Result { + let dispatch_started = Instant::now(); + + let request_id = shared.next_request_id(); + let request = Request { + request_id, + tenant_id: target.tenant_id, + database_id: target.database_id, + vshard_id: target.vshard_id, + plan: target.plan, + deadline: target.deadline, + priority: Priority::Normal, + trace_id: target.trace_id, + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: target.event_source, + user_roles: Vec::new(), + user_id: target.user_id, + statement_digest: None, + txn_id: target.txn_id, + wal_lsn: target.wal_lsn, + resolved_now_ms: target.resolved_now_ms, + admission: target.admission, + }; + + let rx = shared.tracker.register(request_id); + + let dispatched = if waits_for_capacity { + dispatch_when_capacity_frees(shared, request).await + } else { + match shared.dispatcher.lock() { + Ok(mut d) => d.dispatch(request), + Err(poisoned) => poisoned.into_inner().dispatch(request), + } + }; + if dispatched.is_err() { + // No response will ever arrive for a refused request. + shared.tracker.cancel(&request_id); + } + rollback_on_err(shared, ddl_transition, dispatched)?; + + // Release the write-admission guards immediately after the enqueue, before + // the Data-Plane round-trip. The per-database WFQ is strict FIFO, so once LSN + // order equals enqueue order the apply order follows from the queue alone; + // holding the guards across the response await would only serialize same-key + // throughput needlessly. + // + // EXCEPTION — a post-apply-redo write mints its durable redo AFTER apply, + // from the write-set on the response; the guards MUST stay held across the + // response collect + that append so two concurrent same-surrogate writes + // cannot reorder their redo appends. Both guard types are `Send`, so + // holding them across the `.await` is sound. Moved into an `Option` so the + // release is a single, unconditional `drop` below regardless of which path + // took it. (`None` guard slots when no lock manager was registered / for + // the exempt-read / Calvin / already-ordered cases.) + let deferred_guards = if post_apply_pending { + Some((admission_guard, order_guard)) + } else { + drop(admission_guard); + drop(order_guard); + None + }; + + Ok(DispatchOutcome { + request_id, + rx, + dispatch_started, + deferred_guards, + }) +} + +/// Dispatch `request`, waiting for freed capacity after each capacity +/// refusal, until the request's deadline. Returns the last refusal once the +/// deadline passes. +async fn dispatch_when_capacity_frees(shared: &SharedState, request: Request) -> crate::Result<()> { + let deadline = tokio::time::Instant::from_std(request.deadline); + let capacity_freed = match shared.dispatcher.lock() { + Ok(d) => d.capacity_freed(), + Err(poisoned) => poisoned.into_inner().capacity_freed(), + }; + let mut request = request; + loop { + // Registered before the attempt, so a slot freed between the refusal + // and the wait still wakes it. + let freed = capacity_freed.notified(); + tokio::pin!(freed); + freed.as_mut().enable(); + let attempt = match shared.dispatcher.lock() { + Ok(mut d) => d.try_dispatch(request), + Err(poisoned) => poisoned.into_inner().try_dispatch(request), + }; + let refusal = match attempt { + Ok(()) => return Ok(()), + Err(refusal) => *refusal, + }; + if !matches!(refusal.error, crate::Error::DispatchCapacity { .. }) { + return Err(refusal.error); + } + if tokio::time::timeout_at(deadline, freed).await.is_err() { + return Err(refusal.error); + } + request = refusal.request; + } +} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs new file mode 100644 index 000000000..7f18bc5dc --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs @@ -0,0 +1,317 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The funnel's enqueue phase: runs admission, WAL append, and dispatch in +//! that fixed order for one write. [`super::pending::PendingWrite::finish`] +//! runs the response phase. + +use crate::control::server::dispatch_utils::change_events::extract_write_change_set; +use crate::control::server::dispatch_utils::durability_barrier::funnel_minted_redo_engine; +use crate::control::server::dispatch_utils::minted::{MintedRecords, RecordOwner}; +use crate::control::server::shared::session::statement_deadline; +use crate::control::server::shared::write_admission::{bare_ok_response, route_write_to_calvin}; +use crate::control::server::wal_dispatch; +use crate::control::state::SharedState; + +use super::super::params::{ + ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, +}; +use super::admission::{AdmissionOutcome, admit_write}; +use super::dispatch::{DispatchTarget, dispatch_to_data_plane}; +use super::pending::PendingWrite; +use super::response::{ResponsePhaseInput, UserWriteMark}; +use super::wal_append::authorize_and_append; + +/// Admit, make durable, and enqueue one write on its core. +/// +/// The write is on its core's queue when this returns, so a caller that +/// enqueues writes one after another fixes their arrival order at the core. +/// [`PendingWrite::finish`] collects the outcome. +pub(crate) async fn enqueue_write( + shared: &SharedState, + params: SubmitWrite, +) -> crate::Result { + let SubmitWrite { + tenant_id, + database_id, + vshard_id, + plan, + trace_id, + event_source, + txn_id, + user_id, + mut durability, + ordering, + change_feed, + } = params; + let owner = RecordOwner { + tenant_id, + database_id, + vshard_id, + }; + // On a single node no lease exists, so a write to a permission-tree source + // is acknowledged only once the local permission cache holds it. In a + // cluster the lease barrier on the Raft proposal path covers it instead. + let binds_authorization = shared.authorization_fence.timing().is_none() + && plan.named_collections().iter().any(|collection| { + shared + .authorization_fence + .sources() + .is_source_collection(collection) + }); + // Only a user data write advances the tenant's observed write-HLC, which + // the RESTORE staleness gate compares envelopes against. A schema install + // such as a constraint set writes no row, so it records no mark. The mark + // keeps the path and collection of the write, so a refused restore names it. + let user_write_origin = + crate::control::server::shared::write_admission::plan_writes_user_data(&plan).then(|| { + let site = match &durability { + WalDurability::AppendHere { apply_key: 0, .. } => "write funnel (autocommit)", + WalDurability::AppendHere { .. } => "write funnel (replicated apply)", + WalDurability::CallerSupplied { .. } => "write funnel (caller-appended)", + }; + let collection = plan.named_collections().first().map(|c| (*c).to_owned()); + (site, collection) + }); + // The instant the write committed, which is the value its mark carries: + // - a replicated entry carries its proposer's stamp; + // - a caller that appended upstream committed before this call, so the + // instant this call starts bounds it from above; + // - otherwise the append below is the commit, stamped once it lands. + let upstream_commit_hlc = match &durability { + WalDurability::AppendHere { commit_hlc, .. } => *commit_hlc, + WalDurability::CallerSupplied { .. } => Some(shared.hlc_clock.now().wall_ns), + }; + // Records the caller appended for this write, under their outcome-floor + // window. Every path below closes the window. + let caller_minted = durability.take_minted(); + + // The running statement's deadline, pinned once at the session boundary and + // shared by every request the statement fans out into. Used for both the + // envelope the Data Plane enforces and the Control-Plane collect below, so + // the two halves cannot disagree about when this statement expires. + let deadline = statement_deadline(shared.tuning.network.default_deadline_secs); + + // Change metadata is derived from the plan HERE, before it is moved into + // the request — the publish itself happens after apply, once the response + // (which carries the event's LSN) exists, by which point the plan is gone. + // Extraction is a pure match that clones out collection / document + // identity, so a caller whose change feed is `Unowned` skips it rather than + // allocating tuples nothing will read. + let change_set = match change_feed { + ChangeFeedOwner::Funnel => Some(extract_write_change_set(&plan, tenant_id)), + ChangeFeedOwner::Unowned => None, + }; + + // Post-apply redo classification, computed before `plan` is moved (the + // RouteToCalvin admit arm moves it). For a write whose autocommit WAL path + // mints no redo of its own but whose effect must survive a WAL-only restart + // (a document PointUpdate on a collection carrying a secondary vector + // index), the durable redo is minted AFTER apply from the surrogate + + // post-image the Data Plane returns in `Response::write_set`. + // `Some(collection)` for such a write, else `None`. + let post_apply = wal_dispatch::plan_post_apply_redo(&plan); + let appends_here = matches!(&durability, WalDurability::AppendHere { .. }); + let apply_key = match &durability { + WalDurability::AppendHere { apply_key, .. } => *apply_key, + WalDurability::CallerSupplied { .. } => 0, + }; + // A committed proposal applies in log order against the same state on + // every replica, so a final refusal is its outcome everywhere. The abort + // marker of a final refusal carries the proposal's key, so the proposal + // ledger counts the refusal as the entry's outcome after a restart and a + // redelivered copy is never applied. A write no proposal carries has key + // `0`. + let final_refusal_key = apply_key; + // A write whose order is already final waits out a full dispatcher queue + // rather than failing: a committed entry that fails for local load leaves + // this replica without a write every other replica applied. + let waits_for_capacity = matches!(ordering, WriteOrdering::AlreadyOrdered); + + // Durable-at-ack obligation, also computed before `plan` moves. `Some` only + // for a write whose redo record THIS funnel is required to mint; a caller + // that appended upstream (or declared durability owned elsewhere) is not + // held to it, because the LSN it does or does not supply is its own + // contract. See `durability_barrier` for why this is narrower than + // "write-class plan with no LSN". + let funnel_redo_engine = if appends_here { + funnel_minted_redo_engine(&plan) + } else { + None + }; + + // Write-admission gate: every write-class plan whose ordering is not already + // final passes here. On the Calvin route the deterministic scheduler applies + // the write, emits its own WriteEvents, and owns durability (the sequenced + // TxClass plus its own `CalvinApplied` WAL record), so no local WAL append or + // enqueue happens. A plain write with no RETURNING rows yields `None`, + // synthesized into a bare `Ok`. + let (admission, admission_guard, order_guard) = + match admit_write(shared, tenant_id, database_id, vshard_id, &plan, ordering).await { + AdmissionOutcome::Proceed { + admission, + admission_guard, + order_guard, + } => (admission, admission_guard, order_guard), + AdmissionOutcome::RouteToCalvin => { + // The scheduler applies the write from its own records, so + // the caller's records never apply. + let superseded = caller_minted.map(|minted| { + minted.supersede(std::sync::Arc::clone(&shared.wal), owner, "calvin_route") + }); + let routed = + route_write_to_calvin(shared, tenant_id, database_id, vshard_id, plan).await; + if let Some(superseded) = superseded { + superseded.finish().await; + } + let routed = routed?; + return Ok(PendingWrite::done(SubmitOutcome { + response: routed + .unwrap_or_else(|| bare_ok_response(crate::types::RequestId::new(0))), + wal_lsn: None, + })); + } + }; + + // On a server with no Raft groups a user write that mints its own record + // takes its commit stamp now, and its mark is durable before the mint: a + // crash after the record reaches disk keeps the mark. The stamp stays open + // until the mint, so a backup cut at or above it waits for the record. + let local_stamp = match &user_write_origin { + Some((_, collection)) + if appends_here + && caller_minted.is_none() + && upstream_commit_hlc.is_none() + && shared.async_raft_proposer().is_none() => + { + Some(shared.tenant_marks.stamp_local_write( + &shared.hlc_clock, + shared.credentials.catalog(), + tenant_id.as_u64(), + collection.as_deref(), + )?) + } + _ => None, + }; + + // A write that mints its own LSN opens its outcome-floor window before the + // mint, and appends through it. + let minted = match caller_minted { + Some(minted) => Some(minted), + None => appends_here.then(|| MintedRecords::open(&shared.outcome_floor)), + }; + + // Array DDL authorization + durability, under the admission guard, + // immediately before the enqueue below. + let wal_append_outcome = match authorize_and_append( + shared, + owner, + plan, + durability, + minted.as_ref(), + event_source, + ) { + Ok(outcome) => outcome, + Err(error) => { + // No record of this write reaches a core. + if let Some(minted) = minted { + minted.cancel(&shared.wal, owner, 0).await?; + } + return Err(error); + } + }; + let commit_hlc = local_stamp + .as_ref() + .map(|stamp| stamp.hlc()) + .or(upstream_commit_hlc) + .unwrap_or_else(|| shared.hlc_clock.now().wall_ns); + // The record is minted: a backup cut now waits for it through the outcome + // floor. + drop(local_stamp); + let user_write = user_write_origin.map(|(site, collection)| UserWriteMark { + site, + collection, + commit_hlc, + }); + let ddl_transition = wal_append_outcome.ddl_transition; + let plan = wal_append_outcome.plan; + let wal_lsn = wal_append_outcome.wal_lsn; + let resolved_now_ms = wal_append_outcome.resolved_now_ms; + + // A crash test parks one collection's logged write here: its LSN is + // minted, no core holds it, and only this task waits, so a later write + // with a higher LSN applies first. The write still holds its own per-key + // admission guards, which no write to another key contends on. + #[cfg(feature = "failpoints")] + crate::control::fail_gate::before_dispatch(shared.node_id, &plan, wal_lsn).await; + + // Build the wire request and hand it to the Data-Plane dispatcher. + let dispatched = dispatch_to_data_plane( + shared, + &ddl_transition, + DispatchTarget { + tenant_id, + database_id, + vshard_id, + plan, + deadline, + trace_id, + event_source, + user_id, + txn_id, + wal_lsn, + resolved_now_ms, + admission, + }, + admission_guard, + order_guard, + post_apply.is_some(), + waits_for_capacity, + ) + .await; + let dispatch_outcome = match dispatched { + Ok(outcome) => { + // A core holds the request now. From here the records close + // from its final response, never from a drop. + if let Some(minted) = &minted { + minted.mark_sent(); + } + outcome + } + Err(error) => { + // The dispatcher refused the request, so no core applied it. A + // dispatch refusal depends on this node's load, so the markers + // carry no proposal key. + if let Some(minted) = minted { + minted.cancel(&shared.wal, owner, 0).await?; + } + return Err(error); + } + }; + + // The response phase collects the outcome and runs the post-apply steps a + // successful write still owes. + Ok(PendingWrite::dispatched( + ResponsePhaseInput { + request_id: dispatch_outcome.request_id, + rx: dispatch_outcome.rx, + deadline, + dispatch_started: dispatch_outcome.dispatch_started, + tenant_id, + database_id, + vshard_id, + wal_lsn, + appends_here, + final_refusal_key, + apply_key, + event_source, + post_apply, + funnel_redo_engine, + change_set, + ddl_transition, + deferred_guards: dispatch_outcome.deferred_guards, + minted, + user_write, + }, + binds_authorization, + )) +} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs new file mode 100644 index 000000000..9e710788a --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs @@ -0,0 +1,44 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! THE single Control-Plane write funnel. +//! +//! Every Control-Plane path that puts a write on the SPSC bridge routes through +//! [`submit_write`]: the autocommit / internal funnel +//! ([`crate::control::server::dispatch_utils::dispatch`]), the pgwire +//! local-dispatch path (`pgwire::handler::submit`), and the Raft apply loop +//! (`distributed_applier::apply_loop`). The funnel owns write admission, the +//! WAL redo append, the enqueue, the bounded response collect, the post-apply +//! redo, the durable-at-ack barrier, and — for the caller that owns it (see +//! [`super::params::ChangeFeedOwner`]) — the CDC publish, in that order, which +//! is the correctness contract. +//! +//! A path that reimplements these steps drifts silently: it is not a compile +//! error to omit the redo append or the change-event publish, and the omission +//! only surfaces as lost data after a crash, or as a change stream that never +//! fires. Add the step here, once, and every caller gets it. +//! +//! It also owns the mirror of the redo append: every record of the write, the +//! ones it appended and the ones a caller appended under their outcome-floor +//! window, is cancelled when the Data Plane refuses the write with a verdict +//! that proves nothing applied. See +//! [`crate::control::server::dispatch_utils::write_abort`] for which verdicts +//! qualify, and the `minted` module for how each path closes the window. +//! +//! Split by concern, run in this fixed order by [`driver::submit_write`]: +//! - [`admission`]: the write-admission gate. +//! - [`wal_append`]: Array DDL authorization and the WAL redo append/stamp. +//! - [`dispatch`]: building the wire `Request` and handing it to the Data +//! Plane. +//! - [`response`]: collecting the response, classifying the outcome, and the +//! post-apply steps a successful write still owes. +//! - [`pending`]: the write between its enqueue and its response phase. + +mod admission; +mod dispatch; +mod driver; +mod pending; +mod response; +mod wal_append; + +pub(crate) use driver::enqueue_write; +pub(crate) use pending::{PendingWrite, submit_write}; diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/pending.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/pending.rs new file mode 100644 index 000000000..5dc04cf15 --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/pending.rs @@ -0,0 +1,85 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A write the funnel enqueued on its core, and the response phase that +//! finishes it. +//! +//! The enqueue and the response phase are separate so a caller can enqueue +//! writes in a fixed order and collect their outcomes in any order. The +//! data-group apply loop does this: it enqueues committed entries in log +//! order and collects each outcome independently, so one parked write never +//! holds back the writes behind it. + +use crate::control::state::SharedState; + +use super::super::params::{SubmitOutcome, SubmitWrite}; +use super::driver::enqueue_write; +use super::response::{ResponsePhaseInput, collect_classify_and_finish}; + +/// A write past its enqueue. +pub(crate) struct PendingWrite { + stage: Stage, +} + +enum Stage { + /// The write has its outcome already: the Calvin scheduler applied it. + Done(SubmitOutcome), + /// A core holds the write. The response phase collects its outcome. + Dispatched { + input: Box, + /// The write changes a permission-tree source on a node with no + /// lease, so its ack waits until the local permission cache holds it. + binds_authorization: bool, + }, +} + +impl PendingWrite { + pub(super) fn done(outcome: SubmitOutcome) -> Self { + Self { + stage: Stage::Done(outcome), + } + } + + pub(super) fn dispatched(input: ResponsePhaseInput, binds_authorization: bool) -> Self { + Self { + stage: Stage::Dispatched { + input: Box::new(input), + binds_authorization, + }, + } + } + + /// Collect the outcome, classify it, and run every step a completed write + /// still owes before it is acknowledged. + /// + /// See [`SubmitOutcome`] for what comes back. + pub(crate) async fn finish(self, shared: &SharedState) -> crate::Result { + let (input, binds_authorization) = match self.stage { + Stage::Done(outcome) => return Ok(outcome), + Stage::Dispatched { + input, + binds_authorization, + } => (input, binds_authorization), + }; + let max_result_bytes = shared.tuning.network.max_query_result_bytes as usize; + let outcome = collect_classify_and_finish(shared, max_result_bytes, *input).await?; + if binds_authorization { + crate::control::security::auth_lease::await_local_coverage( + shared, + std::time::Instant::now() + + std::time::Duration::from_secs(shared.tuning.network.default_deadline_secs), + ) + .await?; + } + Ok(outcome) + } +} + +/// Admit, make durable, enqueue, collect, and publish one write. +/// +/// See [`SubmitOutcome`] for what comes back. +pub(crate) async fn submit_write( + shared: &SharedState, + params: SubmitWrite, +) -> crate::Result { + enqueue_write(shared, params).await?.finish(shared).await +} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs new file mode 100644 index 000000000..dced5d147 --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs @@ -0,0 +1,339 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Collect the Data Plane's response, classify the outcome, and run the +//! post-apply steps a successful write still owes: the post-apply redo, the +//! durable-at-ack barrier, DDL finalization, and the change-event publish. + +use std::sync::Arc; +use std::time::Instant; + +use crate::bridge::envelope::Status; +use crate::control::array_catalog::ddl::AuthorizedDdlTransition; +use crate::control::local_dispatch::{DispatchCollectError, collect_bounded_response}; +use crate::control::server::dispatch_utils::change_events::{WriteChangeSet, publish_change_set}; +use crate::control::server::dispatch_utils::durability_barrier::assert_durable_before_ack; +use crate::control::server::dispatch_utils::minted::{ + Collect, MintedRecords, OwnedResponse, OwnedWait, RecordOwner, await_response_owned, +}; +use crate::control::server::dispatch_utils::submit_write::ambiguous_ddl::preserve_ambiguous_array_ddl; +use crate::control::server::wal_dispatch; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, Lsn, RequestId, TenantId, VShardId}; + +use super::super::params::SubmitOutcome; +use super::wal_append::rollback_on_err; + +/// Everything the response phase needs, gathered from the admission, WAL +/// append, and dispatch phases that ran before it. +pub(super) struct ResponsePhaseInput { + pub request_id: RequestId, + pub rx: crate::control::ResponseReceiver, + pub deadline: Instant, + pub dispatch_started: Instant, + pub tenant_id: TenantId, + pub database_id: DatabaseId, + pub vshard_id: VShardId, + pub wal_lsn: Option, + pub appends_here: bool, + /// The idempotency key every record this write appends carries (see + /// `WalDurability::AppendHere`). + pub apply_key: u64, + /// The event source the write runs with. Its post-apply redo records + /// carry it. + pub event_source: crate::event::EventSource, + /// The key a final refusal's abort marker carries, `0` when this write's + /// refusals are not final. A final refusal is the proposal's outcome: the + /// proposal ledger rebuilt at boot counts the key as applied. + pub final_refusal_key: u64, + pub post_apply: Option, + pub funnel_redo_engine: Option<&'static str>, + pub change_set: Option, + pub ddl_transition: AuthorizedDdlTransition, + pub deferred_guards: super::dispatch::DeferredGuards, + /// The records minted for this write, under their outcome-floor window. + pub minted: Option, + /// `Some` when the plan writes user data, so a success advances the + /// tenant's observed write-HLC. Reads and system operations never do. + pub user_write: Option, +} + +/// The origin a successful user data write records on the tenant's observed +/// write-HLC. +pub(super) struct UserWriteMark { + /// Which funnel path dispatched the write. + pub site: &'static str, + /// The collection the plan wrote, when it named one. + pub collection: Option, + /// HLC wall time, in nanoseconds, at which the write committed. + pub commit_hlc: u64, +} + +/// Collect the response(s), classify the outcome, and run every step a +/// completed write still owes before the funnel returns. +/// +/// For non-streaming queries, exactly one response arrives. For streaming +/// queries, multiple partial chunks arrive before the final. The partial +/// channel is bounded (see `RequestTracker::register`); here the *total* accumulated +/// payload is additionally capped so a runaway scan can't pin Control-Plane +/// RAM — any query whose combined result exceeds +/// `tuning.network.max_query_result_bytes` is cancelled with a typed +/// `ExecutionLimitExceeded` error. +pub(super) async fn collect_classify_and_finish( + shared: &SharedState, + max_result_bytes: usize, + input: ResponsePhaseInput, +) -> crate::Result { + let ResponsePhaseInput { + request_id, + rx, + deadline, + dispatch_started, + tenant_id, + database_id, + vshard_id, + wal_lsn, + appends_here, + apply_key, + event_source, + final_refusal_key, + post_apply, + funnel_redo_engine, + change_set, + ddl_transition, + deferred_guards, + minted, + user_write, + } = input; + let owner = RecordOwner { + tenant_id, + database_id, + vshard_id, + }; + + let vshard_u32 = vshard_id.as_u32(); + let observe = |shared: &SharedState| { + let latency_us = dispatch_started.elapsed().as_micros().min(u64::MAX as u128) as u64; + shared.per_vshard_metrics.observe(vshard_u32, latency_us); + }; + + // Wait to the same instant the envelope carries. The Data Plane normally + // answers with `DeadlineExceeded` first; this bounds the wait when it is + // inside a stage that carries no safe point yet. A write's records close + // in a task this future does not own, so a caller dropped mid-wait still + // closes them. A refusal that arrives after the deadline still cancels + // them. + let outcome = match minted { + Some(minted) => { + await_response_owned( + OwnedWait { + wal: Arc::clone(&shared.wal), + owner, + final_refusal_key, + deadline, + collect: Collect::Merged { max_result_bytes }, + }, + rx, + minted, + ) + .await? + } + None => collect_unminted(shared, request_id, rx, deadline, max_result_bytes).await, + }; + + let response = match outcome { + OwnedResponse::Answered { response, closed } => { + if response.status != Status::Ok { + let _ = ddl_transition.rollback(shared); + } + // A failed cancel holds the window and fails the write here. + closed?; + response + } + OwnedResponse::DeadlineExceeded => { + observe(shared); + // Dispatch completed, but the Data Plane may have applied CREATE + // or ALTER before this deadline. Never roll that catalog state + // back on an ambiguous post-enqueue outcome. + preserve_ambiguous_array_ddl(shared, &ddl_transition); + if !ddl_transition.preserves_on_ambiguous_apply() { + let _ = ddl_transition.rollback(shared); + } + return Err(crate::Error::DeadlineExceeded { request_id }); + } + OwnedResponse::OverBudget { bytes } => { + observe(shared); + // A partial response proves dispatch began but not whether an + // Array DDL completed; preserve CREATE/ALTER and fail-stop. + preserve_ambiguous_array_ddl(shared, &ddl_transition); + if !ddl_transition.preserves_on_ambiguous_apply() { + let _ = ddl_transition.rollback(shared); + } + return Err(crate::Error::ExecutionLimitExceeded { + detail: format!( + "query result exceeded max_query_result_bytes \ + ({bytes} > {max_result_bytes} bytes)" + ), + }); + } + OwnedResponse::ChannelClosed => { + observe(shared); + // The producer can close after applying but before sending its + // response. CREATE/ALTER must remain catalog-finalized here. + preserve_ambiguous_array_ddl(shared, &ddl_transition); + if !ddl_transition.preserves_on_ambiguous_apply() { + let _ = ddl_transition.rollback(shared); + } + // A producer that stopped after the deadline stopped because the + // statement ran out of time. Reporting the closure would hand the + // client an internal error for its own timeout. + if std::time::Instant::now() >= deadline { + return Err(crate::Error::DeadlineExceeded { request_id }); + } + return Err(crate::Error::Dispatch { + detail: "response channel closed".into(), + }); + } + }; + + // Mint the post-apply redo record while the guards are still held, then + // release them. A PointUpdate whose collection carries a secondary vector + // index returns its surrogate + post-image in `write_set`; without this + // durable `Put` a WAL-only restart rebuilds the HNSW from the pre-update body + // and resurrects the old embedding. + let post_apply_lsn = if let Some(collection) = &post_apply + && appends_here + && response.status == Status::Ok + { + rollback_on_err( + shared, + &ddl_transition, + wal_dispatch::append_write_set_redo( + shared + .wal + .appender(apply_key) + .with_event_source(event_source), + tenant_id, + vshard_id, + database_id, + collection, + &response.write_set, + ), + )? + } else { + None + }; + drop(deferred_guards); + + // Durable-at-ack barrier: an acknowledged write must be WAL-fsync-durable + // before this response (the client ack) returns. `WalAppender::append_*` only + // buffers the record and mints its `Lsn`; without this barrier a `kill -9` + // loses the buffered bytes, which is invisible for engines whose rows are + // committed durably by redb but silently destroys every engine whose only + // durability path is WAL replay: the KV hash tables, the HNSW graphs, the + // columnar / timeseries memtables, the graph node labels, the CRDT states, + // and the FTS index. `wal_lsn` is the forward write's LSN — minted above + // under the admission guard for a write that owns its durability, or supplied + // by a caller that appended upstream (procedural batch flush, + // interactive-COMMIT transaction redo). `post_apply_lsn` covers the + // post-apply redo appended just above. Both records are already buffered in + // the shared WAL; one group-commit fsync coalesces concurrent writers (see + // `WalManager::wait_durable`), and it runs here — after the admission guards + // are released — so it never serializes same-key throughput. Reads / control + // ops / trigger / staged-write dispatch carry no LSN and skip the barrier; + // `durability_barrier` decides which of those skips are legitimate and makes + // the rest loud instead of letting them ack a write nothing can recover. + // On a server with no Raft groups the write's mark lives in the catalog. + // It is durable before the ack. A write that minted its own record made it + // durable before the mint, so this costs no catalog commit there. + if response.status == Status::Ok + && let Some(mark) = &user_write + && shared.async_raft_proposer().is_none() + { + rollback_on_err( + shared, + &ddl_transition, + shared.tenant_marks.record_local_write( + shared.credentials.catalog(), + tenant_id.as_u64(), + mark.commit_hlc, + mark.collection.as_deref(), + ), + )?; + } + + if response.status == Status::Ok { + let durable_target = match (wal_lsn, post_apply_lsn) { + (Some(a), Some(b)) => Some(a.max(b)), + (a, b) => a.or(b), + }; + match durable_target { + Some(lsn) => { + rollback_on_err(shared, &ddl_transition, shared.wal.wait_durable(lsn).await)? + } + // Nothing to fsync. Legitimate for most plans, but if this funnel + // was the one required to mint the record, the ack below promises + // durability the engine cannot deliver — silent until a `kill -9` + // proves it, hence the check. + None => assert_durable_before_ack(funnel_redo_engine), + } + } + + if response.status == Status::Ok { + ddl_transition.finalize(shared)?; + } + + // Publish change events for successful writes whose change feed this funnel + // owns. `None` is a caller whose change feed is `Unowned` — see + // [`super::super::params::ChangeFeedOwner`] for why the node that applies those + // writes is not the node that publishes them. + if response.status == Status::Ok { + if let Some(change_set) = change_set { + publish_change_set(shared, tenant_id, database_id, change_set, &response); + } + + // Record the write's commit HLC on the tenant's observed high-water + // before this response, the ack, returns. The RESTORE staleness gate + // refuses an envelope older than the mark, so the mark is the instant + // the write committed, never the instant this bookkeeping ran: a + // backup taken after the ack then always carries a newer watermark. + if let Some(mark) = &user_write { + shared.advance_tenant_write_hlc( + tenant_id.as_u64(), + mark.commit_hlc, + mark.site, + mark.collection.as_deref(), + ); + } + } + + observe(shared); + Ok(SubmitOutcome { response, wal_lsn }) +} + +/// Collect a response that carries no records, under the same deadline and +/// byte budget a write's owned wait applies. +async fn collect_unminted( + shared: &SharedState, + request_id: RequestId, + mut rx: crate::control::ResponseReceiver, + deadline: Instant, + max_result_bytes: usize, +) -> OwnedResponse { + let collected = tokio::time::timeout_at( + tokio::time::Instant::from_std(deadline), + collect_bounded_response(&mut rx, max_result_bytes), + ) + .await; + match collected { + Ok(Ok(response)) => OwnedResponse::Answered { + response, + closed: Ok(()), + }, + Ok(Err(DispatchCollectError::OverBudget { bytes })) => { + shared.tracker.cancel(&request_id); + OwnedResponse::OverBudget { bytes } + } + Ok(Err(DispatchCollectError::ChannelClosed)) => OwnedResponse::ChannelClosed, + Err(_) => OwnedResponse::DeadlineExceeded, + } +} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs new file mode 100644 index 000000000..0f3900c94 --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs @@ -0,0 +1,128 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Array DDL authorization and WAL redo append/stamp for the funnel. +//! +//! Array DDL conversion is intentionally read-only. Once a task has passed +//! authorization and admission, its durable catalog state installs +//! immediately before the Data-Plane dispatch; the mirror changes only after +//! the redb transaction commits. The DDL transition this creates must be +//! rolled back by every later phase that can fail before the Data Plane has +//! proved it applied the write — see [`rollback_on_err`]. + +use crate::bridge::envelope::PhysicalPlan; +use crate::control::array_catalog::ddl::AuthorizedDdlTransition; +use crate::control::server::dispatch_utils::minted::{MintedRecords, RecordOwner}; +use crate::control::server::wal_dispatch::{self, WalAppendRequest}; +use crate::control::state::SharedState; +use crate::types::Lsn; + +use super::super::params::WalDurability; + +/// What the authorize-and-append phase produced: the DDL transition every +/// later phase must roll back on failure, the plan (stamped with its minted +/// LSN, if any), and the resolved durability values. +pub(super) struct WalAppendOutcome { + pub ddl_transition: AuthorizedDdlTransition, + pub plan: PhysicalPlan, + pub wal_lsn: Option, + pub resolved_now_ms: Option, +} + +/// Roll `ddl_transition` back and convert `result`'s error, or pass a success +/// through unchanged. Every phase after DDL authorization that can fail calls +/// this instead of returning its error directly, so an authorized Array +/// CREATE/DROP/ALTER never survives a failure later in the sequence. +pub(super) fn rollback_on_err( + shared: &SharedState, + ddl_transition: &AuthorizedDdlTransition, + result: crate::Result, +) -> crate::Result { + match result { + Ok(value) => Ok(value), + Err(error) => { + let _ = ddl_transition.rollback(shared); + Err(error) + } + } +} + +/// Install the plan's Array DDL catalog transition, then make the write +/// durable: append its WAL redo record here (under the write-admission guard +/// the caller already holds) or take the LSN a caller supplied upstream. +/// +/// Durability, under the guard, immediately before the enqueue: the LSN is +/// minted in the same order the request is about to be enqueued. +/// +/// Writes the resolved LSN back into the plan itself. The envelope's +/// `wal_lsn` is where most engines read their committed version from, but the +/// array engine stamps its tile versions from the LSN carried in the plan +/// while replay stamps them from the record header — so the plan the Data +/// Plane is about to execute must name the record that reproduces it. This is +/// the only place that knows both, and it knows them for every caller: no +/// upstream path may allocate an LSN of its own and hope it matches. +/// +/// An `AppendHere` write appends through `minted`, so every record it writes +/// joins the write's outcome-floor window. +pub(super) fn authorize_and_append( + shared: &SharedState, + owner: RecordOwner, + mut plan: PhysicalPlan, + durability: WalDurability, + minted: Option<&MintedRecords>, + event_source: crate::event::EventSource, +) -> crate::Result { + let RecordOwner { + tenant_id, + database_id, + vshard_id, + } = owner; + let ddl_transition = crate::control::array_catalog::ddl::apply_authorized_ddl( + shared, + tenant_id, + database_id, + &plan, + )?; + + let (wal_lsn, resolved_now_ms) = match durability { + WalDurability::AppendHere { + now_override, + apply_key, + .. + } => { + let outcome = rollback_on_err( + shared, + &ddl_transition, + wal_dispatch::wal_append(WalAppendRequest { + wal: match minted { + Some(minted) => minted.appender(&shared.wal, apply_key), + None => shared.wal.appender(apply_key), + }, + event_source, + tenant_id, + vshard_id, + database_id, + plan: &plan, + credentials: None, + now_override, + }), + )?; + (outcome.lsn, outcome.resolved_now_ms) + } + WalDurability::CallerSupplied { + wal_lsn, + resolved_now_ms, + .. + } => (wal_lsn, resolved_now_ms), + }; + + if let Some(lsn) = wal_lsn { + wal_dispatch::stamp_minted_lsn(&mut plan, lsn); + } + + Ok(WalAppendOutcome { + ddl_transition, + plan, + wal_lsn, + resolved_now_ms, + }) +} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/mod.rs b/nodedb/src/control/server/dispatch_utils/submit_write/mod.rs index 7da9ef9a5..e6cdd4c6c 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/mod.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/mod.rs @@ -4,7 +4,7 @@ mod ambiguous_ddl; mod funnel; mod params; -pub(crate) use funnel::submit_write; +pub(crate) use funnel::{PendingWrite, enqueue_write, submit_write}; pub(crate) use params::{ ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, }; diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/params.rs b/nodedb/src/control/server/dispatch_utils/submit_write/params.rs index 642e12d5e..7bdaab306 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/params.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/params.rs @@ -10,6 +10,7 @@ use std::sync::Arc; use crate::bridge::envelope::{PhysicalPlan, Response}; +use crate::control::server::dispatch_utils::minted::MintedRecords; use crate::types::{DatabaseId, Lsn, TenantId, TraceId, TxnId, VShardId}; /// Who owns this write's durable redo record. @@ -20,18 +21,60 @@ pub(crate) enum WalDurability { /// is what makes WAL-LSN order equal dispatcher-enqueue order per key; the /// strict-FIFO per-database WFQ then makes apply order follow enqueue /// order, so restart replay (in LSN order) cannot diverge from live state. - AppendHere { now_override: Option }, + /// + /// `apply_key` is the idempotency key of the replicated proposal this + /// write applies, `0` for a write no proposal carries. Every record the + /// funnel appends for the write carries it in its header, so the record + /// names the proposal it applied in the same durable write. + /// + /// `commit_hlc` is the HLC wall time, in nanoseconds, at which the write + /// committed upstream: the proposer's stamp on a replicated entry. `None` + /// when this append is the commit, so the funnel stamps the instant of the + /// append itself. + AppendHere { + now_override: Option, + apply_key: u64, + commit_hlc: Option, + }, /// The caller already recorded this write's durability elsewhere — COMMIT's /// single `Transaction` record, the procedural batch flush, a trigger / /// sync path that owns its own funnel — and supplies the LSN it minted. /// The funnel appends nothing and stamps these values through unchanged; /// the supplied LSN names the record that replays this write. + /// + /// `minted` holds the records the caller appended for this write under + /// their outcome-floor window. The funnel closes the window from the + /// write's outcome: it cancels the records on a refusal that applied + /// nothing, and on a Calvin route that applies the write from its own + /// records. CallerSupplied { wal_lsn: Option, resolved_now_ms: Option, + minted: Option, }, } +impl WalDurability { + /// Whether the caller supplied records it appended for this write. + pub(crate) fn has_minted(&self) -> bool { + matches!( + self, + Self::CallerSupplied { + minted: Some(_), + .. + } + ) + } + + /// Take the caller's minted records out, leaving `None` in their place. + pub(crate) fn take_minted(&mut self) -> Option { + match self { + Self::AppendHere { .. } => None, + Self::CallerSupplied { minted, .. } => minted.take(), + } + } +} + /// Where this write's ordering was decided. pub(crate) enum WriteOrdering { /// Run the write-admission gate: fast path, per-key order lock, or a route diff --git a/nodedb/src/control/server/dispatch_utils/types.rs b/nodedb/src/control/server/dispatch_utils/types.rs index 0255da7a9..317b11d77 100644 --- a/nodedb/src/control/server/dispatch_utils/types.rs +++ b/nodedb/src/control/server/dispatch_utils/types.rs @@ -33,12 +33,16 @@ pub(crate) struct WriteDispatch { pub txn_id: Option, pub wal_lsn: Option, /// Wall-clock instant (ms since epoch) the Control Plane resolved at - /// WAL-append time for a TTL-bearing KV write's `expire_at_ms`. Stamped - /// onto the `Request` (same as `wal_lsn`) so the Data Plane installs the - /// SAME instant the durable WAL record carries instead of re-reading the - /// clock at apply time. `None` for reads, non-TTL writes, and writes whose - /// resolved instant is not (yet) threaded. + /// WAL-append time: a TTL-bearing KV write's expiry base, or a timeseries + /// ingest's default row timestamp. Stamped onto the `Request` (same as + /// `wal_lsn`) so the Data Plane installs the SAME instant the durable WAL + /// record carries instead of re-reading the clock at apply time. `None` + /// for reads and other writes. pub resolved_now_ms: Option, + /// The records the caller appended for this write, under their + /// outcome-floor window. The funnel closes the window from the write's + /// outcome. `None` when the caller appended nothing for this dispatch. + pub minted: Option, } /// Inputs for `dispatch_to_data_plane_inner`: the Data Plane request identity diff --git a/nodedb/src/control/server/dispatch_utils/write_abort.rs b/nodedb/src/control/server/dispatch_utils/write_abort.rs index 373523bef..d6ccc2f1f 100644 --- a/nodedb/src/control/server/dispatch_utils/write_abort.rs +++ b/nodedb/src/control/server/dispatch_utils/write_abort.rs @@ -22,86 +22,82 @@ //! The match is exhaustive on purpose. A new [`ErrorCode`] must be classified by //! whoever adds it, not silently inherit either answer. //! -//! [`abort_refused_write`] is the one place that acts on that verdict, and the -//! write funnel is its only caller — every `AppendHere` write in every engine, -//! including the Raft apply loop, passes through there. +//! `minted::resolve_on_response` is the one place that acts on that +//! verdict. Every write that mints a record for a Data-Plane dispatch +//! resolves its records there. -use crate::bridge::envelope::{ErrorCode, Response}; -use crate::control::state::SharedState; -use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; +use crate::bridge::envelope::ErrorCode; -/// Identity of the forward record an abort marker would name. -pub(crate) struct AbortTarget { - pub tenant_id: TenantId, - pub database_id: DatabaseId, - pub vshard_id: VShardId, - /// The forward write's redo LSN, if one was minted at all. - pub wal_lsn: Option, - /// Whether the write funnel appended the forward record. A caller that - /// recorded durability elsewhere owns the undo semantics of its own record. - pub appends_here: bool, +/// Whether a replicated proposal refused with `code` is refused for good: a +/// redelivery of the same entry against the same state refuses it again. +/// +/// A verdict that depends on this node's momentary load or on a transient +/// precondition is not final: another replica can apply the same entry, and +/// a redelivery here can too. That covers admission and capacity verdicts, +/// a task that expired before it started, concurrency retries, the staging +/// byte budget, a sync hold, which depends on the core's own stream mark, +/// and `RetryableRefusal`, +/// which a committed-redo apply answers with after it rolled a failed +/// install back. +pub(crate) fn refusal_is_final(code: &ErrorCode) -> bool { + write_definitely_not_applied(code) && !is_transient_verdict(code) } -/// Cancel a forward write record the Data Plane refused. -/// -/// The write funnel appends the redo record before the Data Plane has decided, -/// so a refusal arrives with the record already in the log and restart replay -/// would re-apply the very write the client was told was rejected. This writes a -/// `WriteAborted` marker naming that record, then waits for it to be fsynced -/// BEFORE the error is returned: the refusal path performs no fsync of its own, -/// so an abort left buffered is volatile while the forward record may already be -/// durable via a concurrent writer's group commit. -/// -/// Cost: a rejected write now pays a WAL append plus an fsync wait it did not -/// pay before. That is a deliberate trade of refusal latency for the guarantee -/// that a refusal, once acknowledged, stays refused. -/// -/// **Known residual, not closed:** a crash BEFORE the abort record is durable -/// can still resurrect the write, because the forward record's durability is not -/// gated on the verdict. Closing that window means holding the per-key order -/// guard across the whole Data-Plane round trip, which would serialize same-key -/// writes on every engine's common path. What this guarantees is the ACKED case: -/// once the client has been told the write was refused, a restart cannot make it -/// appear. -/// -/// An append or fsync failure here is propagated, not logged: continuing would -/// return the refusal while leaving the forward record replayable, which is -/// exactly the bug this exists to prevent. -pub(crate) async fn abort_refused_write( - shared: &SharedState, - target: AbortTarget, - response: &Response, -) -> crate::Result<()> { - if !target.appends_here { - return Ok(()); - } - let Some(wal_lsn) = target.wal_lsn else { - return Ok(()); - }; - // A rejection is only cancellable when the verdict itself proves nothing - // was installed. An ambiguous failure keeps its forward record, because - // erasing a write that actually landed is worse than replaying one that - // did not. - let Some(code) = response.error_code.as_deref() else { - return Ok(()); - }; - if !write_definitely_not_applied(code) { - return Ok(()); +/// Whether `code` depends on this node's momentary load or on a transient +/// precondition, so a redelivery of the same entry can apply it. +fn is_transient_verdict(code: &ErrorCode) -> bool { + match code { + ErrorCode::RetryableRefusal { .. } + | ErrorCode::SyncNotApplied { .. } + | ErrorCode::RateExceeded { .. } + | ErrorCode::CollectionDraining { .. } + | ErrorCode::DispatchCapacity { .. } + | ErrorCode::ExpiredBeforeExecution + | ErrorCode::ConflictRetry + | ErrorCode::OllpRetryRequired + | ErrorCode::TxnOverlayMemoryExceeded { .. } + | ErrorCode::TransactionRollback { .. } => true, + ErrorCode::DeadlineExceeded + | ErrorCode::ActiveSqlTransaction { .. } + | ErrorCode::DependentObjectsExist { .. } + | ErrorCode::RejectedConstraint { .. } + | ErrorCode::RejectedPrevalidation { .. } + | ErrorCode::SyncRejected { .. } + | ErrorCode::NotFound + | ErrorCode::RejectedAuthz { .. } + | ErrorCode::CrdtFrontierMismatch { .. } + | ErrorCode::FanOutExceeded + | ErrorCode::ResourcesExhausted + | ErrorCode::RejectedDanglingEdge { .. } + | ErrorCode::DuplicateWrite + | ErrorCode::AppendOnlyViolation { .. } + | ErrorCode::BalanceViolation { .. } + | ErrorCode::PeriodLocked { .. } + | ErrorCode::PeriodLockMisconfigured { .. } + | ErrorCode::RetentionViolation { .. } + | ErrorCode::LegalHoldActive { .. } + | ErrorCode::StateTransitionViolation { .. } + | ErrorCode::TransitionCheckViolation { .. } + | ErrorCode::TypeGuardViolation { .. } + | ErrorCode::TypeMismatch { .. } + | ErrorCode::CounterFault { .. } + | ErrorCode::InsufficientBalance { .. } + | ErrorCode::RecursionDepthExceeded { .. } + | ErrorCode::UndefinedColumn { .. } + | ErrorCode::Internal { .. } + | ErrorCode::Unsupported { .. } + | ErrorCode::RollbackFailed { .. } + | ErrorCode::DivisionByZero + | ErrorCode::UndefinedFunction { .. } + | ErrorCode::DataException { .. } + | ErrorCode::BadRequest { .. } => false, } +} - let abort_lsn = shared.wal.append_write_aborted( - target.tenant_id, - target.vshard_id, - target.database_id, - wal_lsn, - )?; - shared.wal.wait_durable(abort_lsn).await?; - tracing::debug!( - aborted_lsn = wal_lsn.as_u64(), - abort_lsn = abort_lsn.as_u64(), - "refused write cancelled in the WAL" - ); - Ok(()) +/// Whether a committed proposal's apply `error` is a final refusal: the +/// entry's outcome, which a redelivery must answer with and never apply. +pub(crate) fn error_is_final_refusal(error: &crate::Error) -> bool { + matches!(error, crate::Error::DataPlane(code) if refusal_is_final(code)) } /// Whether `code` proves the write was refused without applying anything. @@ -114,6 +110,10 @@ pub(crate) fn write_definitely_not_applied(code: &ErrorCode) -> bool { // refusal instead of installing the write. ErrorCode::RejectedConstraint { .. } | ErrorCode::RejectedPrevalidation { .. } + // The sync gate refused the frame before its delta installed. + | ErrorCode::SyncRejected { .. } + // The sync gate held the frame back before anything installed. + | ErrorCode::SyncNotApplied { .. } | ErrorCode::RejectedAuthz { .. } | ErrorCode::RejectedDanglingEdge { .. } | ErrorCode::AppendOnlyViolation { .. } @@ -126,12 +126,15 @@ pub(crate) fn write_definitely_not_applied(code: &ErrorCode) -> bool { | ErrorCode::TransitionCheckViolation { .. } | ErrorCode::TypeGuardViolation { .. } | ErrorCode::TypeMismatch { .. } - | ErrorCode::OverflowError { .. } + | ErrorCode::CounterFault { .. } | ErrorCode::InsufficientBalance { .. } // Admission verdicts: the request never reached the mutation at all. | ErrorCode::RateExceeded { .. } | ErrorCode::CollectionDraining { .. } + | ErrorCode::DispatchCapacity { .. } | ErrorCode::Unsupported { .. } + // The deadline passed before the core started the task. + | ErrorCode::ExpiredBeforeExecution // The target row or collection did not exist, so the write had nothing // to mutate. | ErrorCode::NotFound @@ -141,11 +144,19 @@ pub(crate) fn write_definitely_not_applied(code: &ErrorCode) -> bool { // Concurrency verdicts that abort the whole attempt before install. | ErrorCode::ConflictRetry | ErrorCode::OllpRetryRequired + // The whole transaction aborted before any read-set was validated. + | ErrorCode::TransactionRollback { .. } + // Refused by the transaction state, or by the target's dependents, + // before the statement ran. + | ErrorCode::ActiveSqlTransaction { .. } + | ErrorCode::DependentObjectsExist { .. } // The staging overlay hit its byte budget, so the transaction's writes // were discarded from the overlay and never installed. | ErrorCode::TxnOverlayMemoryExceeded { .. } // Expression evaluation failed before producing a value to write. | ErrorCode::DivisionByZero + | ErrorCode::UndefinedFunction { .. } + | ErrorCode::DataException { .. } | ErrorCode::UndefinedColumn { .. } => true, // NOT established — every one of these can be reported by a request @@ -166,7 +177,10 @@ pub(crate) fn write_definitely_not_applied(code: &ErrorCode) -> bool { // * `DuplicateWrite` — the idempotency gate fired because the write // ALREADY applied under the original request; nothing to undo, and // the duplicate record replays to the same state. + // * `BadRequest` — raised by many engine paths, some of them after a + // multi-row plan already wrote rows. ErrorCode::DeadlineExceeded + | ErrorCode::BadRequest { .. } | ErrorCode::RollbackFailed { .. } | ErrorCode::ResourcesExhausted | ErrorCode::Internal { .. } @@ -205,6 +219,25 @@ mod tests { )); } + /// A dispatcher capacity refusal enqueued nothing, so the record aborts. + #[test] + fn dispatch_capacity_refusal_aborts_the_record() { + assert!(write_definitely_not_applied(&ErrorCode::DispatchCapacity { + reason: "core 0 queue is full at 64 requests".into(), + })); + } + + /// A task that expired before its core started it ran nothing, so the + /// record aborts. A redelivery can still run it, so the refusal is not + /// final. + #[test] + fn a_task_that_never_started_aborts_the_record_but_is_not_final() { + assert!(write_definitely_not_applied( + &ErrorCode::ExpiredBeforeExecution + )); + assert!(!refusal_is_final(&ErrorCode::ExpiredBeforeExecution)); + } + /// The asymmetry that keeps this safe: an ambiguous outcome must never /// produce an abort, because the write it would erase may have landed. #[test] @@ -222,4 +255,20 @@ mod tests { })); assert!(!write_definitely_not_applied(&ErrorCode::DuplicateWrite)); } + + #[test] + fn a_constraint_verdict_is_final_and_a_retryable_one_is_not() { + assert!(refusal_is_final(&ErrorCode::RejectedPrevalidation { + reason: "sub-record does not decode".into(), + })); + assert!(!refusal_is_final(&ErrorCode::RetryableRefusal { + reason: "install rolled back".into(), + })); + assert!(!refusal_is_final(&ErrorCode::DispatchCapacity { + reason: "core 0 queue is full".into(), + })); + assert!(!refusal_is_final(&ErrorCode::Internal { + detail: "io_uring".into(), + })); + } } diff --git a/nodedb/src/control/server/exchange/all_cores/dispatch.rs b/nodedb/src/control/server/exchange/all_cores/dispatch.rs index 223a56739..37112a77b 100644 --- a/nodedb/src/control/server/exchange/all_cores/dispatch.rs +++ b/nodedb/src/control/server/exchange/all_cores/dispatch.rs @@ -168,7 +168,8 @@ pub(crate) async fn execute_plan_all_local_cores( | MetaOp::RollbackToSavepoint { .. } | MetaOp::RecordCalvinWriteVersions { .. } | MetaOp::CalvinFlush { .. } - | MetaOp::CalvinDrop { .. } => { + | MetaOp::CalvinDrop { .. } + | MetaOp::ApplyTransactionRedo { .. } => { generic_gather(state, tenant_id, database_id, plan, trace_id, txn_id).await } }, @@ -194,7 +195,17 @@ pub(crate) async fn execute_plan_all_local_cores( } } -/// Generic gather path: delegate to [`gather_all_cores`] and wrap. +/// Generic gather path. +/// +/// A plan on one collection that is not cluster-partitioned lives wholly on +/// the core that owns the collection's vShard. It runs there alone, and its +/// payload returns verbatim: the shape a single core produces. That shape is +/// not always a msgpack array. A KV point read answers with the stored value +/// itself, and wrapping it as an array element hands the requesting node a +/// different value than a local read returns. +/// +/// Every other plan fans across all local cores, and their row arrays merge +/// into one. async fn generic_gather( state: &SharedState, tenant_id: TenantId, @@ -204,9 +215,31 @@ async fn generic_gather( txn_id: Option, ) -> crate::Result { use crate::control::server::exchange::gather::gather_all_cores; + use crate::control::server::exchange::owning_core::dispatch_single_owning_core; // Forwarded `txn_id`, if any, is stamped on each core's request so a // transactional read honours its staged overlay. Inert when `None`. + if !nodedb_physical::physical_plan::plan_contains_cluster_partitioned_leaf(&plan) + && let Some(collection) = plan.collection() + { + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?.vshard(); + let resp = dispatch_single_owning_core( + state, + tenant_id, + database_id, + plan, + vshard_id, + trace_id, + txn_id, + ) + .await?; + return Ok(NodeLevelResult { + payload: resp.payload.to_vec(), + watermark_lsn: resp.watermark_lsn, + read_version_lsn: resp.read_version_lsn, + }); + } let outcome = gather_all_cores(state, tenant_id, database_id, plan, trace_id, txn_id).await?; Ok(NodeLevelResult { payload: outcome.merged_array, diff --git a/nodedb/src/control/server/exchange/all_cores/fanout.rs b/nodedb/src/control/server/exchange/all_cores/fanout.rs index 31522bac3..117b8b76f 100644 --- a/nodedb/src/control/server/exchange/all_cores/fanout.rs +++ b/nodedb/src/control/server/exchange/all_cores/fanout.rs @@ -14,11 +14,15 @@ use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId, TraceId, TxnId}; use nodedb_physical::physical_plan::{GraphOp, PhysicalPlan}; +/// One core's outcome: its response, `None` for a `NotFound` refusal, or the +/// typed error that stopped it. +type CoreOutcome = crate::Result>; + /// Shared per-core fan for a BSP/WCC superstep plan: dispatch to every local /// core, gather bounded responses, drop `NotFound`/empty-CSR cores. /// -/// Must scope `owned_vshards` to `vshard % num_cores == core_id`, or every core -/// claims sibling-homed nodes in its local CSR, duplicating them in the merge. +/// A core that fails is dropped while any other core answers. The call fails +/// only when no core answers. pub(super) async fn gather_graph_op_all_cores( state: &SharedState, tenant_id: TenantId, @@ -28,6 +32,69 @@ pub(super) async fn gather_graph_op_all_cores( txn_id: Option, label: &'static str, ) -> crate::Result> { + let outcomes = + dispatch_all_cores(state, tenant_id, database_id, plan, trace_id, txn_id, label).await?; + let mut out = Vec::with_capacity(outcomes.len()); + // First error seen across cores, kept as a TYPED error: a core cut short by + // the statement's deadline reports the deadline, and a constraint refusal + // keeps its own SQLSTATE. + let mut first_error: Option = None; + for outcome in outcomes { + match outcome { + Ok(Some(resp)) => out.push(resp), + Ok(None) => {} + Err(error) => { + if first_error.is_none() { + first_error = Some(error); + } + } + } + } + if out.is_empty() + && let Some(error) = first_error + { + return Err(error); + } + Ok(out) +} + +/// Fan `plan` to every local core and require every core to answer. +/// +/// Each core holds only the state its own vShards home to. A merge that +/// dropped a failed core returns that core's state as absent, so the first +/// core error fails the whole call. A `NotFound` refusal contributes nothing. +pub(super) async fn gather_every_core( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + plan: PhysicalPlan, + trace_id: TraceId, + label: &'static str, +) -> crate::Result> { + let outcomes = + dispatch_all_cores(state, tenant_id, database_id, plan, trace_id, None, label).await?; + let mut out = Vec::with_capacity(outcomes.len()); + for outcome in outcomes { + if let Some(resp) = outcome? { + out.push(resp); + } + } + Ok(out) +} + +/// Dispatch `plan` to every local core and collect each core's outcome. +/// +/// Must scope `owned_vshards` to `vshard % num_cores == core_id`, or every core +/// claims sibling-homed nodes in its local CSR, duplicating them in the merge. +async fn dispatch_all_cores( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + plan: PhysicalPlan, + trace_id: TraceId, + txn_id: Option, + label: &'static str, +) -> crate::Result> { // Shared broadcast call counter (parity with gather_all_cores). crate::control::server::broadcast::broadcast_call_count_increment(); @@ -103,9 +170,9 @@ pub(super) async fn gather_graph_op_all_cores( .into_iter() .map(|(core_id, request_id, mut rx)| async move { let context = format!("{label} gather on core {core_id}"); - crate::control::server::dispatch_utils::collect_under_deadline( + crate::control::local_dispatch::collect_under_deadline( &mut rx, - crate::control::server::dispatch_utils::DeadlineCollect { + crate::control::local_dispatch::DeadlineCollect { request_id, deadline, max_result_bytes, @@ -117,43 +184,16 @@ pub(super) async fn gather_graph_op_all_cores( let results: Vec> = join_all(response_futures).await; - let mut out = Vec::with_capacity(num_cores); - // First error seen across cores, kept as a TYPED error: a core cut short by - // the statement's deadline reports the deadline, and a constraint refusal - // keeps its own SQLSTATE. Stringifying either one made both read as - // internal. - let mut first_error: Option = None; - - for result in results { - let resp = match result { - Ok(r) => r, - Err(error) => { - if first_error.is_none() { - first_error = Some(error); - } - continue; - } - }; - - if resp.status == Status::Error { - // `NotFound` is an empty CSR slice on this core, not an error. - if let Err(error) = - crate::control::server::dispatch_utils::reject_data_plane_error(&resp) - && first_error.is_none() - { - first_error = Some(error); + Ok(results + .into_iter() + .map(|result| { + let resp = result?; + if resp.status == Status::Error { + // `NotFound` is an empty slice on this core, not an error. + crate::control::local_dispatch::reject_data_plane_error(&resp)?; + return Ok(None); } - continue; - } - - out.push(resp); - } - - if out.is_empty() - && let Some(error) = first_error - { - return Err(error); - } - - Ok(out) + Ok(Some(resp)) + }) + .collect()) } diff --git a/nodedb/src/control/server/exchange/all_cores/mod.rs b/nodedb/src/control/server/exchange/all_cores/mod.rs index 7ec345020..f24c3a374 100644 --- a/nodedb/src/control/server/exchange/all_cores/mod.rs +++ b/nodedb/src/control/server/exchange/all_cores/mod.rs @@ -30,3 +30,4 @@ mod wcc; pub use dispatch::NodeLevelResult; pub(crate) use dispatch::execute_plan_all_local_cores; +pub(crate) use snapshot::snapshot_tenant_on_local_cores; diff --git a/nodedb/src/control/server/exchange/all_cores/snapshot.rs b/nodedb/src/control/server/exchange/all_cores/snapshot.rs index ffa8ba8b9..cc9184945 100644 --- a/nodedb/src/control/server/exchange/all_cores/snapshot.rs +++ b/nodedb/src/control/server/exchange/all_cores/snapshot.rs @@ -3,12 +3,44 @@ //! Tenant-snapshot fan: dispatch `CreateTenantSnapshot` across all local cores //! and merge the per-core partial `TenantDataSnapshot` into one blob. +use std::time::Duration; + use crate::control::state::SharedState; use crate::types::{DatabaseId, Lsn, TenantId, TraceId}; -use nodedb_physical::physical_plan::PhysicalPlan; +use nodedb_physical::physical_plan::{MetaOp, PhysicalPlan}; use super::dispatch::NodeLevelResult; -use super::fanout::gather_graph_op_all_cores; +use super::fanout::gather_every_core; + +/// Snapshot `tenant_id` in `database_id` on every local core and return the +/// one merged `TenantDataSnapshot` blob. +/// +/// Every local caller that snapshots a tenant goes through this function. Each +/// core stores only the collections its vShards home to, so a snapshot taken +/// on one core leaves out every collection homed on another core. +pub(crate) async fn snapshot_tenant_on_local_cores( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + timeout: Duration, +) -> crate::Result> { + let plan = PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { + tenant_id: tenant_id.as_u64(), + cut_watermark: None, + }); + let fan = + fan_tenant_snapshot_all_cores(state, tenant_id, database_id, plan, TraceId::generate()); + match tokio::time::timeout(timeout, fan).await { + Ok(result) => result.map(|merged| merged.payload), + Err(_) => Err(crate::Error::Dispatch { + detail: format!( + "tenant snapshot of tenant {} in database {} did not finish within {timeout:?}", + tenant_id.as_u64(), + database_id.as_u64() + ), + }), + } +} /// Tenant-snapshot fan: dispatch `CreateTenantSnapshot` across all local cores, /// decode each core's partial [`TenantDataSnapshot`], merge them by field @@ -16,11 +48,10 @@ use super::fanout::gather_graph_op_all_cores; /// /// Each core scans only the engine state for the vShards homed on that core, so /// the per-core snapshots cover DISJOINT key sets — concatenating every `Vec` -/// field requires no dedup, exactly like the BSP/WCC superstep merges. The -/// result is byte-shape-identical to the local `snapshot_self` path's single -/// `TenantDataSnapshot` blob, so backup sections from the local and remote -/// transports converge on the same `from_msgpack::` decode. -/// At 1 core/node this yields the lone core's snapshot unchanged. +/// field requires no dedup, exactly like the BSP/WCC superstep merges. Every +/// core must answer: a core that fails would leave its collections out of the +/// snapshot, so its error fails the snapshot. At 1 core/node this yields the +/// lone core's snapshot unchanged. pub(super) async fn fan_tenant_snapshot_all_cores( state: &SharedState, tenant_id: TenantId, @@ -30,13 +61,12 @@ pub(super) async fn fan_tenant_snapshot_all_cores( ) -> crate::Result { use crate::types::TenantDataSnapshot; - let responses = gather_graph_op_all_cores( + let responses = gather_every_core( state, tenant_id, database_id, plan, trace_id, - None, "tenant-snapshot", ) .await?; @@ -74,6 +104,9 @@ pub(super) async fn fan_tenant_snapshot_all_cores( index_configs, surrogate_pk, tenant_edges, + group_write_marks, + documents_versioned, + indexes_versioned, } = part; merged.documents.extend(documents); merged.indexes.extend(indexes); @@ -89,6 +122,9 @@ pub(super) async fn fan_tenant_snapshot_all_cores( merged.index_configs.extend(index_configs); merged.surrogate_pk.extend(surrogate_pk); merged.tenant_edges.extend(tenant_edges); + merged.group_write_marks.extend(group_write_marks); + merged.documents_versioned.extend(documents_versioned); + merged.indexes_versioned.extend(indexes_versioned); } let payload = zerompk::to_msgpack_vec(&merged).map_err(|e| crate::Error::Serialization { diff --git a/nodedb/src/control/server/exchange/gather.rs b/nodedb/src/control/server/exchange/gather.rs index 540afd36f..07b57d9b8 100644 --- a/nodedb/src/control/server/exchange/gather.rs +++ b/nodedb/src/control/server/exchange/gather.rs @@ -60,7 +60,7 @@ pub(crate) fn eager_dispatch_to_all_cores( Vec<( usize, crate::types::RequestId, - tokio::sync::mpsc::Receiver, + crate::control::ResponseReceiver, )>, > { // Every core in this fan-out belongs to ONE statement, so all of them @@ -177,9 +177,9 @@ pub(crate) async fn gather_all_cores( .into_iter() .map(|(core_id, request_id, mut rx)| async move { let context = format!("gather on core {core_id}"); - let result = crate::control::server::dispatch_utils::collect_under_deadline( + let result = crate::control::local_dispatch::collect_under_deadline( &mut rx, - crate::control::server::dispatch_utils::DeadlineCollect { + crate::control::local_dispatch::DeadlineCollect { request_id, deadline, max_result_bytes, @@ -218,7 +218,7 @@ pub(crate) async fn gather_all_cores( }; if resp.status == Status::Error { - if let Err(e) = crate::control::server::dispatch_utils::reject_data_plane_error(&resp) + if let Err(e) = crate::control::local_dispatch::reject_data_plane_error(&resp) && first_error.is_none() { first_error = Some(e); @@ -340,8 +340,8 @@ pub(crate) fn gather_all_cores_stream( /// timeseries, spatial, vector, text) /// /// Standard collections are *single-vShard-homed*: all rows for a collection -/// live on exactly one vShard determined by `vshard_for_collection(database_id, -/// &name)`. The data-plane scan is **not** vshard-scoped, so broadcasting the +/// live on exactly one vShard determined by `vshard_for_collection` over the +/// collection's canonical key. The data-plane scan is **not** vshard-scoped, so broadcasting the /// plan to every vShard via `Exchange{Gather}` causes the owning node to return /// the full collection once per route that lands on it — 1 024× duplication. /// diff --git a/nodedb/src/control/server/exchange/mod.rs b/nodedb/src/control/server/exchange/mod.rs index c9484c41c..09156cc84 100644 --- a/nodedb/src/control/server/exchange/mod.rs +++ b/nodedb/src/control/server/exchange/mod.rs @@ -17,7 +17,7 @@ pub mod response; pub mod streamable; pub use all_cores::NodeLevelResult; -pub(crate) use all_cores::execute_plan_all_local_cores; +pub(crate) use all_cores::{execute_plan_all_local_cores, snapshot_tenant_on_local_cores}; pub(crate) use gather::gather_all_cores; pub use gather::{GatherOutcome, finalize_aggregate}; pub use resolve::{ diff --git a/nodedb/src/control/server/exchange/owning_core.rs b/nodedb/src/control/server/exchange/owning_core.rs index 2727e77a0..698e9dddd 100644 --- a/nodedb/src/control/server/exchange/owning_core.rs +++ b/nodedb/src/control/server/exchange/owning_core.rs @@ -15,9 +15,8 @@ //! minus the empty cores' spurious contributions. use crate::bridge::envelope::PhysicalPlan; -use crate::control::server::dispatch_utils::{ - dispatch_to_data_plane_with_txn, reject_data_plane_error, -}; +use crate::control::local_dispatch::reject_data_plane_error; +use crate::control::server::dispatch_utils::dispatch_to_data_plane_with_txn; use crate::control::server::payload_merge::{encode_msgpack_array, extract_msgpack_elements}; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId, TraceId, TxnId, VShardId}; @@ -51,7 +50,8 @@ pub async fn gather_single_node( return gather_all_cores(state, tenant_id, database_id, plan, trace_id, txn_id).await; } if let Some(collection) = plan.collection() { - let vshard_id = VShardId::from_collection_in_database(database_id, collection); + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?.vshard(); return gather_single_owning_core( state, tenant_id, @@ -69,8 +69,8 @@ pub async fn gather_single_node( /// Dispatch `plan` to the single Data-Plane core that owns `vshard_id` and /// gather the one bounded response into a [`GatherOutcome`]. /// -/// `vshard_id` is the collection's owning vShard -/// (`VShardId::from_collection_in_database(database_id, collection)`); the +/// `vshard_id` is the collection's owning vShard (the vShard of its +/// canonical `CollectionKey`); the /// dispatcher's `VShardRouter` resolves it to the one core holding the /// collection's rows. /// @@ -78,7 +78,7 @@ pub async fn gather_single_node( /// `read_version_lsn` and exactly one `shard_watermarks` entry keyed to the /// collection's vShard — matching the cluster `dispatch_local` path so an /// in-transaction read records the same OCC read-set entry the write-set uses -/// (writes home to the same `from_collection_in_database` vShard). Aggregate +/// (writes home to the same `CollectionKey` vShard). Aggregate /// finalization (`finalize_aggregate`) is a passthrough over the merged array, /// so one complete aggregate row in yields one row out. pub async fn gather_single_owning_core( @@ -90,6 +90,40 @@ pub async fn gather_single_owning_core( trace_id: TraceId, txn_id: Option, ) -> crate::Result { + let resp = dispatch_single_owning_core( + state, + tenant_id, + database_id, + plan, + vshard_id, + trace_id, + txn_id, + ) + .await?; + let payload_bytes: &[u8] = resp.payload.as_ref(); + let all_elements = extract_msgpack_elements(payload_bytes); + let merged_array = encode_msgpack_array(&all_elements); + + Ok(GatherOutcome { + raw: payload_bytes.to_vec(), + merged_array, + watermark_lsn: resp.watermark_lsn, + read_version_lsn: resp.read_version_lsn, + shard_watermarks: vec![(vshard_id, resp.watermark_lsn)], + }) +} + +/// Dispatch `plan` to the single Data-Plane core that owns `vshard_id` and +/// return that core's response, its payload in the shape the core produced. +pub async fn dispatch_single_owning_core( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + plan: PhysicalPlan, + vshard_id: VShardId, + trace_id: TraceId, + txn_id: Option, +) -> crate::Result { // `Box::pin` breaks an async-fn recursion cycle: `dispatch_to_data_plane_*` // re-enters `resolve_exchange_in_plan`. The plan handed here is the bare, // Exchange-free child of the resolved Gather, so the re-entrant resolve is a @@ -109,16 +143,5 @@ pub async fn gather_single_owning_core( // validatable) observation; any other error status surfaces with its typed // code rather than being swallowed as an empty success. reject_data_plane_error(&resp)?; - - let payload_bytes: &[u8] = resp.payload.as_ref(); - let all_elements = extract_msgpack_elements(payload_bytes); - let merged_array = encode_msgpack_array(&all_elements); - - Ok(GatherOutcome { - raw: payload_bytes.to_vec(), - merged_array, - watermark_lsn: resp.watermark_lsn, - read_version_lsn: resp.read_version_lsn, - shard_watermarks: vec![(vshard_id, resp.watermark_lsn)], - }) + Ok(resp) } diff --git a/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs b/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs index 8230cffec..b71d33e08 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs @@ -21,7 +21,6 @@ use crate::data::executor::response_codec::{ flatten_hybrid_hits_to_relational_rows, flatten_to_relational_rows, flatten_vector_hits_to_relational_rows, }; -use crate::types::VShardId; use super::dispatch::{ResolveCtx, resolve_exchange}; use super::entry::Resolved; @@ -214,7 +213,7 @@ pub(super) async fn materialize_child_rows( tenant_id, database_id, child, - VShardId::from_collection_in_database(database_id, ""), + nodedb_types::CollectionKey::from_bare(database_id, "").vshard(), trace_id, txn_id, ) diff --git a/nodedb/src/control/server/exchange/resolve/mod.rs b/nodedb/src/control/server/exchange/resolve/mod.rs index 0d397fbb7..db5f35366 100644 --- a/nodedb/src/control/server/exchange/resolve/mod.rs +++ b/nodedb/src/control/server/exchange/resolve/mod.rs @@ -40,4 +40,3 @@ mod shuffle_aggregate; pub use capture::DistributedReadCapture; pub use exchange::{Resolved, resolve_and_materialize, resolve_exchange_in_plan}; -pub(crate) use peers::register_peers_from_topology; diff --git a/nodedb/src/control/server/exchange/resolve/peers.rs b/nodedb/src/control/server/exchange/resolve/peers.rs index 8c19e4bca..6418081e7 100644 --- a/nodedb/src/control/server/exchange/resolve/peers.rs +++ b/nodedb/src/control/server/exchange/resolve/peers.rs @@ -4,10 +4,10 @@ //! resolvers (`shuffle` = shuffle-join, `shuffle_aggregate` = shuffle GROUP BY). //! //! Both resolvers fan producer/consumer RPCs across the cluster and need the -//! same primitives: resolve a collection's owner nodes, register peer addresses -//! from the live topology before dispatch, and count the cluster's data nodes -//! for the default partition count. They live here (rather than duplicated in -//! each resolver) so the two paths share one implementation. +//! same primitives: resolve a collection's owner nodes, count the cluster's +//! data nodes for the default partition count, and send a produce request. +//! They live here (rather than duplicated in each resolver) so the two paths +//! share one implementation. use std::collections::BTreeSet; @@ -15,10 +15,10 @@ use nodedb_cluster::{ METADATA_GROUP_ID, RaftRpc, RoutingTable, ShuffleProduceRequest, ShuffleProduceResponse, }; -use crate::control::state::SharedState; -use crate::types::{DatabaseId, VShardId}; +use crate::types::DatabaseId; -/// Producer nodes that own `collection`'s data: resolve its vShard → owning +/// Producer nodes that own `collection`'s data. `collection` is the plan's +/// database-qualified name. Resolve its canonical key's vShard → owning /// group → leader. A user collection is single-vShard-homed, so this is one /// node; returned as a deduped sorted vec for generality. pub(super) fn producer_nodes( @@ -26,7 +26,9 @@ pub(super) fn producer_nodes( database_id: DatabaseId, collection: &str, ) -> crate::Result> { - let vshard = VShardId::from_collection_in_database(database_id, collection).as_u32(); + let vshard = nodedb_types::CollectionKey::from_qualified_str(database_id, collection)? + .vshard() + .as_u32(); let group = routing .group_for_vshard(vshard) .map_err(|e| crate::Error::Internal { @@ -59,34 +61,6 @@ pub(super) fn distinct_data_node_count(routing: &RoutingTable) -> usize { nodes.len() } -/// Register each target node's address with the transport from the live cluster -/// topology (idempotent). Makes the shuffle fan-out robust to a peer the -/// transport has not warmed yet — without it `send_rpc` to an unregistered (but -/// topology-known) node fails with `NodeUnreachable`. Self IS registered too: -/// when this coordinator also owns one of the sides it dispatches that -/// producer/consumer to itself via `send_rpc`, which loops back through the local -/// QUIC endpoint and runs the same handler (an extra local hop, functionally -/// correct). Missing topology / address for a node is left alone so the -/// subsequent `send_rpc` surfaces the typed `NodeUnreachable` rather than this -/// silently masking it. -pub(crate) fn register_peers_from_topology( - state: &SharedState, - transport: &nodedb_cluster::NexarTransport, - nodes: &BTreeSet, -) { - let Some(topology) = state.cluster_topology.as_ref() else { - return; - }; - let topo = topology.read().unwrap_or_else(|p| p.into_inner()); - for &node in nodes { - if let Some(info) = topo.get_node(node) - && let Some(addr) = info.socket_addr() - { - transport.register_peer(node, addr); - } - } -} - /// Send one `ShuffleProduceRequest` and map the reply / RPC error to a typed /// coordinator error, returning the producer's observed per-collection /// read-version LSN on a clean produce. Fail-fast: a producer-reported terminal diff --git a/nodedb/src/control/server/exchange/resolve/shuffle.rs b/nodedb/src/control/server/exchange/resolve/shuffle.rs index 9a56c69a8..57c06f438 100644 --- a/nodedb/src/control/server/exchange/resolve/shuffle.rs +++ b/nodedb/src/control/server/exchange/resolve/shuffle.rs @@ -38,6 +38,7 @@ use nodedb_cluster::{ use nodedb_physical::physical_plan::wire as plan_wire; use nodedb_physical::physical_plan::{PhysicalPlan, QueryOp}; +use crate::control::cluster::warm_peers::register_peers_from_topology; use crate::control::server::exchange::full_scan::{ScanSide, full_scan_plan_for_collection}; use crate::control::server::exchange::gather::outcome_to_response; use crate::control::server::payload_merge::merge_msgpack_arrays; @@ -46,9 +47,7 @@ use crate::types::{DatabaseId, Lsn, TenantId, TraceId}; use super::capture::DistributedReadCapture; use super::exchange::Resolved; -use super::peers::{ - distinct_data_node_count, producer_nodes, register_peers_from_topology, send_produce, -}; +use super::peers::{distinct_data_node_count, producer_nodes, send_produce}; /// Orchestrate a distributed shuffle hash join. /// diff --git a/nodedb/src/control/server/exchange/resolve/shuffle_aggregate.rs b/nodedb/src/control/server/exchange/resolve/shuffle_aggregate.rs index dc805a441..a40d9b134 100644 --- a/nodedb/src/control/server/exchange/resolve/shuffle_aggregate.rs +++ b/nodedb/src/control/server/exchange/resolve/shuffle_aggregate.rs @@ -46,15 +46,14 @@ use nodedb_cluster::{ use nodedb_physical::physical_plan::wire as plan_wire; use nodedb_physical::physical_plan::{PhysicalPlan, QueryOp}; +use crate::control::cluster::warm_peers::register_peers_from_topology; use crate::control::server::exchange::gather::outcome_to_response; use crate::control::server::payload_merge::{encode_msgpack_array, extract_msgpack_elements}; use crate::control::state::SharedState; use crate::types::{DatabaseId, Lsn, TenantId, TraceId}; use super::exchange::Resolved; -use super::peers::{ - distinct_data_node_count, producer_nodes, register_peers_from_topology, send_produce, -}; +use super::peers::{distinct_data_node_count, producer_nodes, send_produce}; /// Orchestrate a distributed shuffle GROUP BY aggregate. /// diff --git a/nodedb/src/control/server/graph_dispatch/match_broadcast.rs b/nodedb/src/control/server/graph_dispatch/match_broadcast.rs index ecf264c57..86a225ac1 100644 --- a/nodedb/src/control/server/graph_dispatch/match_broadcast.rs +++ b/nodedb/src/control/server/graph_dispatch/match_broadcast.rs @@ -224,9 +224,9 @@ pub async fn broadcast_match_to_all_cores( .into_iter() .map(|(core_id, request_id, mut rx)| async move { let context = format!("match gather on core {core_id}"); - crate::control::server::dispatch_utils::collect_under_deadline( + crate::control::local_dispatch::collect_under_deadline( &mut rx, - crate::control::server::dispatch_utils::DeadlineCollect { + crate::control::local_dispatch::DeadlineCollect { request_id, deadline, max_result_bytes, @@ -260,8 +260,7 @@ pub async fn broadcast_match_to_all_cores( if resp.status == Status::Error { // `NotFound` is an empty CSR slice on this core, not an error. - if let Err(error) = - crate::control::server::dispatch_utils::reject_data_plane_error(&resp) + if let Err(error) = crate::control::local_dispatch::reject_data_plane_error(&resp) && first_error.is_none() { first_error = Some(error); diff --git a/nodedb/src/control/server/http/auth.rs b/nodedb/src/control/server/http/auth.rs index a40e3f2ef..d45ace268 100644 --- a/nodedb/src/control/server/http/auth.rs +++ b/nodedb/src/control/server/http/auth.rs @@ -320,6 +320,9 @@ pub enum ApiError { status: StatusCode, message: String, code: nodedb_types::error::ErrorCode, + /// The typed error that caused this one, such as the Data-Plane + /// refusal behind a DDL phase failure. `None` when there is none. + cause: Option>, }, } @@ -343,8 +346,13 @@ impl IntoResponse for ApiError { status, message, code, + cause, } => { let body = HttpError::with_code(message, code.to_string()); + let body = match cause { + Some(cause) => body.caused_by(&cause), + None => body, + }; (status, axum::Json(body)).into_response() } other => { @@ -424,22 +432,19 @@ impl FromRequestParts for ResolvedAuth { } } +/// The status and message come from the one gateway mapping, +/// [`GatewayErrorMap::to_http`](crate::control::gateway::GatewayErrorMap::to_http). +/// A rate refusal also carries its `Retry-After` hint. impl From for ApiError { fn from(e: crate::Error) -> Self { - match &e { - crate::Error::RejectedAuthz { .. } => Self::Forbidden(e.to_string()), - crate::Error::RateExceeded { retry_after_ms, .. } => Self::RateLimited { - message: e.to_string(), + let (status, message) = crate::control::gateway::GatewayErrorMap::to_http(&e); + if let crate::Error::RateExceeded { retry_after_ms, .. } = &e { + return Self::RateLimited { + message, retry_after_secs: retry_after_ms.div_ceil(1000).max(1), - }, - crate::Error::BadRequest { .. } - | crate::Error::PlanError { .. } - | crate::Error::Config { .. } => Self::BadRequest(e.to_string()), - crate::Error::CollectionNotFound { .. } | crate::Error::DocumentNotFound { .. } => { - Self::BadRequest(e.to_string()) - } - _ => Self::Internal(e.to_string()), + }; } + Self::HttpStatus(status, message) } } @@ -538,6 +543,90 @@ mod tests { assert_authorization_fields_unchanged(&context, &result); } + fn api_status(error: crate::Error) -> StatusCode { + ApiError::from(error).into_response().status() + } + + /// Every error takes the status the gateway mapping gives it. + #[test] + fn api_errors_take_the_gateway_status() { + use crate::types::{RequestId, VShardId}; + + let cases = [ + ( + crate::Error::DataPlane(crate::bridge::envelope::ErrorCode::NotFound), + StatusCode::NOT_FOUND, + ), + ( + crate::Error::DeadlineExceeded { + request_id: RequestId::new(1), + }, + StatusCode::GATEWAY_TIMEOUT, + ), + ( + crate::Error::NotLeader { + vshard_id: VShardId::new(1), + leader_node: 2, + leader_addr: "10.0.0.1:9000".into(), + }, + StatusCode::SERVICE_UNAVAILABLE, + ), + ( + crate::Error::ConflictRetry { + collection: "orders".into(), + document_id: "o1".into(), + }, + StatusCode::CONFLICT, + ), + ( + crate::Error::CollectionNotFound { + tenant_id: TenantId::new(1), + collection: "orders".into(), + }, + StatusCode::NOT_FOUND, + ), + ( + crate::Error::RejectedAuthz { + tenant_id: TenantId::new(1), + resource: "orders".into(), + }, + StatusCode::FORBIDDEN, + ), + ( + crate::Error::BadRequest { + detail: "bad".into(), + }, + StatusCode::BAD_REQUEST, + ), + ]; + for (error, expected) in cases { + let label = format!("{error:?}"); + let gateway = crate::control::gateway::GatewayErrorMap::to_http(&error).0; + let status = api_status(error); + assert_eq!(status, expected, "{label}"); + assert_eq!(status.as_u16(), gateway, "{label}"); + } + } + + /// A rate refusal keeps its 429 and carries its `Retry-After` hint. + #[test] + fn a_rate_refusal_carries_retry_after() { + let response = ApiError::from(crate::Error::RateExceeded { + gate: "write".into(), + detail: "over budget".into(), + retry_after_ms: 1500, + }) + .into_response(); + assert_eq!(response.status(), StatusCode::TOO_MANY_REQUESTS); + assert_eq!( + response + .headers() + .get("Retry-After") + .and_then(|value| value.to_str().ok()), + Some("2") + ); + } + #[test] fn x_on_deny_malformed_value_is_ignored() { let context = auth_context(); diff --git a/nodedb/src/control/server/http/mod.rs b/nodedb/src/control/server/http/mod.rs index e533b4cb8..3cfe7f079 100644 --- a/nodedb/src/control/server/http/mod.rs +++ b/nodedb/src/control/server/http/mod.rs @@ -6,6 +6,7 @@ pub mod peer; pub(crate) mod rate_limit_headers; pub mod routes; pub mod server; +pub mod startup_gate; pub(crate) mod tls_accept; pub mod transport; pub mod types; diff --git a/nodedb/src/control/server/http/routes/crdt.rs b/nodedb/src/control/server/http/routes/crdt.rs index 511aebcc5..f45ed92bf 100644 --- a/nodedb/src/control/server/http/routes/crdt.rs +++ b/nodedb/src/control/server/http/routes/crdt.rs @@ -100,12 +100,11 @@ pub async fn crdt_apply( .shared .surrogate_assigner .assign( - crate::types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(crate::types::DatabaseId::DEFAULT, &collection), identity.tenant_id, - &collection, body.doc_id.as_bytes(), ) - .map_err(|e| ApiError::Internal(e.to_string()))?; + .map_err(ApiError::from)?; let plan = PhysicalPlan::Crdt(CrdtOp::Apply { collection: nodedb_types::QualifiedCollection::new( @@ -125,10 +124,11 @@ pub async fn crdt_apply( let task = PhysicalTask { tenant_id: identity.tenant_id, - vshard_id: crate::types::VShardId::from_collection_in_database( + vshard_id: nodedb_types::CollectionKey::from_bare( crate::types::DatabaseId::DEFAULT, &collection, - ), + ) + .vshard(), database_id: crate::types::DatabaseId::DEFAULT, plan, post_set_op: PostSetOp::None, diff --git a/nodedb/src/control/server/http/routes/health.rs b/nodedb/src/control/server/http/routes/health.rs index 2dd592efa..a348bbaa7 100644 --- a/nodedb/src/control/server/http/routes/health.rs +++ b/nodedb/src/control/server/http/routes/health.rs @@ -6,7 +6,7 @@ //! |-------------------|--------|-----------------------------|---------------| //! | `/healthz` | GET | Ready to serve traffic | readiness | //! | `/health/live` | GET | Process alive (always 200) | liveness | -//! | `/health/ready` | GET | WAL recovered | readiness alt | +//! | `/health/ready` | GET | WAL recovered, serving | readiness alt | //! | `/health/drain` | POST | Trigger graceful drain | preStop hook | use std::sync::atomic::Ordering; @@ -34,13 +34,13 @@ pub async fn live() -> impl IntoResponse { /// GET /healthz — k8s-style readiness probe. /// -/// Returns `200 OK` when the node has reached `GatewayEnable`, is +/// Returns `200 OK` when the node has reached `Serving`, is /// serving traffic, is NOT draining/decommissioned, and — on a node that /// runs a Calvin sequencer — can actually sequence a cross-shard write. /// Returns `503 Service Unavailable` otherwise. /// /// Every condition is evaluated live on each call, not latched: sequencer -/// leadership and the epoch seed can both be lost long after `GatewayEnable`. +/// leadership and the epoch seed can both be lost long after `Serving`. pub async fn healthz(State(state): State) -> impl IntoResponse { // The coordinator signals this canonical watch before progressing drain // phases, so readiness must fail immediately even before lifecycle state @@ -109,6 +109,41 @@ pub async fn healthz(State(state): State) -> impl IntoResponse { return (StatusCode::SERVICE_UNAVAILABLE, axum::Json(body)); } + // A halted Calvin scheduler holds one vShard's sequenced txns unapplied. + // The node serves everything else, so it reports degraded, like a halted + // sequencer. + if let Some(halt) = state.shared.sequencer_halt.apply_halt().report() { + let body = json!({ + "status": "degraded", + "reason": "calvin_apply_halted", + "node_id": state.shared.node_id, + "vshard_id": halt.vshard_id, + "epoch": halt.epoch, + "position": halt.position, + "halt_reason": halt.reason, + "step": halt.step, + "error": halt.error, + }); + return (StatusCode::SERVICE_UNAVAILABLE, axum::Json(body)); + } + + // A fail-stopped core refuses every request routed to it: its state is + // unknown until restart. The other cores serve, so the node is degraded. + if let Some(stops) = state + .shared + .system_metrics + .as_ref() + .map(|metrics| &metrics.core_fail_stops) + && let Some(report) = stops.report() + { + let (status, mut body) = crate::control::metrics::system::core_fail_stop::to_http_response( + report, + stops.stopped_cores(), + ); + body["node_id"] = json!(state.shared.node_id); + return (status, axum::Json(body)); + } + // A core that stops completing event-loop iterations panics nothing, so // the per-core panic watchdog stays quiet and every other check above // still passes. Fail readiness and name the cores: work routed to a @@ -121,8 +156,27 @@ pub async fn healthz(State(state): State) -> impl IntoResponse { return (status, axum::Json(body)); } + // A write window open past the outcome-floor bound holds the floor, so + // no checkpoint on this node advances past it. The node serves, so it + // reports degraded. + if let Some(body) = outcome_floor_stuck_body(&state, outcome_floor_bound(&state)) { + return (StatusCode::SERVICE_UNAVAILABLE, axum::Json(body)); + } + let health = crate::control::startup::health::observe(&state.shared.startup); - let (status, body) = crate::control::startup::health::to_http_response(&health); + let (status, mut body) = crate::control::startup::health::to_http_response(&health); + // A held window keeps the outcome floor below it by design, so it never + // degrades readiness. The count shows how many restart replay will reach. + body["held_windows"] = json!(state.shared.outcome_floor.held_windows()); + // A node that lost its authorization lease refuses permission-checked + // statements until it renews, and serves everything else. Checked once + // the startup gate is green: boot holds the gateway until the first lease. + if status == StatusCode::OK + && let Some(body) = lease_invalid_body(&state) + { + return (StatusCode::SERVICE_UNAVAILABLE, axum::Json(body)); + } + body["authorization_lease"] = json!(lease_label(&state)); // Checked only once the startup gate is otherwise green, so a node still // advancing through phases keeps reporting the phase it is stuck in. if status == StatusCode::OK @@ -138,6 +192,76 @@ pub async fn healthz(State(state): State) -> impl IntoResponse { (status, axum::Json(body)) } +/// The lease state `/healthz` reports: `valid`, `invalid`, `sole_voter` for +/// a pinned lease, or `not_required` on a single node without a cluster. +fn lease_label(state: &AppState) -> &'static str { + use crate::control::security::auth_lease::{LeaseStatus, lease_status}; + match lease_status(&state.shared, std::time::Instant::now()) { + LeaseStatus::NotRequired => "not_required", + LeaseStatus::Valid { .. } => "valid", + LeaseStatus::SoleVoter => "sole_voter", + LeaseStatus::Invalid { .. } => "invalid", + } +} + +/// The degraded body for a node that holds no valid authorization lease, or +/// `None` when it holds one or needs none. +fn lease_invalid_body(state: &AppState) -> Option { + use crate::control::security::auth_lease::{LeaseStatus, lease_status}; + match lease_status(&state.shared, std::time::Instant::now()) { + LeaseStatus::NotRequired | LeaseStatus::Valid { .. } | LeaseStatus::SoleVoter => None, + LeaseStatus::Invalid { expired_for } => Some(json!({ + "status": "degraded", + "reason": "authorization_lease_invalid", + "detail": "this node holds no valid authorization lease; it refuses \ + permission-checked statements until it renews", + "node_id": state.shared.node_id, + "lease_expired_ms_ago": expired_for + .map(|expired| u64::try_from(expired.as_millis()).unwrap_or(u64::MAX)), + })), + } +} + +/// The degraded body for an outcome floor held past `bound` by a window that +/// is not held, or `None` when no such window exists. +fn outcome_floor_stuck_body( + state: &AppState, + bound: std::time::Duration, +) -> Option { + let floor = &state.shared.outcome_floor; + let stuck = floor.stuck(bound)?; + Some(json!({ + "status": "degraded", + "reason": "outcome_floor_stuck", + "node_id": state.shared.node_id, + "outcome_floor": stuck.floor.as_u64(), + "oldest_window_horizon": stuck.horizon.as_u64(), + "oldest_window_open_secs": stuck.open_for.as_secs(), + "open_windows": stuck.open_windows, + "leaked_windows": floor.leaked_windows(), + "held_windows": floor.held_windows(), + })) +} + +/// How long a write window can hold the outcome floor before readiness reports +/// it: the longest path from a window's open to its final outcome. +/// +/// - A statement write reaches its outcome by the longest statement deadline, +/// plus the wait the node gives a committed entry to apply. +/// - A vector index install waits two core dispatch deadlines. +/// +/// A window older than both is stuck. +fn outcome_floor_bound(state: &AppState) -> std::time::Duration { + let network = &state.shared.tuning.network; + let statement = std::time::Duration::from_secs( + network + .default_deadline_secs + .max(network.copy_deadline_secs), + ) + .saturating_add(crate::control::metadata_proposer::DEFAULT_PROPOSE_TIMEOUT); + statement.max(crate::control::catalog_entry::post_apply::vector_install_longest_core_wait()) +} + /// Why a cross-shard Calvin write would be refused on this node right now, /// or `None` when one would be accepted. /// @@ -179,16 +303,23 @@ fn sequencer_not_servable( None } -/// GET /health/ready — readiness check (WAL recovered, cores initialized). +/// GET /health/ready — readiness check: WAL recovered and the `Serving` phase reached. +/// +/// Boot recovers the WAL long before it listens on the client protocols, so +/// the WAL alone never makes the node ready. pub async fn ready(State(state): State) -> impl IntoResponse { - let wal_ready = state.shared.wal.next_lsn().as_u64() > 0; - let status = if wal_ready { + let serving = matches!( + crate::control::startup::health::observe(&state.shared.startup), + crate::control::startup::health::HealthState::Ok + ); + let ready = serving && state.shared.wal.next_lsn().as_u64() > 0; + let status = if ready { StatusCode::OK } else { StatusCode::SERVICE_UNAVAILABLE }; let body = json!({ - "status": if wal_ready { "ready" } else { "not_ready" }, + "status": if ready { "ready" } else { "not_ready" }, "wal_lsn": state.shared.wal.next_lsn().as_u64(), "node_id": state.shared.node_id, }); @@ -252,3 +383,168 @@ pub async fn drain( })), )) } + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::config::auth::AuthMode; + use crate::control::cluster::CalvinApplyHalt; + use crate::control::state::SharedState; + use crate::wal::WalManager; + + fn app_state(dir: &tempfile::TempDir) -> AppState { + let wal = Arc::new( + WalManager::open_for_testing(&dir.path().join("health.wal")).expect("open WAL"), + ); + let (dispatcher, _data_sides) = Dispatcher::new(1, 64); + let shared = SharedState::new(dispatcher, wal).expect("shared state"); + AppState { + shutdown_bus: crate::control::shutdown::ShutdownBus::new(Arc::clone(&shared.shutdown)) + .0, + query_ctx: Arc::new(crate::control::planner::context::QueryContext::for_state( + &shared, + )), + shared, + auth_mode: AuthMode::Trust, + } + } + + async fn healthz_body(state: AppState) -> (StatusCode, serde_json::Value) { + let response = healthz(State(state)).await.into_response(); + let status = response.status(); + let bytes = axum::body::to_bytes(response.into_body(), usize::MAX) + .await + .expect("read healthz body"); + let body = sonic_rs::from_slice::(&bytes).expect("healthz body is JSON"); + (status, body) + } + + #[tokio::test] + async fn healthz_reports_a_halted_calvin_scheduler_as_degraded() { + let dir = tempfile::tempdir().expect("tempdir"); + let state = app_state(&dir); + state + .shared + .sequencer_halt + .apply_halt() + .record(CalvinApplyHalt { + vshard_id: 12, + epoch: 40, + position: 3, + reason: "flush_failed", + step: "flush", + error: "CalvinFlush returned Error".to_string(), + }); + + let (status, body) = healthz_body(state).await; + + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE); + assert_eq!(body["status"], "degraded"); + assert_eq!(body["reason"], "calvin_apply_halted"); + assert_eq!(body["vshard_id"], 12); + assert_eq!(body["epoch"], 40); + assert_eq!(body["position"], 3); + assert_eq!(body["halt_reason"], "flush_failed"); + assert_eq!(body["step"], "flush"); + } + + #[tokio::test] + async fn a_node_without_a_valid_lease_reports_degraded_with_the_reason() { + let dir = tempfile::tempdir().expect("tempdir"); + let state = app_state(&dir); + assert!( + lease_invalid_body(&state).is_none(), + "a node without lease timing needs no lease" + ); + let timing = crate::control::security::auth_lease::LeaseTiming::from_raft( + std::time::Duration::from_millis(1000), + std::time::Duration::from_millis(100), + ) + .expect("timing"); + assert!(state.shared.authorization_fence.install_timing(timing)); + + let body = lease_invalid_body(&state).expect("no lease was granted"); + assert_eq!(body["status"], "degraded"); + assert_eq!(body["reason"], "authorization_lease_invalid"); + assert!(body["lease_expired_ms_ago"].is_null()); + + state + .shared + .authorization_fence + .holder() + .install(std::time::Instant::now() + std::time::Duration::from_secs(60)); + assert!(lease_invalid_body(&state).is_none()); + } + + #[tokio::test] + async fn healthz_without_a_calvin_halt_names_no_calvin_halt() { + let dir = tempfile::tempdir().expect("tempdir"); + let state = app_state(&dir); + + let (_status, body) = healthz_body(state).await; + + assert_ne!(body["reason"], "calvin_apply_halted"); + } + + /// The bound covers the longest statement deadline plus the apply wait, + /// and the vector install's two core dispatch deadlines. + #[tokio::test] + async fn the_outcome_floor_bound_covers_every_path_to_a_final_outcome() { + let dir = tempfile::tempdir().expect("tempdir"); + let state = app_state(&dir); + let network = &state.shared.tuning.network; + let statement = std::time::Duration::from_secs( + network + .default_deadline_secs + .max(network.copy_deadline_secs), + ) + crate::control::metadata_proposer::DEFAULT_PROPOSE_TIMEOUT; + let vector = crate::control::catalog_entry::post_apply::vector_install_longest_core_wait(); + + let bound = outcome_floor_bound(&state); + + assert!(bound >= statement); + assert!(bound >= vector); + assert_eq!(bound, statement.max(vector)); + } + + /// A window open past the bound degrades readiness and names the floor + /// it holds. + #[tokio::test] + async fn a_window_open_past_the_bound_reports_a_stuck_floor() { + let dir = tempfile::tempdir().expect("tempdir"); + let state = app_state(&dir); + let window = state.shared.outcome_floor.open_write(); + window.note_minted(crate::types::Lsn::new(7)); + std::thread::sleep(std::time::Duration::from_millis(2)); + + let body = outcome_floor_stuck_body(&state, std::time::Duration::ZERO) + .expect("the window is older than a zero bound"); + + assert_eq!(body["reason"], "outcome_floor_stuck"); + assert_eq!(body["open_windows"], 1); + assert_eq!(body["oldest_window_horizon"], 1); + assert_eq!(body["held_windows"], 0); + window.settle(); + assert!(outcome_floor_stuck_body(&state, std::time::Duration::ZERO).is_none()); + } + + /// A held window never degrades readiness. The healthz body counts it. + #[tokio::test] + async fn a_held_window_is_reported_without_degrading_readiness() { + let dir = tempfile::tempdir().expect("tempdir"); + let state = app_state(&dir); + let window = state.shared.outcome_floor.open_write(); + window.note_minted(crate::types::Lsn::new(7)); + window.hold(); + std::thread::sleep(std::time::Duration::from_millis(2)); + + assert!(outcome_floor_stuck_body(&state, std::time::Duration::ZERO).is_none()); + let (_status, body) = healthz_body(state).await; + + assert_ne!(body["reason"], "outcome_floor_stuck"); + assert_eq!(body["held_windows"], 1); + } +} diff --git a/nodedb/src/control/server/http/routes/metrics.rs b/nodedb/src/control/server/http/routes/metrics.rs index 99ef5cb0f..3afb56ad1 100644 --- a/nodedb/src/control/server/http/routes/metrics.rs +++ b/nodedb/src/control/server/http/routes/metrics.rs @@ -50,6 +50,42 @@ pub async fn metrics( output.push_str("# TYPE nodedb_wal_next_lsn gauge\n"); output.push_str(&format!("nodedb_wal_next_lsn {wal_lsn}\n\n")); + // Outcome floor: every engine watermark and WAL truncation stays at or + // below it. + let outcome_floor = &state.shared.outcome_floor; + output.push_str( + "# HELP nodedb_outcome_floor_lsn Highest WAL LSN at or below which every dispatched record has a final outcome.\n", + ); + output.push_str("# TYPE nodedb_outcome_floor_lsn gauge\n"); + output.push_str(&format!( + "nodedb_outcome_floor_lsn {}\n\n", + outcome_floor.floor().as_u64() + )); + output.push_str( + "# HELP nodedb_outcome_floor_oldest_window_seconds Age of the oldest write window holding the outcome floor.\n", + ); + output.push_str("# TYPE nodedb_outcome_floor_oldest_window_seconds gauge\n"); + output.push_str(&format!( + "nodedb_outcome_floor_oldest_window_seconds {}\n\n", + outcome_floor.oldest_open_for().as_secs_f64() + )); + output.push_str( + "# HELP nodedb_outcome_floor_windows_leaked_total Write windows dropped before their write's outcome was final.\n", + ); + output.push_str("# TYPE nodedb_outcome_floor_windows_leaked_total counter\n"); + output.push_str(&format!( + "nodedb_outcome_floor_windows_leaked_total {}\n\n", + outcome_floor.leaked_windows() + )); + output.push_str( + "# HELP nodedb_outcome_floor_windows_held Write windows held until restart; the floor stays below each.\n", + ); + output.push_str("# TYPE nodedb_outcome_floor_windows_held gauge\n"); + output.push_str(&format!( + "nodedb_outcome_floor_windows_held {}\n\n", + outcome_floor.held_windows() + )); + // Node ID. output.push_str("# HELP nodedb_node_id This node's cluster ID.\n"); output.push_str("# TYPE nodedb_node_id gauge\n"); @@ -238,6 +274,10 @@ pub async fn metrics( // Auth observability: method-specific counters, duration histograms, anomaly detection. output.push_str(&state.shared.auth_metrics.to_prometheus()); + // Authorization lease validity. A node without a valid lease refuses + // permission-checked statements. + crate::control::security::auth_lease::status::render_prometheus(&state.shared, &mut output); + // Metering capacity: dropped-entry counters, so a refused (i.e. never // billed) usage record is observable without reading server logs. crate::control::security::metering::metrics::render_prometheus( diff --git a/nodedb/src/control/server/http/routes/promql/remote.rs b/nodedb/src/control/server/http/routes/promql/remote.rs index f694c788c..8f613cda5 100644 --- a/nodedb/src/control/server/http/routes/promql/remote.rs +++ b/nodedb/src/control/server/http/routes/promql/remote.rs @@ -23,7 +23,7 @@ use crate::control::promql::{self, types::DEFAULT_LOOKBACK_MS}; use crate::control::server::http::admission::admit_without_rate_limit; use crate::control::server::http::auth::{AppState, ResolvedIdentity}; use crate::control::server::http::peer::PeerAddr; -use crate::types::{DatabaseId, TraceId, VShardId}; +use crate::types::{DatabaseId, TraceId}; use nodedb_physical::physical_plan::{PhysicalPlan, TimeseriesOp}; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; @@ -103,7 +103,8 @@ pub async fn remote_write( continue; } - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &collection); + let vshard = + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection).vshard(); let plan = PhysicalPlan::Timeseries(TimeseriesOp::Ingest { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, &collection), payload: ilp_payload.into_bytes(), @@ -182,7 +183,7 @@ pub async fn remote_write( }; gw.execute(&gw_ctx, checked).await } - None => crate::control::server::dispatch_utils::dispatch_authorized_autocommit_write( + None => crate::control::server::dispatch_utils::dispatch_authorized_durable_write( &state.shared, checked, TraceId::generate(), diff --git a/nodedb/src/control/server/http/routes/query.rs b/nodedb/src/control/server/http/routes/query.rs index 7a5e51513..083164a18 100644 --- a/nodedb/src/control/server/http/routes/query.rs +++ b/nodedb/src/control/server/http/routes/query.rs @@ -50,9 +50,13 @@ pub(crate) fn resolve_database_id( match catalog.get_database_id_by_name(&db_name) { Ok(Some(id)) => Ok(id), - Ok(None) => Err(ApiError::BadRequest(format!( - "3D000 database '{db_name}' does not exist" - ))), + // The status of `3D000` from the one gateway status table. + Ok(None) => Err(ApiError::HttpStatus( + crate::control::gateway::GatewayErrorMap::sqlstate_to_http( + nodedb_types::error::sqlstate::INVALID_CATALOG_NAME, + ), + format!("3D000 database '{db_name}' does not exist"), + )), Err(e) => Err(ApiError::Internal(format!("catalog lookup failed: {e}"))), } } diff --git a/nodedb/src/control/server/http/routes/query/materialized/encode.rs b/nodedb/src/control/server/http/routes/query/materialized/encode.rs index f6d4be59c..190b7b8e9 100644 --- a/nodedb/src/control/server/http/routes/query/materialized/encode.rs +++ b/nodedb/src/control/server/http/routes/query/materialized/encode.rs @@ -4,31 +4,34 @@ use super::super::super::super::auth::ApiError; +/// Map a DDL error to the HTTP error the client reads. The status follows the +/// SQLSTATE through the gateway status table. The code and the typed cause +/// travel in the body, as they do on native and pgwire. pub(super) fn ddl_error_to_api(error: crate::control::server::shared::ddl::DdlError) -> ApiError { - let status = if error.sqlstate == "42501" { - axum::http::StatusCode::FORBIDDEN - } else { - axum::http::StatusCode::BAD_REQUEST - }; + let status = crate::control::gateway::GatewayErrorMap::sqlstate_to_http(&error.sqlstate); ApiError::Coded { - status, + status: axum::http::StatusCode::from_u16(status) + .unwrap_or(axum::http::StatusCode::INTERNAL_SERVER_ERROR), message: error.message, code: error.code, + cause: error.cause, } } +/// Map a gateway error to the HTTP error the client reads, through the one +/// `crate::Error` to `ApiError` conversion. pub(super) fn gateway_error(error: crate::Error) -> ApiError { - let (status, msg) = crate::control::gateway::GatewayErrorMap::to_http(&error); - ApiError::HttpStatus(status, msg) + ApiError::from(error) } +/// Map a Data-Plane refusal to the HTTP error the client reads. A typed +/// refusal takes the status its code maps to. Only a refusal with no code is +/// an internal error. pub(super) fn response_error(response: &crate::bridge::envelope::Response) -> ApiError { - let detail = response - .error_code - .as_ref() - .map(|code| format!("{code:?}")) - .unwrap_or_else(|| "unknown error".into()); - ApiError::Internal(detail) + match response.error_code.as_deref() { + Some(code) => gateway_error(crate::Error::DataPlane(code.clone())), + None => ApiError::Internal("data plane returned an error status with no error code".into()), + } } #[cfg(test)] @@ -44,7 +47,7 @@ mod tests { assert!(matches!( ddl_error_to_api(error), - ApiError::Coded { status, message, code } + ApiError::Coded { status, message, code, .. } if status == axum::http::StatusCode::FORBIDDEN && message == "write permission denied" && code == nodedb_types::error::ErrorCode::AUTHORIZATION_DENIED @@ -71,5 +74,85 @@ mod tests { nodedb_types::error::ErrorCode::AUTHORIZATION_DENIED.to_string() ); assert_eq!(json["error"], "write permission denied"); + assert!(json.get("cause").is_none(), "no cause, no cause field"); + } + + async fn response_json(error: ApiError) -> (axum::http::StatusCode, serde_json::Value) { + let response = error.into_response(); + let status = response.status(); + let body = axum::body::to_bytes(response.into_body(), usize::MAX) + .await + .expect("read response body"); + let json = serde_json::from_slice(&body).expect("valid JSON body"); + (status, json) + } + + /// An internal DDL error is a server fault, and its typed cause reaches + /// the client with its own message and code. + #[tokio::test] + async fn internal_ddl_error_is_500_with_its_cause() { + let cause = nodedb_types::NodeDbError::from(crate::Error::DataPlane( + crate::bridge::envelope::ErrorCode::Unsupported { + detail: "not on this engine".into(), + }, + )); + let mut error = crate::control::server::shared::ddl::DdlError::move_tenant_cutover_failed( + "MOVE TENANT cutover failed", + ); + error.cause = Some(Box::new(cause)); + + let (status, json) = response_json(ddl_error_to_api(error)).await; + assert_eq!(status, axum::http::StatusCode::INTERNAL_SERVER_ERROR); + assert_eq!(json["error"], "MOVE TENANT cutover failed"); + assert_eq!( + json["code"], + nodedb_types::error::ErrorCode::MOVE_TENANT_CUTOVER_FAILED.to_string() + ); + assert_eq!( + json["cause"]["code"], + nodedb_types::error::ErrorCode::SQL_NOT_ENABLED.to_string() + ); + assert_eq!(json["cause"]["error"], "not on this engine"); + } + + /// An internal DDL error is a server fault, never a client error. + #[tokio::test] + async fn xx000_ddl_error_is_500() { + let error = crate::control::server::shared::ddl::DdlError::internal("catalog write failed"); + let (status, _) = response_json(ddl_error_to_api(error)).await; + assert_eq!(status, axum::http::StatusCode::INTERNAL_SERVER_ERROR); + } + + /// A feature-not-supported DDL error is 501 Not Implemented. + #[tokio::test] + async fn feature_not_supported_ddl_error_is_501() { + let error = crate::control::server::shared::ddl::DdlError::new( + "0A000", + "changing vector index params is not supported", + ); + let (status, json) = response_json(ddl_error_to_api(error)).await; + assert_eq!(status, axum::http::StatusCode::NOT_IMPLEMENTED); + assert_eq!( + json["code"], + nodedb_types::error::ErrorCode::SQL_NOT_ENABLED.to_string() + ); + } + + /// A conflict, a constraint violation and a rate limit take their own + /// status, never a blanket 400. + #[test] + fn ddl_conflicts_and_rate_limits_take_their_class_status() { + for (state, expected) in [ + ("42P07", axum::http::StatusCode::CONFLICT), + ("23505", axum::http::StatusCode::CONFLICT), + ("53300", axum::http::StatusCode::TOO_MANY_REQUESTS), + ("42P01", axum::http::StatusCode::NOT_FOUND), + ] { + let error = crate::control::server::shared::ddl::DdlError::new(state, "refused"); + match ddl_error_to_api(error) { + ApiError::Coded { status, .. } => assert_eq!(status, expected, "{state}"), + other => panic!("expected a coded error for {state}, got {other:?}"), + } + } } } diff --git a/nodedb/src/control/server/http/routes/query/materialized/shape.rs b/nodedb/src/control/server/http/routes/query/materialized/shape.rs index ad9148e54..073777685 100644 --- a/nodedb/src/control/server/http/routes/query/materialized/shape.rs +++ b/nodedb/src/control/server/http/routes/query/materialized/shape.rs @@ -323,7 +323,7 @@ pub(super) async fn run_task_loop( None => { // Single-node boot: gateway not yet initialised — dispatch locally. let response = - crate::control::server::dispatch_utils::dispatch_authorized_autocommit_write( + crate::control::server::dispatch_utils::dispatch_authorized_durable_write( &state.shared, checked, trace_id, diff --git a/nodedb/src/control/server/http/routes/query/ndjson.rs b/nodedb/src/control/server/http/routes/query/ndjson.rs index 097f16619..9d99551af 100644 --- a/nodedb/src/control/server/http/routes/query/ndjson.rs +++ b/nodedb/src/control/server/http/routes/query/ndjson.rs @@ -307,8 +307,9 @@ pub async fn query_ndjson( }; gw.execute(&gw_ctx, checked).await } + // A write takes the durable route, a read the read route. None => { - crate::control::server::dispatch_utils::dispatch_authorized_to_data_plane( + crate::control::server::dispatch_utils::dispatch_authorized_task_by_class( &state.shared, checked, trace_id, diff --git a/nodedb/src/control/server/http/routes/result_shape.rs b/nodedb/src/control/server/http/routes/result_shape.rs index a3022bd33..e217ff02d 100644 --- a/nodedb/src/control/server/http/routes/result_shape.rs +++ b/nodedb/src/control/server/http/routes/result_shape.rs @@ -16,33 +16,26 @@ use crate::control::server::response_shape::cell::row_to_wire_json; use crate::control::server::response_shape::compose::{ShapeOutcome, shape_response_materialized}; use crate::control::server::response_shape::request::MaterializedShapeRequest; use nodedb_types::NodeDbError; -use nodedb_types::error::ErrorCode; use super::super::auth::ApiError; /// Map a shaping error to the HTTP error the client reads, keeping its -/// numeric code. A statement-level refusal the shaper raises per row — an -/// unknown sequence, `currval` before `nextval`, a bad accessor argument, -/// division by zero — is the caller's error and answers `400`; anything -/// else is the server's and answers `500`. +/// numeric code. The status follows the SQLSTATE the code renders on +/// pgwire, through the one gateway status table. A per-row refusal, such as +/// an unknown sequence or a division by zero, answers its client class. +/// An internal error answers `500`. pub(super) fn shape_error_to_api(e: NodeDbError) -> ApiError { let code = e.code(); - let status = if matches!( - code, - ErrorCode::UNDEFINED_OBJECT - | ErrorCode::OBJECT_NOT_READY - | ErrorCode::PLAN_ERROR - | ErrorCode::DIVISION_BY_ZERO - | ErrorCode::BAD_REQUEST - ) { - StatusCode::BAD_REQUEST - } else { - StatusCode::INTERNAL_SERVER_ERROR - }; + let state = crate::control::server::pgwire::types::error_map::numeric_code_to_sqlstate(code); + let status = StatusCode::from_u16(crate::control::gateway::GatewayErrorMap::sqlstate_to_http( + state, + )) + .unwrap_or(StatusCode::INTERNAL_SERVER_ERROR); ApiError::Coded { status, message: e.message().to_string(), code, + cause: e.cause().map(|cause| Box::new(cause.clone())), } } diff --git a/nodedb/src/control/server/http/routes/ws_rpc/execute_sql.rs b/nodedb/src/control/server/http/routes/ws_rpc/execute_sql.rs index 12806adc2..4fd83ccea 100644 --- a/nodedb/src/control/server/http/routes/ws_rpc/execute_sql.rs +++ b/nodedb/src/control/server/http/routes/ws_rpc/execute_sql.rs @@ -295,8 +295,9 @@ pub async fn execute_sql( gw.execute(&gw_ctx, checked).await } None => { - // Single-node boot: gateway not yet initialised — dispatch locally. - crate::control::server::dispatch_utils::dispatch_authorized_to_data_plane( + // Single-node boot: gateway not yet initialised — dispatch + // locally. A write takes the durable route, a read the read route. + crate::control::server::dispatch_utils::dispatch_authorized_task_by_class( shared, checked, trace_id, ) .await diff --git a/nodedb/src/control/server/http/server.rs b/nodedb/src/control/server/http/server.rs index fcabd76cc..7e8981a2e 100644 --- a/nodedb/src/control/server/http/server.rs +++ b/nodedb/src/control/server/http/server.rs @@ -3,11 +3,14 @@ //! HTTP API server using axum + axum-server (for TLS). //! //! Probe routes (unversioned, always reachable): -//! - GET /healthz — k8s readiness/liveness (always reachable; 503 until GatewayEnable) +//! - GET /healthz — k8s readiness (always reachable; 503 until the `Serving` phase) //! - GET /health/live — unconditional liveness probe -//! - GET /health/ready — readiness (WAL recovered) +//! - GET /health/ready — readiness (WAL recovered and the `Serving` phase reached) //! - POST /health/drain — trigger graceful drain -//! - GET /metrics — Prometheus-format metrics (requires monitor role) +//! - GET /metrics — Prometheus-format metrics (requires monitor role; always reachable) +//! +//! Every other route returns 503 until the `Serving` startup phase (see +//! [`super::startup_gate`]). //! //! All other routes are versioned under `/v1/`. //! @@ -40,9 +43,8 @@ use std::sync::Arc; use axum::Router; -use axum::extract::{DefaultBodyLimit, State}; -use axum::middleware::{self, Next}; -use axum::response::Response; +use axum::extract::DefaultBodyLimit; +use axum::middleware; use axum::routing::{get, post, put}; use tracing::info; @@ -64,7 +66,7 @@ use super::routes; /// SSE and WebSocket routes are kept on a separate sub-router that does NOT /// carry the `map_response` layer — those handlers set their own /// `Content-Type` (text/event-stream, or the WS upgrade response). -fn build_router(state: AppState) -> Router { +pub(super) fn build_router(state: AppState) -> Router { // ── Streaming / non-JSON routes (no Content-Type stamp) ────────────────── let streaming_routes = Router::new() // WebSocket RPC — upgrade response, not JSON. @@ -172,50 +174,11 @@ fn build_router(state: AppState) -> Router { .merge(streaming_routes) .layer(middleware::from_fn_with_state( state.clone(), - startup_gate_middleware, + super::startup_gate::startup_gate_middleware, )) .with_state(state) } -/// Axum middleware that gates non-health routes on [`StartupPhase::GatewayEnable`]. -/// -/// All `/health*` paths (liveness, readiness, drain) are always let through so -/// k8s probes can observe startup progress. All other routes receive a -/// `503 Service Unavailable` until the node reaches `GatewayEnable`. -async fn startup_gate_middleware( - State(app_state): State, - req: axum::http::Request, - next: Next, -) -> Response { - use axum::http::StatusCode; - use axum::response::IntoResponse; - - let path = req.uri().path(); - // Health-probe paths bypass the gate — these must be reachable during startup. - let is_health_path = path == "/healthz" || path.starts_with("/health/"); - - if !is_health_path { - let gate = &app_state.shared.startup; - let snap = gate.current_phase(); - if let Some(err) = gate.is_failed() { - let body = serde_json::json!({ - "status": "failed", - "error": err.to_string(), - }); - return (StatusCode::SERVICE_UNAVAILABLE, axum::Json(body)).into_response(); - } - if snap < crate::control::startup::StartupPhase::GatewayEnable { - let body = serde_json::json!({ - "status": "starting", - "phase": snap.name(), - }); - return (StatusCode::SERVICE_UNAVAILABLE, axum::Json(body)).into_response(); - } - } - - next.run(req).await -} - /// Start the HTTP API server from an already-bound [`tokio::net::TcpListener`]. /// /// Alias of [`run`], kept for call sites (tests, harnesses) that bind an diff --git a/nodedb/src/control/server/http/startup_gate.rs b/nodedb/src/control/server/http/startup_gate.rs new file mode 100644 index 000000000..c8fc31382 --- /dev/null +++ b/nodedb/src/control/server/http/startup_gate.rs @@ -0,0 +1,184 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The HTTP gate that holds every non-probe route until the node serves. +//! +//! The HTTP listener serves from early in boot so orchestrator probes can +//! watch startup. Until the startup gate reports [`HealthState::Ok`], only +//! these routes answer: +//! +//! - `/healthz`, which reports `starting` with `503`; +//! - `/health/*` (liveness, readiness, drain); +//! - `/metrics`. +//! +//! Every other route, an unmatched path included, returns `503` with a +//! `starting` body. The one readiness signal is the startup phase, read +//! through [`observe`]: the node is ready at [`StartupPhase::Serving`], which +//! boot enters where it opens the client protocols. +//! +//! [`StartupPhase::Serving`]: crate::control::startup::StartupPhase::Serving + +use axum::extract::State; +use axum::middleware::Next; +use axum::response::{IntoResponse, Response}; +use serde_json::json; + +use crate::control::startup::health::{HealthState, observe, to_http_response}; + +use super::auth::AppState; + +/// The error text of a route refused while the node boots. +pub const NODE_STARTING: &str = "node is starting"; + +/// Whether `path` is a probe or metrics route, which answers during boot. +fn answers_during_boot(path: &str) -> bool { + path == "/healthz" || path.starts_with("/health/") || path == "/metrics" +} + +/// Refuse every non-probe route until the node reaches the final phase. +pub(super) async fn startup_gate_middleware( + State(app_state): State, + req: axum::http::Request, + next: Next, +) -> Response { + if answers_during_boot(req.uri().path()) { + return next.run(req).await; + } + let health = observe(&app_state.shared.startup); + if matches!(health, HealthState::Ok) { + return next.run(req).await; + } + let (status, mut body) = to_http_response(&health); + if matches!(health, HealthState::Starting { .. }) { + body["error"] = json!(NODE_STARTING); + } + (status, axum::Json(body)).into_response() +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use axum::body::Body; + use axum::http::{Method, Request, StatusCode}; + use tower::ServiceExt; + + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::config::auth::AuthMode; + use crate::control::startup::{ReadyGate, StartupPhase, StartupSequencer}; + use crate::control::state::SharedState; + use crate::wal::WalManager; + + /// A router over a state whose boot reached `GatewayEnable` but not + /// `Serving`, as between `await_cluster_ready` and the opening of the + /// client protocols. Firing the returned gate enters `Serving`. + fn booting_router(dir: &tempfile::TempDir) -> (axum::Router, StartupSequencer, ReadyGate) { + let wal = + Arc::new(WalManager::open_for_testing(&dir.path().join("gate.wal")).expect("open WAL")); + let (dispatcher, _data_sides) = Dispatcher::new(1, 64); + let mut shared = SharedState::new(dispatcher, wal).expect("shared state"); + let (sequencer, gate) = StartupSequencer::new(); + let serving = sequencer.register_gate(StartupPhase::Serving, "test-serving"); + sequencer + .register_gate(StartupPhase::GatewayEnable, "test-gateway") + .fire(); + assert_eq!(gate.current_phase(), StartupPhase::GatewayEnable); + Arc::get_mut(&mut shared) + .expect("state is uniquely owned here") + .startup = Arc::clone(&gate); + let state = AppState { + shutdown_bus: crate::control::shutdown::ShutdownBus::new(Arc::clone(&shared.shutdown)) + .0, + query_ctx: Arc::new(crate::control::planner::context::QueryContext::for_state( + &shared, + )), + shared, + auth_mode: AuthMode::Trust, + }; + let router = super::super::server::build_router(state).layer(axum::Extension( + crate::control::security::tls_policy::TransportSecurity::Cleartext, + )); + (router, sequencer, serving) + } + + async fn get(router: &axum::Router, path: &str) -> (StatusCode, String) { + let req = Request::builder() + .method(Method::GET) + .uri(path) + .body(Body::empty()) + .expect("request"); + let response = router.clone().oneshot(req).await.expect("response"); + let status = response.status(); + let bytes = axum::body::to_bytes(response.into_body(), usize::MAX) + .await + .expect("read body"); + (status, String::from_utf8_lossy(&bytes).into_owned()) + } + + #[tokio::test] + async fn healthz_reports_starting_until_serving() { + let dir = tempfile::tempdir().expect("tempdir"); + let (router, _sequencer, serving) = booting_router(&dir); + + let (status, body) = get(&router, "/healthz").await; + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE); + let body: serde_json::Value = sonic_rs::from_str(&body).expect("healthz body is JSON"); + assert_eq!(body["status"], "starting"); + + let (status, _) = get(&router, "/health/live").await; + assert_eq!(status, StatusCode::OK); + + serving.fire(); + let (status, body) = get(&router, "/healthz").await; + assert_eq!(status, StatusCode::OK, "healthz after admission: {body}"); + let body: serde_json::Value = sonic_rs::from_str(&body).expect("healthz body is JSON"); + assert_eq!(body["status"], "ok"); + } + + #[tokio::test] + async fn a_non_probe_route_is_refused_until_serving() { + let dir = tempfile::tempdir().expect("tempdir"); + let (router, _sequencer, serving) = booting_router(&dir); + + let (status, body) = get(&router, "/v1/status").await; + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE); + let parsed: serde_json::Value = sonic_rs::from_str(&body).expect("refusal body is JSON"); + assert_eq!(parsed["status"], "starting"); + assert_eq!(parsed["error"], NODE_STARTING); + + // Metrics answer during boot. + let (_, body) = get(&router, "/metrics").await; + assert!(!body.contains(NODE_STARTING)); + + serving.fire(); + let (_, body) = get(&router, "/v1/status").await; + assert!(!body.contains(NODE_STARTING)); + } + + /// The gate layer also wraps the fallback: an unmatched path returns 503 + /// during boot and 404 once the node serves. + #[tokio::test] + async fn an_unmatched_path_is_refused_until_serving_then_not_found() { + let dir = tempfile::tempdir().expect("tempdir"); + let (router, _sequencer, serving) = booting_router(&dir); + + let (status, body) = get(&router, "/health").await; + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE); + assert!(body.contains(NODE_STARTING)); + + serving.fire(); + let (status, _) = get(&router, "/health").await; + assert_eq!(status, StatusCode::NOT_FOUND); + } + + #[test] + fn probe_and_metrics_paths_answer_during_boot() { + assert!(answers_during_boot("/healthz")); + assert!(answers_during_boot("/health/live")); + assert!(answers_during_boot("/health/ready")); + assert!(answers_during_boot("/metrics")); + assert!(!answers_during_boot("/v1/query")); + assert!(!answers_during_boot("/healthzz")); + assert!(!answers_during_boot("/metrics/extra")); + } +} diff --git a/nodedb/src/control/server/http/types.rs b/nodedb/src/control/server/http/types.rs index c692aadbe..df8abe506 100644 --- a/nodedb/src/control/server/http/types.rs +++ b/nodedb/src/control/server/http/types.rs @@ -126,6 +126,10 @@ pub struct HttpError { pub error: String, #[serde(skip_serializing_if = "Option::is_none")] pub code: Option, + /// The typed error that caused this one, as its message and code. Absent + /// when there is none. + #[serde(skip_serializing_if = "Option::is_none")] + pub cause: Option>, } impl HttpError { @@ -133,6 +137,7 @@ impl HttpError { Self { error: error.into(), code: None, + cause: None, } } @@ -140,8 +145,18 @@ impl HttpError { Self { error: error.into(), code: Some(code.into()), + cause: None, } } + + /// Carry `cause` as this error's cause, with its own message and code. + pub fn caused_by(mut self, cause: &nodedb_types::NodeDbError) -> Self { + self.cause = Some(Box::new(Self::with_code( + cause.message(), + cause.code().to_string(), + ))); + self + } } #[cfg(test)] diff --git a/nodedb/src/control/server/ilp_batch/dispatch.rs b/nodedb/src/control/server/ilp_batch/dispatch.rs index e75f5dca5..6e18c5e6e 100644 --- a/nodedb/src/control/server/ilp_batch/dispatch.rs +++ b/nodedb/src/control/server/ilp_batch/dispatch.rs @@ -19,7 +19,7 @@ use crate::control::server::ilp_auth::AuthenticatedIlpContext; use crate::control::server::shared::authorization::authorize_task_set; use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TenantId, VShardId}; +use crate::types::{DatabaseId, TenantId}; use nodedb_physical::physical_plan::TimeseriesOp; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; use nodedb_types::Surrogate; @@ -297,7 +297,8 @@ fn build_ilp_calvin_tasks( Ok(PhysicalTask { tenant_id, database_id, - vshard_id: VShardId::from_collection_in_database(database_id, &group.measurement), + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, &group.measurement) + .vshard(), plan: PhysicalPlan::Timeseries(TimeseriesOp::Ingest { collection: nodedb_types::QualifiedCollection::new( database_id, @@ -338,7 +339,7 @@ mod tests { use crate::control::security::identity::{AuthMethod, AuthenticatedIdentity, DatabaseSet}; use crate::control::security::permission::PermissionStore; use crate::control::security::role::RoleStore; - use crate::types::{DatabaseId, TenantId, VShardId}; + use crate::types::{DatabaseId, TenantId}; use crate::wal::WalManager; use nodedb_physical::physical_plan::{PhysicalPlan, TimeseriesOp}; use nodedb_types::Surrogate; @@ -636,7 +637,7 @@ mod tests { assert_eq!(tasks[0].database_id, database_id); assert_eq!( tasks[0].vshard_id, - VShardId::from_collection_in_database(database_id, "cpu") + nodedb_types::CollectionKey::from_bare(database_id, "cpu").vshard() ); } } diff --git a/nodedb/src/control/server/ilp_listener.rs b/nodedb/src/control/server/ilp_listener.rs index 32e0852cd..79eb1ad59 100644 --- a/nodedb/src/control/server/ilp_listener.rs +++ b/nodedb/src/control/server/ilp_listener.rs @@ -51,7 +51,11 @@ pub struct IlpListener { impl IlpListener { /// Bind to the given address. pub async fn bind(addr: SocketAddr) -> crate::Result { - let tcp = TcpListener::bind(addr).await.map_err(crate::Error::Io)?; + Self::from_listener(TcpListener::bind(addr).await.map_err(crate::Error::Io)?) + } + + /// Serve on a socket that already listens. + pub fn from_listener(tcp: TcpListener) -> crate::Result { let local_addr = tcp.local_addr().map_err(crate::Error::Io)?; info!(%local_addr, "ILP TCP listener bound"); Ok(Self { diff --git a/nodedb/src/control/server/listener.rs b/nodedb/src/control/server/listener.rs index c3d87df20..5bc131d9b 100644 --- a/nodedb/src/control/server/listener.rs +++ b/nodedb/src/control/server/listener.rs @@ -48,7 +48,11 @@ pub struct ListenerRunParams { impl Listener { /// Bind to the given address. pub async fn bind(addr: SocketAddr) -> crate::Result { - let tcp = TcpListener::bind(addr).await?; + Self::from_listener(TcpListener::bind(addr).await?) + } + + /// Serve on a socket that already listens. + pub fn from_listener(tcp: TcpListener) -> crate::Result { let local_addr = tcp.local_addr()?; info!(%local_addr, "control plane listener bound"); Ok(Self { diff --git a/nodedb/src/control/server/mod.rs b/nodedb/src/control/server/mod.rs index b2653b9db..1e49f7491 100644 --- a/nodedb/src/control/server/mod.rs +++ b/nodedb/src/control/server/mod.rs @@ -16,6 +16,7 @@ pub mod payload_merge; pub mod pgwire; pub mod post_aggregate; pub mod reservation; +pub mod reserved_socket; pub mod resp; pub mod response_shape; pub mod response_translate; diff --git a/nodedb/src/control/server/native/dispatch/conversion.rs b/nodedb/src/control/server/native/dispatch/conversion.rs index 9e8e4649c..d29ab4379 100644 --- a/nodedb/src/control/server/native/dispatch/conversion.rs +++ b/nodedb/src/control/server/native/dispatch/conversion.rs @@ -13,6 +13,15 @@ use crate::control::server::response_shape::types::{ use crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate; use crate::control::server::shared::ddl::{DdlError, DdlResult}; +/// The SQLSTATE, message and numeric code a native error frame carries for +/// one Control-Plane error. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct NativeErrorFields { + pub(crate) sqlstate: &'static str, + pub(crate) message: String, + pub(crate) code: nodedb_types::error::ErrorCode, +} + /// Convert a Control-Plane error into a native error frame. /// /// The stable numeric NodeDB code travels alongside the SQLSTATE, taken from @@ -24,7 +33,14 @@ use crate::control::server::shared::ddl::{DdlError, DdlResult}; /// The SQLSTATE is chosen here because it is a protocol-level rendering, while /// the numeric code is the classification itself. pub(crate) fn error_to_native(seq: u64, e: &crate::Error) -> NativeResponse { - let (code, message) = match e { + let fields = native_error_fields(e); + NativeResponse::error_with_code(seq, fields.sqlstate, fields.message, fields.code.0) +} + +/// The fields [`error_to_native`] puts on the frame. The one native mapping: +/// every native rendering of an `Error` reads it. +pub(crate) fn native_error_fields(e: &crate::Error) -> NativeErrorFields { + let (sqlstate, message) = match e { crate::Error::BadRequest { detail } => ("42601", detail.clone()), crate::Error::RejectedAuthz { resource, .. } => ("42501", resource.clone()), crate::Error::RateExceeded { .. } => ( @@ -35,11 +51,12 @@ pub(crate) fn error_to_native(seq: u64, e: &crate::Error) -> NativeResponse { crate::Error::CollectionNotFound { collection, .. } => { ("42P01", format!("collection '{collection}' not found")) } - // Same SQLSTATE as the "not authenticated" responses in - // `session::request`: the client's stored bearer token expired - // mid-connection and it must re-authenticate with a fresh Auth frame. + // The SQLSTATE pgwire renders, and the one the "not authenticated" + // responses in `session::request` carry. The client's stored bearer + // token expired mid-connection and it must re-authenticate with a + // fresh Auth frame. The message names the native Auth frame. crate::Error::SessionTokenExpired => ( - "28000", + nodedb_types::error::sqlstate::AUTH_TOKEN_EXPIRED.0, "OIDC bearer token expired; re-authenticate with a fresh Auth request".into(), ), // A cross-shard Calvin OCC abort is a serialization failure (40001) — @@ -67,10 +84,36 @@ pub(crate) fn error_to_native(seq: u64, e: &crate::Error) -> NativeResponse { crate::control::server::pgwire::types::error_map::numeric_code_to_sqlstate(e.code()), e.message().to_string(), ), - other => ("XX000", format!("{other}")), + // Every other variant takes the protocol-neutral SQLSTATE pgwire + // renders for it. `XX000` is only for a variant that table leaves + // unclassified. + other => { + let (_severity, sqlstate, message) = + crate::control::server::pgwire::types::error_map::error_to_sqlstate(other); + (sqlstate, message) + } }; - let ndb_code = crate::error_classify::classify(e).code().0; - NativeResponse::error_with_code(seq, code, message, ndb_code) + NativeErrorFields { + sqlstate, + message, + code: crate::error_classify::classify(e).code(), + } +} + +/// Convert a Control-Plane error into a native error frame with `context` +/// before its message. The SQLSTATE and code stay the error's own. +pub(crate) fn error_to_native_in_context( + seq: u64, + context: &str, + e: &crate::Error, +) -> NativeResponse { + let fields = native_error_fields(e); + NativeResponse::error_with_code( + seq, + fields.sqlstate, + format!("{context}: {}", fields.message), + fields.code.0, + ) } /// Convert a Control-Plane error into a native error frame under a SQLSTATE @@ -99,18 +142,26 @@ pub(crate) fn error_to_native_with_sqlstate( /// Convert a `NodeDbError` produced while shaping a response into a /// NativeResponse error frame. /// -/// The numeric code travels alongside the SQLSTATE: the error is already -/// classified here, and rendering only `XX000` would make the client rebuild -/// it as a generic internal failure. +/// The numeric code travels alongside the SQLSTATE its code maps to, the same +/// SQLSTATE pgwire's `shape_error_to_pg` renders. pub(crate) fn shape_error_to_native(seq: u64, e: &nodedb_types::NodeDbError) -> NativeResponse { - NativeResponse::error_with_code(seq, "XX000", e.message().to_string(), e.code().0) + NativeResponse::error_with_code( + seq, + crate::control::server::pgwire::types::error_map::numeric_code_to_sqlstate(e.code()), + e.message().to_string(), + e.code().0, + ) } /// Render a statement-tag fold refusal as a native error frame. Two tasks of /// one statement disagreeing on their verb is a planner bug, so it is an /// internal error, the same class pgwire's `dml_fold_error_to_pg` renders. pub(crate) fn dml_fold_error_to_native(seq: u64, e: &DmlFoldError) -> NativeResponse { - sqlstate_error(seq, "XX000", e.to_string()) + sqlstate_error( + seq, + nodedb_types::error::sqlstate::INTERNAL_ERROR, + e.to_string(), + ) } /// Render an error [`Response`] from the Data Plane as a native error frame. @@ -143,11 +194,16 @@ pub(crate) fn error_code_to_native( code: Option<&crate::bridge::envelope::ErrorCode>, ) -> NativeResponse { let Some(code) = code else { - return sqlstate_error(seq, "XX000", "unknown data plane error"); + return sqlstate_error( + seq, + nodedb_types::error::sqlstate::INTERNAL_ERROR, + "unknown data plane error", + ); }; let (_, sqlstate, message) = error_code_to_sqlstate(code); let public = nodedb_types::NodeDbError::from(crate::Error::DataPlane(code.clone())); NativeResponse::error_with_code(seq, sqlstate, message, public.code().0) + .with_error_details(public.details().clone()) } /// Encode a protocol-neutral DDL dispatch result into a single @@ -174,7 +230,21 @@ pub(crate) fn ddl_result_to_native( sqlstate, code, message, - }) => NativeResponse::error_with_code(seq, sqlstate, message, code.0), + details, + cause, + }) => { + let frame = NativeResponse::error_with_code(seq, sqlstate, message, code.0); + let frame = match details { + Some(details) => frame.with_error_details(*details), + None => frame, + }; + match cause { + Some(cause) => frame.with_error_cause( + nodedb_types::protocol::ErrorCausePayload::from(cause.as_ref()), + ), + None => frame, + } + } // Unknown pgwire response variants are dropped during translation, so // the first element is the first meaningful result — the bridge // returns on the first known variant. @@ -352,6 +422,42 @@ mod tests { assert_eq!(error.code, "28000"); } + /// A shaping error answers the SQLSTATE its code maps to, the one pgwire + /// renders, never a bare internal error. + #[test] + fn a_shaping_error_keeps_its_sqlstate() { + let shaped = shape_error_to_native(1, &nodedb_types::NodeDbError::division_by_zero()); + let error = shaped.error.expect("error responses carry a payload"); + assert_eq!(error.code, "22012"); + assert_eq!( + error.ndb_code, + nodedb_types::error::ErrorCode::DIVISION_BY_ZERO.0 + ); + } + + /// A typed error behind a context prefix keeps its SQLSTATE and code. + #[test] + fn an_error_in_context_keeps_its_class() { + let missing = crate::Error::CollectionNotFound { + tenant_id: crate::types::TenantId::new(1), + collection: "orders".into(), + }; + let response = error_to_native_in_context(1, "database catalog lookup failed", &missing); + let error = response.error.expect("error responses carry a payload"); + assert_eq!(error.code, "42P01"); + assert_eq!( + error.ndb_code, + nodedb_types::error::ErrorCode::COLLECTION_NOT_FOUND.0 + ); + assert!( + error + .message + .starts_with("database catalog lookup failed: "), + "{}", + error.message + ); + } + /// One statement running out of time answers ONE SQLSTATE, whichever half /// of the race reported it: the Control-Plane timer, which raises /// `DeadlineExceeded` directly, or a shard refusing an already-expired @@ -458,6 +564,32 @@ mod tests { ); } + /// A phase failure keeps its own code, and the typed Data-Plane cause + /// rides beside it with its own code, so a client sees both. + #[test] + fn ddl_phase_failure_carries_its_typed_cause() { + let phase = nodedb_types::NodeDbError::move_tenant_snapshot_failed("7", "dispatch") + .with_cause(nodedb_types::NodeDbError::division_by_zero()); + let response = ddl_result_to_native( + 1, + Err(DdlError::move_tenant_snapshot_failed(phase.message()).with_cause_of(&phase)), + ); + + let bytes = zerompk::to_msgpack_vec(&response).expect("encode native response"); + let decoded: NativeResponse = + zerompk::from_msgpack(&bytes).expect("decode native response"); + let error = decoded.error.expect("error responses carry a payload"); + assert_eq!( + error.ndb_code, + nodedb_types::error::ErrorCode::MOVE_TENANT_SNAPSHOT_FAILED.0 + ); + let cause = error.cause.expect("the typed cause survives the wire"); + assert_eq!( + cause.to_error().code(), + nodedb_types::error::ErrorCode::DIVISION_BY_ZERO + ); + } + /// A count-bearing DDL-router status (the `{ ... }` document INSERT) is /// a DML answer: `(rows_affected, command)` exactly as the dispatch /// loop's folded tag reports, never a status row with no verb. @@ -532,11 +664,10 @@ mod tests { ); } - /// The numeric code is populated for every variant, including the ones - /// whose SQLSTATE falls through to `XX000` — otherwise the fix would be a - /// per-variant special case rather than one classification. + /// A variant with no native arm takes the SQLSTATE pgwire renders for it, + /// and keeps its numeric code. #[test] - fn errors_without_a_dedicated_sqlstate_still_carry_a_code() { + fn errors_without_a_native_arm_take_the_pgwire_sqlstate() { let response = error_to_native( 1, &crate::Error::PlanError { @@ -547,11 +678,33 @@ mod tests { let error = response .error .expect("error responses must carry a payload"); - assert_eq!(error.code, "XX000"); + assert_eq!(error.code, nodedb_types::error::sqlstate::SYNTAX_ERROR); assert_eq!( error.ndb_code, nodedb_types::error::ErrorCode::PLAN_ERROR.0, - "an unmapped SQLSTATE must not also erase the numeric classification" + "the numeric classification must survive the SQLSTATE rendering" + ); + } + + /// A constraint refusal that crossed a node boundary keeps its SQLSTATE. + #[test] + fn a_rejected_constraint_keeps_its_sqlstate() { + let response = error_to_native( + 1, + &crate::Error::RejectedConstraint { + collection: "c".to_owned(), + constraint: "unique".to_owned(), + detail: "duplicate key".to_owned(), + }, + ); + + let error = response + .error + .expect("error responses must carry a payload"); + assert_eq!(error.code, nodedb_types::error::sqlstate::UNIQUE_VIOLATION); + assert_eq!( + error.ndb_code, + nodedb_types::error::ErrorCode::CONSTRAINT_VIOLATION.0 ); } } diff --git a/nodedb/src/control/server/native/dispatch/ctx.rs b/nodedb/src/control/server/native/dispatch/ctx.rs index 69c8de36e..9b6245dd3 100644 --- a/nodedb/src/control/server/native/dispatch/ctx.rs +++ b/nodedb/src/control/server/native/dispatch/ctx.rs @@ -10,6 +10,9 @@ use crate::control::security::identity::AuthenticatedIdentity; use crate::control::security::request_scope::RequestAuthScope; use crate::control::server::shared::session::SessionStore; use crate::control::state::SharedState; +use nodedb_physical::physical_plan::{GraphOp, PhysicalPlan}; +use nodedb_types::CollectionKey; + use crate::types::{TenantId, VShardId}; /// Dispatch context: holds references needed by all handlers. @@ -52,7 +55,47 @@ impl DispatchCtx<'_> { self.scope.auth() } - pub(super) fn vshard_for_key(&self, key: &str) -> VShardId { - VShardId::from_key(key.as_bytes()) + /// The vShard a direct-op task carries. + /// + /// A graph plan is homed by node key: an edge write by its source node, + /// any other graph op by `document_id`, else the collection name. Every + /// other plan is homed by its collection's canonical key, the bare name + /// in this request's database. That is the vShard the planner, the + /// gateway and the staging gate use for the same collection. + pub(super) fn task_vshard( + &self, + plan: &PhysicalPlan, + document_id: Option<&str>, + collection: &str, + ) -> VShardId { + if let PhysicalPlan::Graph(op) = plan { + let node_key = match op { + GraphOp::EdgePut { src_id, .. } | GraphOp::EdgeDelete { src_id, .. } => { + src_id.as_str() + } + GraphOp::EdgePutBatch { .. } + | GraphOp::ResolveEdgeDelete(_) + | GraphOp::EdgeDeleteBatch { .. } + | GraphOp::Hop { .. } + | GraphOp::Neighbors { .. } + | GraphOp::NeighborsMulti { .. } + | GraphOp::Path { .. } + | GraphOp::Subgraph { .. } + | GraphOp::RagFusion { .. } + | GraphOp::Algo { .. } + | GraphOp::Match { .. } + | GraphOp::MatchContinuation { .. } + | GraphOp::MatchVarLenResume { .. } + | GraphOp::BspSuperstep(_) + | GraphOp::WccSuperstep(_) + | GraphOp::SetNodeLabels { .. } + | GraphOp::RemoveNodeLabels { .. } + | GraphOp::TemporalNeighbors { .. } + | GraphOp::TemporalAlgorithm { .. } + | GraphOp::Stats { .. } => document_id.unwrap_or(collection), + }; + return VShardId::from_key(node_key.as_bytes()); + } + CollectionKey::from_bare(self.database_id(), collection).vshard() } } diff --git a/nodedb/src/control/server/native/dispatch/direct_ops.rs b/nodedb/src/control/server/native/dispatch/direct_ops.rs index 73d7d7e86..5eece82c3 100644 --- a/nodedb/src/control/server/native/dispatch/direct_ops.rs +++ b/nodedb/src/control/server/native/dispatch/direct_ops.rs @@ -33,13 +33,12 @@ pub(crate) async fn handle_direct_op( .as_deref() .unwrap_or("default") .to_lowercase(); - let vshard_key = fields.document_id.as_deref().unwrap_or(&collection); - let vshard_id = ctx.vshard_for_key(vshard_key); let tenant_id = ctx.tenant_id(); - // CRDT Apply allocates a surrogate while planning; authorize the exact - // collection first. - if matches!(op, OpCode::CrdtApply) { + // CRDT Apply allocates a surrogate while planning, and a KV counter plans + // its fresh row from the catalog, which can evaluate a DEFAULT. Authorize + // the exact collection first. + if matches!(op, OpCode::CrdtApply | OpCode::KvIncr | OpCode::KvIncrFloat) { let audit = crate::control::security::audit::ArcAuditEmitter(std::sync::Arc::clone( &ctx.state.audit, )); @@ -70,6 +69,7 @@ pub(crate) async fn handle_direct_op( Ok(p) => p, Err(e) => return error_to_native_with_sqlstate(seq, "42601", &e), }; + let vshard_id = ctx.task_vshard(&plan, fields.document_id.as_deref(), &collection); // Apply RLS before any special Control-Plane orchestration can observe the plan. if let Err(e) = crate::control::planner::rls_injection::inject_rls_for_single_plan( @@ -336,7 +336,7 @@ pub(crate) async fn handle_direct_op( None => { return sqlstate_error( seq, - "XX000", + nodedb_types::error::sqlstate::INTERNAL_ERROR, "authorization returned no task capability", ); } @@ -358,12 +358,14 @@ pub(crate) async fn handle_direct_op( } // Only reads to widen with are those materialized-sum settlement stamped // on the source rows its shipped balances folded from. + let sum_read_vshards = + match crate::control::planner::calvin::read_vshards_of(&sum_target_reads) { + Ok(vshards) => vshards, + Err(error) => return error_to_native(seq, &error), + }; let route_to_calvin = !in_txn_block && matches!( - classify_dispatch( - &tasks, - &crate::control::planner::calvin::read_vshards_of(&sum_target_reads), - ), + classify_dispatch(&tasks, &sum_read_vshards), DispatchClass::MultiShard { .. } ); if route_to_calvin { diff --git a/nodedb/src/control/server/native/dispatch/graph_match.rs b/nodedb/src/control/server/native/dispatch/graph_match.rs index 401dcaddd..f659ee765 100644 --- a/nodedb/src/control/server/native/dispatch/graph_match.rs +++ b/nodedb/src/control/server/native/dispatch/graph_match.rs @@ -31,8 +31,6 @@ pub(crate) async fn handle_graph_match( .as_deref() .unwrap_or("default") .to_lowercase(); - let vshard_key = fields.document_id.as_deref().unwrap_or(&collection); - let vshard_id = ctx.vshard_for_key(vshard_key); let tenant_id = ctx.tenant_id(); if let Err(error) = super::limits::check_op_limits(ctx.state, fields) { @@ -47,6 +45,7 @@ pub(crate) async fn handle_graph_match( Ok(plan) => plan, Err(error) => return error_to_native_with_sqlstate(seq, "42601", &error), }; + let vshard_id = ctx.task_vshard(&plan, fields.document_id.as_deref(), &collection); if let Err(error) = crate::control::planner::rls_injection::inject_rls_for_single_plan( tenant_id.as_u64(), ctx.database_id(), diff --git a/nodedb/src/control/server/native/dispatch/index_ddl_op.rs b/nodedb/src/control/server/native/dispatch/index_ddl_op.rs new file mode 100644 index 000000000..5b21a87d3 --- /dev/null +++ b/nodedb/src/control/server/native/dispatch/index_ddl_op.rs @@ -0,0 +1,393 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Native index-DDL opcodes run the SQL DDL statements they name. +//! +//! | opcode | SQL | +//! |-------------------------|--------------------------------------------------| +//! | `KvRegisterSortedIndex` | `CREATE SORTED INDEX` | +//! | `KvDropSortedIndex` | `DROP SORTED INDEX` | +//! | `VectorSetParams` | `ALTER VECTOR INDEX`, or `CREATE VECTOR INDEX` | +//! | `KvRegisterIndex` | `CREATE INDEX IF NOT EXISTS ON c (field)` | +//! | `KvDropIndex` | `DROP INDEX` of the index on the field | +//! | `DocumentDropIndex` | `DROP INDEX` of the index on the field | +//! | `DocumentRegister` | `CREATE COLLECTION IF NOT EXISTS`, then one `CREATE INDEX IF NOT EXISTS` per index path | +//! +//! The opcode then reaches the same catalog entries, engine steps and +//! per-connection transaction buffer as the SQL statements. The index is in +//! the catalog, survives a restart, and is listed by `SHOW INDEXES`. Inside +//! an explicit transaction it is visible to later statements, applies at +//! COMMIT and is discarded on ROLLBACK. +//! +//! `DocumentRegister` runs its statements in order and stops at the first +//! error. Each statement is `IF NOT EXISTS`, so a retry after a partial run +//! completes it. +//! +//! The opcode answers like any other opcode: `Ok` with no status row, count +//! or verb on success, the first failing statement's error on failure. +//! +//! Every rendered identifier passes the SQL identifier rules and is a plain +//! token of ASCII letters, digits and underscores, so no field value can +//! change the statement's shape. + +use nodedb_types::protocol::{NativeResponse, OpCode, TextFields}; + +use super::{DispatchCtx, error_to_native_with_sqlstate, handle_sql}; + +/// Run the index-DDL opcode `op` as the SQL statement it names. +pub(crate) async fn handle_index_ddl_op( + ctx: &DispatchCtx<'_>, + seq: u64, + op: OpCode, + fields: &TextFields, +) -> NativeResponse { + let collection = fields + .collection + .as_deref() + .unwrap_or("default") + .to_lowercase(); + let statements = match index_ddl_statements(ctx, op, fields, &collection) { + Ok(statements) => statements, + Err(error) => return error_to_native_with_sqlstate(seq, "42601", &error), + }; + for sql in &statements { + let response = handle_sql(ctx, seq, sql, None).await; + // A failure keeps the statement's error. + if response.status != nodedb_types::protocol::ResponseStatus::Ok { + return response; + } + } + // An opcode answers as every opcode does: success carries no DDL status + // row, no count and no verb. + NativeResponse::ok(seq) +} + +/// The DDL statements opcode `op` names, in the order they run. +fn index_ddl_statements( + ctx: &DispatchCtx<'_>, + op: OpCode, + fields: &TextFields, + collection: &str, +) -> crate::Result> { + match op { + OpCode::KvRegisterSortedIndex => Ok(vec![create_sorted_index(fields, collection)?]), + OpCode::KvDropSortedIndex => { + let name = ident(required(fields.index_name.as_deref(), "index_name")?)?; + Ok(vec![format!("DROP SORTED INDEX {name}")]) + } + OpCode::VectorSetParams => Ok(vec![vector_set_params(ctx, fields, collection)?]), + OpCode::KvRegisterIndex => Ok(vec![kv_register_index(fields, collection)?]), + OpCode::KvDropIndex | OpCode::DocumentDropIndex => { + Ok(vec![drop_index_on_field(ctx, fields, collection)?]) + } + OpCode::DocumentRegister => document_register(fields, collection), + other => Err(bad_request(format!( + "opcode {other:?} is not an index DDL opcode" + ))), + } +} + +/// `KvRegisterIndex` indexes one field of the collection, backfilled from +/// every row it holds. +/// +/// `backfill = false` is refused: a catalog index answers lookups for every +/// row, and an index that skipped the existing rows would miss them. +fn kv_register_index(fields: &TextFields, collection: &str) -> crate::Result { + if fields.backfill == Some(false) { + return Err(bad_request( + "KvRegisterIndex with backfill = false is not supported: an index covers every \ + row, so it is built from the rows the collection already holds", + )); + } + let collection = ident(collection)?; + let field = index_field(required(fields.field.as_deref(), "field")?)?; + Ok(format!( + "CREATE INDEX IF NOT EXISTS ON {collection} ({field})" + )) +} + +/// `DocumentRegister` makes sure the document collection exists, then +/// indexes each of its index paths. +fn document_register(fields: &TextFields, collection: &str) -> crate::Result> { + let collection = ident(collection)?; + let mut statements = vec![format!( + "CREATE COLLECTION IF NOT EXISTS {collection} WITH (engine='document_schemaless')" + )]; + for path in fields.index_paths.as_deref().unwrap_or_default() { + let field = index_field(path)?; + statements.push(format!( + "CREATE INDEX IF NOT EXISTS ON {collection} ({field})" + )); + } + Ok(statements) +} + +/// A top-level field named bare or as `$.field`. +fn index_field(raw: &str) -> crate::Result { + ident(raw.strip_prefix("$.").unwrap_or(raw)) +} + +fn create_sorted_index(fields: &TextFields, collection: &str) -> crate::Result { + let name = ident(required(fields.index_name.as_deref(), "index_name")?)?; + let collection = ident(collection)?; + let columns = fields + .sort_columns + .as_deref() + .filter(|columns| !columns.is_empty()) + .ok_or_else(|| bad_request("missing 'sort_columns'"))? + .iter() + .map(|(column, direction)| { + let direction = direction.to_ascii_uppercase(); + if direction != "ASC" && direction != "DESC" { + return Err(bad_request(format!( + "invalid sort direction '{direction}', expected ASC or DESC" + ))); + } + Ok(format!("{} {direction}", ident(column)?)) + }) + .collect::>>()? + .join(", "); + let key = ident(required(fields.key_column.as_deref(), "key_column")?)?; + let mut sql = format!("CREATE SORTED INDEX {name} ON {collection} ({columns}) KEY {key}"); + match fields.window_type.as_deref().map(str::to_ascii_lowercase) { + None => {} + Some(window) if window == "none" => {} + Some(window) if matches!(window.as_str(), "daily" | "weekly" | "monthly") => { + sql.push_str(&format!(" WINDOW {}", window.to_ascii_uppercase())); + if let Some(column) = fields.window_timestamp_column.as_deref() { + sql.push_str(&format!(" ON {}", ident(column)?)); + } + } + Some(window) if window == "custom" => { + sql.push_str(&format!( + " WINDOW CUSTOM START {} END {}", + fields.window_start_ms.unwrap_or(0), + fields.window_end_ms.unwrap_or(0) + )); + if let Some(column) = fields.window_timestamp_column.as_deref() { + sql.push_str(&format!(" ON {}", ident(column)?)); + } + } + Some(window) => { + return Err(bad_request(format!( + "invalid window type '{window}', expected none, daily, weekly, monthly or \ + custom" + ))); + } + } + Ok(sql) +} + +/// `VectorSetParams` alters the vector index the column already carries, or +/// creates one when it has none. +fn vector_set_params( + ctx: &DispatchCtx<'_>, + fields: &TextFields, + collection: &str, +) -> crate::Result { + let collection = ident(collection)?; + let column = match fields.field_name.as_deref().filter(|f| !f.is_empty()) { + Some(column) => Some(ident(column)?), + None => None, + }; + let index_type = fields.index_type.as_deref().map(plain_token).transpose()?; + let metric = fields.metric.as_deref().map(plain_token).transpose()?; + let existing = ctx.state.credentials.catalog().get_vector_index_params( + ctx.database_id().as_u64(), + ctx.tenant_id().as_u64(), + &collection, + column.as_deref().unwrap_or(""), + )?; + + if let Some(existing) = existing { + if metric.as_deref().is_some_and(|m| m != existing.metric) { + return Err(bad_request(format!( + "the vector index on '{collection}' uses metric '{}'; a metric change needs \ + a new index", + existing.metric + ))); + } + let mut set = Vec::new(); + if let Some(m) = fields.m { + set.push(format!("m = {m}")); + } + if let Some(ef) = fields.ef_construction { + set.push(format!("ef_construction = {ef}")); + } + if let Some(index_type) = &index_type { + set.push(format!("index_type = {index_type}")); + } + if set.is_empty() { + return Err(bad_request( + "VectorSetParams on an existing index needs m, ef_construction or index_type", + )); + } + let target = match &column { + Some(column) => format!("{collection}.{column}"), + None => collection.clone(), + }; + return Ok(format!( + "ALTER VECTOR INDEX ON {target} SET ({})", + set.join(", ") + )); + } + + let name = match fields.index_name.as_deref() { + Some(name) => ident(name)?, + None => match &column { + Some(column) => ident(&format!("vec_{collection}_{column}"))?, + None => ident(&format!("vec_{collection}"))?, + }, + }; + let mut sql = format!("CREATE VECTOR INDEX {name} ON {collection}"); + if let Some(column) = &column { + sql.push_str(&format!(" ({column})")); + } + sql.push_str(&format!(" DIM {}", fields.vector_dim.unwrap_or(0))); + if let Some(metric) = &metric { + sql.push_str(&format!(" METRIC {metric}")); + } + if let Some(m) = fields.m { + sql.push_str(&format!(" M {m}")); + } + if let Some(ef) = fields.ef_construction { + sql.push_str(&format!(" EF_CONSTRUCTION {ef}")); + } + if let Some(index_type) = &index_type { + sql.push_str(&format!(" INDEX_TYPE {index_type}")); + } + Ok(sql) +} + +/// `KvDropIndex` and `DocumentDropIndex` name a field; the SQL statement +/// names the index on it. +fn drop_index_on_field( + ctx: &DispatchCtx<'_>, + fields: &TextFields, + collection: &str, +) -> crate::Result { + let field = required(fields.field.as_deref(), "field")?; + let path = if field.starts_with('$') { + field.to_string() + } else { + format!("$.{field}") + }; + let stored = ctx + .state + .credentials + .catalog() + .get_collection(ctx.database_id(), ctx.tenant_id().as_u64(), collection)? + .ok_or_else(|| crate::Error::CollectionNotFound { + tenant_id: ctx.tenant_id(), + collection: collection.to_string(), + })?; + let index = stored + .indexes + .iter() + .find(|index| index.field == path) + .ok_or_else(|| bad_request(format!("no index on field '{field}' of '{collection}'")))?; + Ok(format!("DROP INDEX {}", ident(&index.name)?)) +} + +fn required<'a>(value: Option<&'a str>, name: &str) -> crate::Result<&'a str> { + value + .filter(|v| !v.is_empty()) + .ok_or_else(|| bad_request(format!("missing '{name}'"))) +} + +/// A SQL identifier that renders without quoting. +fn ident(raw: &str) -> crate::Result { + let normalized = + nodedb_sql::reserved::check_identifier(raw).map_err(|e| bad_request(e.to_string()))?; + plain_token(&normalized) +} + +/// A token of ASCII letters, digits and underscores. +fn plain_token(raw: &str) -> crate::Result { + let plain = !raw.is_empty() && raw.bytes().all(|b| b.is_ascii_alphanumeric() || b == b'_'); + if plain { + Ok(raw.to_string()) + } else { + Err(bad_request(format!( + "'{raw}' must be letters, digits and underscores" + ))) + } +} + +fn bad_request(detail: impl Into) -> crate::Error { + crate::Error::BadRequest { + detail: detail.into(), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_sorted_index_opcode_renders_its_statement() { + let fields = TextFields { + index_name: Some("lb".into()), + sort_columns: Some(vec![("score".into(), "desc".into())]), + key_column: Some("player".into()), + window_type: Some("daily".into()), + window_timestamp_column: Some("ts".into()), + ..TextFields::default() + }; + assert_eq!( + create_sorted_index(&fields, "scores").expect("renders"), + "CREATE SORTED INDEX lb ON scores (score DESC) KEY player WINDOW DAILY ON ts" + ); + } + + #[test] + fn a_kv_index_opcode_renders_an_idempotent_create() { + let fields = TextFields { + field: Some("$.region".into()), + ..TextFields::default() + }; + assert_eq!( + kv_register_index(&fields, "sessions").expect("renders"), + "CREATE INDEX IF NOT EXISTS ON sessions (region)" + ); + let skip_backfill = TextFields { + field: Some("region".into()), + backfill: Some(false), + ..TextFields::default() + }; + assert!(kv_register_index(&skip_backfill, "sessions").is_err()); + } + + #[test] + fn a_register_opcode_creates_the_collection_then_each_index() { + let fields = TextFields { + index_paths: Some(vec!["$.region".into(), "status".into()]), + ..TextFields::default() + }; + assert_eq!( + document_register(&fields, "orders").expect("renders"), + vec![ + "CREATE COLLECTION IF NOT EXISTS orders WITH (engine='document_schemaless')" + .to_string(), + "CREATE INDEX IF NOT EXISTS ON orders (region)".to_string(), + "CREATE INDEX IF NOT EXISTS ON orders (status)".to_string(), + ] + ); + let nested = TextFields { + index_paths: Some(vec!["$.a.b".into()]), + ..TextFields::default() + }; + assert!(document_register(&nested, "orders").is_err()); + } + + #[test] + fn a_field_that_would_change_the_statement_is_refused() { + let fields = TextFields { + index_name: Some("lb; DROP".into()), + sort_columns: Some(vec![("score".into(), "DESC".into())]), + key_column: Some("player".into()), + ..TextFields::default() + }; + assert!(create_sorted_index(&fields, "scores").is_err()); + assert!(plain_token("cosine) KEY x").is_err()); + } +} diff --git a/nodedb/src/control/server/native/dispatch/mod.rs b/nodedb/src/control/server/native/dispatch/mod.rs index 344088f22..14cd8c68b 100644 --- a/nodedb/src/control/server/native/dispatch/mod.rs +++ b/nodedb/src/control/server/native/dispatch/mod.rs @@ -10,12 +10,14 @@ mod ctx; mod direct_ops; mod edge_recon_gate; mod graph_match; +mod index_ddl_op; mod limits; mod plan_builder; pub(crate) mod raw_dispatch; pub(crate) mod response; mod session_ops; mod single_task; +mod sorted_read_op; mod sql; mod sql_admin; mod sql_dispatch_task; @@ -29,13 +31,16 @@ pub(crate) use admission_op::admission_operation; pub(crate) use auth::{NativeAuthOutcome, handle_auth, handle_ping}; pub(crate) use conversion::{ apply_dml_outcome, ddl_result_to_native, dml_fold_error_to_native, error_code_to_native, - error_response_to_native, error_to_native, error_to_native_with_sqlstate, - shape_error_to_native, to_native_columns_rows, + error_response_to_native, error_to_native, error_to_native_in_context, + error_to_native_with_sqlstate, native_error_fields, shape_error_to_native, + to_native_columns_rows, }; pub(crate) use ctx::DispatchCtx; pub(crate) use direct_ops::handle_direct_op; pub(crate) use graph_match::handle_graph_match; +pub(crate) use index_ddl_op::handle_index_ddl_op; pub(crate) use session_ops::{handle_reset, handle_set, handle_show, show_all}; +pub(crate) use sorted_read_op::handle_sorted_read_op; pub(crate) use sql::{handle_sql, handle_sql_streaming}; pub(crate) use streaming::{SqlOutcome, SqlStream}; pub(crate) use transaction::{NativeTxnDp, handle_begin, handle_commit, handle_rollback}; diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/columnar.rs b/nodedb/src/control/server/native/dispatch/plan_builder/columnar.rs index 89e0de06e..5f1d19d5b 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/columnar.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/columnar.rs @@ -97,7 +97,11 @@ fn derive_surrogates( if pk.is_empty() { out.push(Surrogate::ZERO); } else { - out.push(assigner.assign(ctx.database_id(), ctx.tenant_id(), collection, &pk)?); + out.push(assigner.assign( + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + &pk, + )?); } } Ok(out) diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/crdt.rs b/nodedb/src/control/server/native/dispatch/plan_builder/crdt.rs index 9da6c3706..da2c204c9 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/crdt.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/crdt.rs @@ -42,9 +42,8 @@ pub(crate) fn build_apply( }); let surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, document_id.as_bytes(), )?; @@ -144,9 +143,8 @@ pub(crate) fn build_list_insert( })?; let surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, document_id.as_bytes(), )?; @@ -170,9 +168,8 @@ pub(crate) fn build_list_delete( let index = require_list_index(fields.list_index, "list_index")?; let surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, document_id.as_bytes(), )?; @@ -196,9 +193,8 @@ pub(crate) fn build_list_move( let to_index = require_list_index(fields.list_to_index, "list_to_index")?; let surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, document_id.as_bytes(), )?; diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/dispatch.rs b/nodedb/src/control/server/native/dispatch/plan_builder/dispatch.rs index fc2b5b2bb..f86701665 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/dispatch.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/dispatch.rs @@ -8,7 +8,10 @@ use nodedb_types::protocol::{OpCode, TextFields}; use crate::bridge::envelope::PhysicalPlan; use super::super::DispatchCtx; -use super::{columnar, crdt, document, graph, kv, query, spatial, text, timeseries, vector}; +use super::{ + columnar, crdt, document, document_bulk, graph, kv, kv_counter, query, spatial, text, + timeseries, vector, +}; /// Build a PhysicalPlan from an opcode and request fields. pub(crate) fn build_plan( @@ -27,8 +30,8 @@ pub(crate) fn build_plan( OpCode::DocumentUpdate => document::build_update(ctx, fields, collection), OpCode::DocumentScan => document::build_scan(ctx, fields, collection), OpCode::DocumentUpsert => document::build_upsert(ctx, fields, collection), - OpCode::DocumentBulkUpdate => document::build_bulk_update(ctx, fields, collection), - OpCode::DocumentBulkDelete => document::build_bulk_delete(ctx, fields, collection), + OpCode::DocumentBulkUpdate => document_bulk::build_bulk_update(ctx, fields, collection), + OpCode::DocumentBulkDelete => document_bulk::build_bulk_delete(ctx, fields, collection), // Vector. OpCode::VectorSearch => vector::build_search(ctx, fields, collection), OpCode::VectorBatchInsert => vector::build_batch_insert(ctx, fields, collection), @@ -74,30 +77,18 @@ pub(crate) fn build_plan( OpCode::GraphAlgo => graph::build_algo(fields, collection), OpCode::GraphMatch => graph::build_match(fields, collection), // Document DDL. - OpCode::DocumentTruncate => document::build_truncate(ctx, collection), - OpCode::DocumentEstimateCount => document::build_estimate_count(ctx, fields, collection), - OpCode::DocumentInsertSelect => document::build_insert_select(ctx, fields, collection), - OpCode::DocumentRegister => document::build_register(ctx, fields, collection), - OpCode::DocumentDropIndex => document::build_drop_index(ctx, fields, collection), - // KV DDL. - OpCode::KvRegisterIndex => kv::build_register_index(ctx, fields, collection), - OpCode::KvDropIndex => kv::build_drop_index(ctx, fields, collection), + OpCode::DocumentTruncate => document_bulk::build_truncate(ctx, collection), + OpCode::DocumentEstimateCount => { + document_bulk::build_estimate_count(ctx, fields, collection) + } + OpCode::DocumentInsertSelect => document_bulk::build_insert_select(ctx, fields, collection), + // KV truncate. OpCode::KvTruncate => kv::build_truncate(ctx, collection), // KV atomic operations. - OpCode::KvIncr => kv::build_incr(ctx, collection, fields), - OpCode::KvIncrFloat => kv::build_incr_float(ctx, collection, fields), + OpCode::KvIncr => kv_counter::build_incr(ctx, collection, fields), + OpCode::KvIncrFloat => kv_counter::build_incr_float(ctx, collection, fields), OpCode::KvCas => kv::build_cas(ctx, collection, fields), OpCode::KvGetSet => kv::build_getset(ctx, collection, fields), - // KV sorted index operations. - OpCode::KvRegisterSortedIndex => kv::build_register_sorted_index(ctx, collection, fields), - OpCode::KvDropSortedIndex => kv::build_drop_sorted_index(fields), - OpCode::KvSortedIndexRank => kv::build_sorted_index_rank(fields), - OpCode::KvSortedIndexTopK => kv::build_sorted_index_top_k(fields), - OpCode::KvSortedIndexRange => kv::build_sorted_index_range(fields), - OpCode::KvSortedIndexCount => kv::build_sorted_index_count(fields), - OpCode::KvSortedIndexScore => kv::build_sorted_index_score(fields), - // Vector DDL. - OpCode::VectorSetParams => vector::build_set_params(ctx, fields, collection), // Query. OpCode::RecursiveScan => query::build_recursive_scan(ctx, fields, collection), _ => Err(crate::Error::BadRequest { diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/document.rs b/nodedb/src/control/server/native/dispatch/plan_builder/document.rs index 0c2b6b3a3..4916459b3 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/document.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/document.rs @@ -44,7 +44,11 @@ pub(crate) fn build_point_get( let surrogate = ctx .state .surrogate_assigner - .lookup(ctx.database_id(), ctx.tenant_id(), collection, &pk_bytes)? + .lookup( + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + &pk_bytes, + )? .unwrap_or(nodedb_types::Surrogate::ZERO); Ok(PhysicalPlan::Document(DocumentOp::PointGet { collection: QualifiedCollection::new(ctx.database_id(), collection), @@ -70,9 +74,8 @@ pub(crate) fn build_point_put( Some(CollectionType::KeyValue(_)) => { let key = doc_id.into_bytes(); let surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, &key, )?; Ok(PhysicalPlan::Kv(KvOp::Put { @@ -83,17 +86,24 @@ pub(crate) fn build_point_put( surrogate, returning: None, rls_filters: Vec::new(), + provenance: None, })) } Some(CollectionType::Columnar(ColumnarProfile::Timeseries { .. })) => { let json_str = String::from_utf8_lossy(&value); let ilp_line = format!("{collection} value={json_str}\n"); + // The line's own surrogate keys its staged row, so a read later in + // the same transaction observes it. + let (surrogate, _identity) = ctx.state.surrogate_assigner.assign_fresh( + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + )?; Ok(PhysicalPlan::Timeseries(TimeseriesOp::Ingest { collection: QualifiedCollection::new(ctx.database_id(), collection), payload: ilp_line.into_bytes(), format: "ilp".to_string(), wal_lsn: None, - surrogates: Vec::new(), + surrogates: vec![surrogate], provenance: None, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: None, @@ -108,9 +118,8 @@ pub(crate) fn build_point_put( Some(CollectionType::Document(_)) | None => { let pk_bytes = doc_id.as_bytes().to_vec(); let surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, &pk_bytes, )?; Ok(PhysicalPlan::Document(DocumentOp::PointPut { @@ -144,6 +153,7 @@ pub(crate) fn build_point_delete( // The native point-delete carries no RETURNING clause. returning: None, rls_filters: Vec::new(), + provenance: None, })), Some(CollectionType::Columnar(ColumnarProfile::Timeseries { .. })) => { Err(crate::Error::BadRequest { @@ -162,7 +172,11 @@ pub(crate) fn build_point_delete( let surrogate = ctx .state .surrogate_assigner - .lookup(ctx.database_id(), ctx.tenant_id(), collection, &pk_bytes)? + .lookup( + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + &pk_bytes, + )? .unwrap_or(nodedb_types::Surrogate::ZERO); Ok(PhysicalPlan::Document(DocumentOp::PointDelete { collection: QualifiedCollection::new(ctx.database_id(), collection), @@ -225,9 +239,8 @@ pub(crate) fn build_batch_insert( detail: format!("failed to serialize document '{}': {e}", d.id), })?; let surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, d.id.as_bytes(), )?; documents.push((d.id.clone(), value_bytes)); @@ -269,7 +282,11 @@ pub(crate) fn build_update( let surrogate = ctx .state .surrogate_assigner - .lookup(ctx.database_id(), ctx.tenant_id(), collection, &pk_bytes)? + .lookup( + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + &pk_bytes, + )? .unwrap_or(nodedb_types::Surrogate::ZERO); Ok(PhysicalPlan::Document(DocumentOp::PointUpdate { collection: QualifiedCollection::new(ctx.database_id(), collection), @@ -333,9 +350,8 @@ pub(crate) fn build_upsert( let doc_id = require_doc_id(fields)?; let value = fields.data.clone().unwrap_or_default(); let surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, doc_id.as_bytes(), )?; Ok(PhysicalPlan::Document(DocumentOp::Upsert { @@ -353,183 +369,3 @@ pub(crate) fn build_upsert( resolved_sum_targets: Vec::new(), })) } - -pub(crate) fn build_bulk_update( - ctx: &DispatchCtx<'_>, - fields: &TextFields, - collection: &str, -) -> crate::Result { - let filters = fields - .filters - .as_ref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'filters'".to_string(), - })? - .clone(); - let updates: Vec<(String, nodedb_physical::physical_plan::UpdateValue)> = fields - .updates - .as_ref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'updates'".to_string(), - })? - .iter() - .map(|(f, b)| { - ( - f.clone(), - nodedb_physical::physical_plan::UpdateValue::Literal(b.clone()), - ) - }) - .collect(); - Ok(PhysicalPlan::Document(DocumentOp::BulkUpdate { - collection: QualifiedCollection::new(ctx.database_id(), collection), - filters, - updates, - returning: None, - ollp_predicted_surrogates: None, - ollp_predicted_edges: None, - rls_filters: Vec::new(), - rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), - // Filled in by the materialized-sum resolution pass. - resolved_sum_targets: Vec::new(), - // See `build_update`: reads the declared PRIMARY KEY from the catalog. - declared_primary_key: declared_primary_key(ctx, collection)?, - })) -} - -pub(crate) fn build_bulk_delete( - ctx: &DispatchCtx<'_>, - fields: &TextFields, - collection: &str, -) -> crate::Result { - let filters = fields - .filters - .as_ref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'filters'".to_string(), - })? - .clone(); - Ok(PhysicalPlan::Document(DocumentOp::BulkDelete { - collection: QualifiedCollection::new(ctx.database_id(), collection), - filters, - returning: None, - ollp_predicted_surrogates: None, - ollp_predicted_edges: None, - rls_filters: Vec::new(), - rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), - // Filled in by the materialized-sum resolution pass. - resolved_sum_targets: Vec::new(), - // See `build_update`: reads the declared PRIMARY KEY from the catalog. - declared_primary_key: declared_primary_key(ctx, collection)?, - })) -} - -pub(crate) fn build_truncate( - ctx: &DispatchCtx<'_>, - collection: &str, -) -> crate::Result { - Ok(PhysicalPlan::Document(DocumentOp::Truncate { - collection: QualifiedCollection::new(ctx.database_id(), collection), - restart_identity: false, - // Filled in by the materialized-sum resolution pass. - resolved_sum_targets: Vec::new(), - // See `build_update`: reads the declared PRIMARY KEY from the catalog. - declared_primary_key: declared_primary_key(ctx, collection)?, - })) -} - -pub(crate) fn build_estimate_count( - ctx: &DispatchCtx<'_>, - fields: &TextFields, - collection: &str, -) -> crate::Result { - let field = fields.field.as_deref().unwrap_or("id").to_string(); - - Ok(PhysicalPlan::Document(DocumentOp::EstimateCount { - collection: QualifiedCollection::new(ctx.database_id(), collection), - field, - })) -} - -pub(crate) fn build_insert_select( - ctx: &DispatchCtx<'_>, - fields: &TextFields, - collection: &str, -) -> crate::Result { - let source = fields - .source_collection - .as_ref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'source_collection'".to_string(), - })? - .clone(); - let filters = fields.filters.clone().unwrap_or_default(); - let limit = fields.limit.unwrap_or(10_000) as usize; - - Ok(PhysicalPlan::Document(DocumentOp::InsertSelect { - target_collection: QualifiedCollection::new(ctx.database_id(), collection), - source_collection: QualifiedCollection::new(ctx.database_id(), &source), - source_filters: filters, - source_limit: limit, - // The native text-field form names no projection, so every source row - // copies unchanged. - column_map: Vec::new(), - })) -} - -pub(crate) fn build_register( - ctx: &DispatchCtx<'_>, - fields: &TextFields, - collection: &str, -) -> crate::Result { - // Native protocol exposes only a legacy `index_paths` text list — promote - // each entry to a `Ready`, non-unique `RegisteredIndex` named after the - // path. UNIQUE / COLLATE / build-state come from SQL DDL only. - let indexes = fields - .index_paths - .clone() - .unwrap_or_default() - .into_iter() - .map(|path| nodedb_physical::physical_plan::RegisteredIndex { - name: path.clone(), - path, - unique: false, - case_insensitive: false, - state: nodedb_physical::physical_plan::RegisteredIndexState::Ready, - predicate: None, - }) - .collect(); - - Ok(PhysicalPlan::Document(DocumentOp::Register { - collection: QualifiedCollection::new(ctx.database_id(), collection), - indexes, - crdt_enabled: false, - storage_mode: nodedb_physical::physical_plan::StorageMode::Schemaless, - enforcement: Box::new(nodedb_physical::physical_plan::EnforcementOptions::default()), - bitemporal: false, - conflict_policy: None, - timeseries: None, - // The native protocol's register frame carries only index paths; a - // vector-primary collection is created through SQL DDL, which goes - // through the catalog-sourced builder instead. - vector_primary: None, - })) -} - -pub(crate) fn build_drop_index( - ctx: &DispatchCtx<'_>, - fields: &TextFields, - collection: &str, -) -> crate::Result { - let field = fields - .field - .as_ref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'field'".to_string(), - })? - .clone(); - - Ok(PhysicalPlan::Document(DocumentOp::DropIndex { - collection: QualifiedCollection::new(ctx.database_id(), collection), - field, - })) -} diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/document_bulk.rs b/nodedb/src/control/server/native/dispatch/plan_builder/document_bulk.rs new file mode 100644 index 000000000..45681ae25 --- /dev/null +++ b/nodedb/src/control/server/native/dispatch/plan_builder/document_bulk.rs @@ -0,0 +1,135 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Document engine plan builders for set-level writes: predicate update and +//! delete, truncate, insert-select, and the row-count estimate. + +use nodedb_types::QualifiedCollection; +use nodedb_types::protocol::TextFields; + +use crate::bridge::envelope::PhysicalPlan; +use nodedb_physical::physical_plan::DocumentOp; + +use super::super::DispatchCtx; +use super::declared_primary_key; + +pub(crate) fn build_bulk_update( + ctx: &DispatchCtx<'_>, + fields: &TextFields, + collection: &str, +) -> crate::Result { + let filters = fields + .filters + .as_ref() + .ok_or_else(|| crate::Error::BadRequest { + detail: "missing 'filters'".to_string(), + })? + .clone(); + let updates: Vec<(String, nodedb_physical::physical_plan::UpdateValue)> = fields + .updates + .as_ref() + .ok_or_else(|| crate::Error::BadRequest { + detail: "missing 'updates'".to_string(), + })? + .iter() + .map(|(f, b)| { + ( + f.clone(), + nodedb_physical::physical_plan::UpdateValue::Literal(b.clone()), + ) + }) + .collect(); + Ok(PhysicalPlan::Document(DocumentOp::BulkUpdate { + collection: QualifiedCollection::new(ctx.database_id(), collection), + filters, + updates, + returning: None, + ollp_predicted_surrogates: None, + ollp_predicted_edges: None, + rls_filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // Filled in by the materialized-sum resolution pass. + resolved_sum_targets: Vec::new(), + // See `build_update`: reads the declared PRIMARY KEY from the catalog. + declared_primary_key: declared_primary_key(ctx, collection)?, + })) +} + +pub(crate) fn build_bulk_delete( + ctx: &DispatchCtx<'_>, + fields: &TextFields, + collection: &str, +) -> crate::Result { + let filters = fields + .filters + .as_ref() + .ok_or_else(|| crate::Error::BadRequest { + detail: "missing 'filters'".to_string(), + })? + .clone(); + Ok(PhysicalPlan::Document(DocumentOp::BulkDelete { + collection: QualifiedCollection::new(ctx.database_id(), collection), + filters, + returning: None, + ollp_predicted_surrogates: None, + ollp_predicted_edges: None, + rls_filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // Filled in by the materialized-sum resolution pass. + resolved_sum_targets: Vec::new(), + // See `build_update`: reads the declared PRIMARY KEY from the catalog. + declared_primary_key: declared_primary_key(ctx, collection)?, + })) +} + +pub(crate) fn build_truncate( + ctx: &DispatchCtx<'_>, + collection: &str, +) -> crate::Result { + Ok(PhysicalPlan::Document(DocumentOp::Truncate { + collection: QualifiedCollection::new(ctx.database_id(), collection), + restart_identity: false, + // Filled in by the materialized-sum resolution pass. + resolved_sum_targets: Vec::new(), + // See `build_update`: reads the declared PRIMARY KEY from the catalog. + declared_primary_key: declared_primary_key(ctx, collection)?, + })) +} + +pub(crate) fn build_estimate_count( + ctx: &DispatchCtx<'_>, + fields: &TextFields, + collection: &str, +) -> crate::Result { + let field = fields.field.as_deref().unwrap_or("id").to_string(); + + Ok(PhysicalPlan::Document(DocumentOp::EstimateCount { + collection: QualifiedCollection::new(ctx.database_id(), collection), + field, + })) +} + +pub(crate) fn build_insert_select( + ctx: &DispatchCtx<'_>, + fields: &TextFields, + collection: &str, +) -> crate::Result { + let source = fields + .source_collection + .as_ref() + .ok_or_else(|| crate::Error::BadRequest { + detail: "missing 'source_collection'".to_string(), + })? + .clone(); + let filters = fields.filters.clone().unwrap_or_default(); + let limit = fields.limit.unwrap_or(10_000) as usize; + + Ok(PhysicalPlan::Document(DocumentOp::InsertSelect { + target_collection: QualifiedCollection::new(ctx.database_id(), collection), + source_collection: QualifiedCollection::new(ctx.database_id(), &source), + source_filters: filters, + source_limit: limit, + // The native text-field form names no projection, so every source row + // copies unchanged. + column_map: Vec::new(), + })) +} diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/graph.rs b/nodedb/src/control/server/native/dispatch/plan_builder/graph.rs index 2408807fa..a69c0c785 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/graph.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/graph.rs @@ -193,15 +193,13 @@ pub(crate) fn build_edge_put( None => String::new(), }; let src_surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, src.as_bytes(), )?; let dst_surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, dst.as_bytes(), )?; Ok(PhysicalPlan::Graph(GraphOp::EdgePut { @@ -247,15 +245,13 @@ pub(crate) fn build_edge_delete( // returns the existing node identities) so a cross-shard delete dual-homes // and locks against a concurrent insert of the same edge. let src_surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, src.as_bytes(), )?; let dst_surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, dst.as_bytes(), )?; Ok(PhysicalPlan::Graph(GraphOp::EdgeDelete { diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs b/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs index c0c0282fb..14232fafe 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs @@ -207,48 +207,6 @@ fn require_key_bytes(fields: &TextFields) -> crate::Result> { }) } -pub(crate) fn build_register_index( - ctx: &DispatchCtx<'_>, - fields: &TextFields, - collection: &str, -) -> crate::Result { - let field = fields - .field - .as_ref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'field'".to_string(), - })? - .clone(); - let field_position = fields.field_position.unwrap_or(0) as usize; - let backfill = fields.backfill.unwrap_or(true); - - Ok(PhysicalPlan::Kv(KvOp::RegisterIndex { - collection: QualifiedCollection::new(ctx.database_id(), collection), - field, - field_position, - backfill, - })) -} - -pub(crate) fn build_drop_index( - ctx: &DispatchCtx<'_>, - fields: &TextFields, - collection: &str, -) -> crate::Result { - let field = fields - .field - .as_ref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'field'".to_string(), - })? - .clone(); - - Ok(PhysicalPlan::Kv(KvOp::DropIndex { - collection: QualifiedCollection::new(ctx.database_id(), collection), - field, - })) -} - pub(crate) fn build_truncate( ctx: &DispatchCtx<'_>, collection: &str, @@ -262,62 +220,16 @@ pub(crate) fn build_truncate( /// Resolve the stable cross-engine surrogate for a KV atomic op, content- /// addressed on `(collection, key)` — the same binding a normal insert of that /// key allocated, so an atomic op on an existing key keeps its identity. -fn assign_kv_surrogate( +pub(super) fn assign_kv_surrogate( ctx: &DispatchCtx<'_>, collection: &str, key: &[u8], ) -> crate::Result { - ctx.state - .surrogate_assigner - .assign(ctx.database_id(), ctx.tenant_id(), collection, key) -} - -pub(crate) fn build_incr( - ctx: &DispatchCtx<'_>, - collection: &str, - fields: &TextFields, -) -> crate::Result { - let key = fields - .key - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'key'".to_string(), - })?; - let delta = fields.incr_delta.unwrap_or(1); - let ttl_ms = fields.ttl_ms.unwrap_or(0); - let surrogate = assign_kv_surrogate(ctx, collection, key.as_bytes())?; - - Ok(PhysicalPlan::Kv(KvOp::Incr { - collection: QualifiedCollection::new(ctx.database_id(), collection), - key: key.as_bytes().to_vec(), - delta, - ttl_ms, - surrogate, - rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), - })) -} - -pub(crate) fn build_incr_float( - ctx: &DispatchCtx<'_>, - collection: &str, - fields: &TextFields, -) -> crate::Result { - let key = fields - .key - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'key'".to_string(), - })?; - let delta = fields.incr_float_delta.unwrap_or(1.0); - let surrogate = assign_kv_surrogate(ctx, collection, key.as_bytes())?; - - Ok(PhysicalPlan::Kv(KvOp::IncrFloat { - collection: QualifiedCollection::new(ctx.database_id(), collection), - key: key.as_bytes().to_vec(), - delta, - surrogate, - rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), - })) + ctx.state.surrogate_assigner.assign( + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + key, + ) } pub(crate) fn build_cas( @@ -378,123 +290,3 @@ pub(crate) fn build_getset( rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), })) } - -pub(crate) fn build_register_sorted_index( - ctx: &DispatchCtx<'_>, - collection: &str, - fields: &TextFields, -) -> crate::Result { - let index_name = fields - .index_name - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'index_name'".into(), - })?; - let sort_columns = fields.sort_columns.clone().unwrap_or_default(); - let key_column = fields.key_column.clone().unwrap_or_default(); - let window_type = fields.window_type.clone().unwrap_or_else(|| "none".into()); - let window_timestamp_column = fields.window_timestamp_column.clone().unwrap_or_default(); - let window_start_ms = fields.window_start_ms.unwrap_or(0); - let window_end_ms = fields.window_end_ms.unwrap_or(0); - - Ok(PhysicalPlan::Kv(KvOp::RegisterSortedIndex { - collection: QualifiedCollection::new(ctx.database_id(), collection), - index_name: index_name.to_string(), - sort_columns, - key_column, - window_type, - window_timestamp_column, - window_start_ms, - window_end_ms, - })) -} - -pub(crate) fn build_drop_sorted_index(fields: &TextFields) -> crate::Result { - let index_name = fields - .index_name - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'index_name'".into(), - })?; - Ok(PhysicalPlan::Kv(KvOp::DropSortedIndex { - index_name: index_name.to_string(), - })) -} - -pub(crate) fn build_sorted_index_rank(fields: &TextFields) -> crate::Result { - let index_name = fields - .index_name - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'index_name'".into(), - })?; - let key = fields - .key - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'key'".into(), - })?; - Ok(PhysicalPlan::Kv(KvOp::SortedIndexRank { - index_name: index_name.to_string(), - primary_key: key.as_bytes().to_vec(), - })) -} - -pub(crate) fn build_sorted_index_top_k(fields: &TextFields) -> crate::Result { - let index_name = fields - .index_name - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'index_name'".into(), - })?; - let k = fields.top_k_count.unwrap_or(10); - Ok(PhysicalPlan::Kv(KvOp::SortedIndexTopK { - index_name: index_name.to_string(), - k, - })) -} - -pub(crate) fn build_sorted_index_range(fields: &TextFields) -> crate::Result { - let index_name = fields - .index_name - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'index_name'".into(), - })?; - Ok(PhysicalPlan::Kv(KvOp::SortedIndexRange { - index_name: index_name.to_string(), - score_min: fields.score_min.clone(), - score_max: fields.score_max.clone(), - })) -} - -pub(crate) fn build_sorted_index_count(fields: &TextFields) -> crate::Result { - let index_name = fields - .index_name - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'index_name'".into(), - })?; - Ok(PhysicalPlan::Kv(KvOp::SortedIndexCount { - index_name: index_name.to_string(), - })) -} - -pub(crate) fn build_sorted_index_score(fields: &TextFields) -> crate::Result { - let index_name = fields - .index_name - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'index_name'".into(), - })?; - let key = fields - .key - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'key'".into(), - })?; - Ok(PhysicalPlan::Kv(KvOp::SortedIndexScore { - index_name: index_name.to_string(), - primary_key: key.as_bytes().to_vec(), - })) -} diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/kv_counter.rs b/nodedb/src/control/server/native/dispatch/plan_builder/kv_counter.rs new file mode 100644 index 000000000..017b9cc1a --- /dev/null +++ b/nodedb/src/control/server/native/dispatch/plan_builder/kv_counter.rs @@ -0,0 +1,86 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Native `KvIncr` / `KvIncrFloat` plan builders. + +use nodedb_physical::physical_plan::KvOp; +use nodedb_sql::planner::dml_helpers::KvCounterKind; +use nodedb_types::QualifiedCollection; +use nodedb_types::protocol::TextFields; + +use super::kv::assign_kv_surrogate; +use crate::bridge::envelope::PhysicalPlan; +use crate::control::planner::sql_plan_convert::kv_counter_shape::kv_counter_shape; +use crate::control::server::native::dispatch::DispatchCtx; + +fn required_key(fields: &TextFields) -> crate::Result<&str> { + fields + .key + .as_deref() + .ok_or_else(|| crate::Error::BadRequest { + detail: "missing 'key'".to_string(), + }) +} + +pub(crate) fn build_incr( + ctx: &DispatchCtx<'_>, + collection: &str, + fields: &TextFields, +) -> crate::Result { + let key = required_key(fields)?; + let delta = fields.incr_delta.unwrap_or(1); + let ttl_ms = fields.ttl_ms.unwrap_or(0); + let surrogate = assign_kv_surrogate(ctx, collection, key.as_bytes())?; + // An absent key takes the collection's shape, as a SQL `KV_INCR` does. + let shape = kv_counter_shape( + ctx.state, + ctx.tenant_id(), + ctx.database_id(), + collection, + key, + KvCounterKind::Integer, + )?; + + Ok(PhysicalPlan::Kv(KvOp::Incr { + collection: QualifiedCollection::new(ctx.database_id(), collection), + key: key.as_bytes().to_vec(), + delta, + ttl_ms, + surrogate, + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape, + })) +} + +pub(crate) fn build_incr_float( + ctx: &DispatchCtx<'_>, + collection: &str, + fields: &TextFields, +) -> crate::Result { + let key = required_key(fields)?; + // The delta stays the client's decimal text, so the engine adds every + // digit the client sent. + let delta = fields.incr_float_delta.as_deref().unwrap_or("1"); + if !nodedb_physical::kv_atomic::float_text::is_decimal_number(delta) { + return Err(crate::Error::BadRequest { + detail: format!("KvIncrFloat: delta must be a decimal number, got '{delta}'"), + }); + } + let surrogate = assign_kv_surrogate(ctx, collection, key.as_bytes())?; + let shape = kv_counter_shape( + ctx.state, + ctx.tenant_id(), + ctx.database_id(), + collection, + key, + KvCounterKind::Float, + )?; + + Ok(PhysicalPlan::Kv(KvOp::IncrFloat { + collection: QualifiedCollection::new(ctx.database_id(), collection), + key: key.as_bytes().to_vec(), + delta: delta.to_string(), + surrogate, + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape, + })) +} diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/mod.rs b/nodedb/src/control/server/native/dispatch/plan_builder/mod.rs index e1d342a31..0d5555db5 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/mod.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/mod.rs @@ -9,9 +9,11 @@ pub(crate) mod columnar; pub(crate) mod crdt; mod dispatch; pub(crate) mod document; +pub(crate) mod document_bulk; pub(crate) mod graph; mod helpers; pub(crate) mod kv; +pub(crate) mod kv_counter; pub(crate) mod query; pub(crate) mod spatial; pub(crate) mod text; diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/vector.rs b/nodedb/src/control/server/native/dispatch/plan_builder/vector.rs index cea66a91c..37a682aac 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/vector.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/vector.rs @@ -75,9 +75,8 @@ pub(crate) fn build_batch_insert( let mut surrogates = Vec::with_capacity(vectors.len()); for _ in &vectors { surrogates.push(assigner.assign_anonymous( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, )?); } @@ -109,15 +108,17 @@ pub(crate) fn build_insert( let (surrogate, pk_bytes) = match fields.document_id.as_deref() { Some(pk) if !pk.is_empty() => ( assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, pk.as_bytes(), )?, Some(pk.as_bytes().to_vec()), ), _ => ( - assigner.assign_anonymous(ctx.database_id(), ctx.tenant_id(), collection)?, + assigner.assign_anonymous( + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + )?, None, ), }; @@ -176,33 +177,3 @@ pub(crate) fn build_delete( vector_id, })) } - -pub(crate) fn build_set_params( - ctx: &DispatchCtx<'_>, - fields: &TextFields, - collection: &str, -) -> crate::Result { - let m = fields.m.unwrap_or(16) as usize; - let ef_construction = fields.ef_construction.unwrap_or(200) as usize; - let metric = fields - .metric - .clone() - .unwrap_or_else(|| "cosine".to_string()); - let index_type = fields - .index_type - .clone() - .unwrap_or_else(|| "hnsw".to_string()); - - Ok(PhysicalPlan::Vector(VectorOp::SetParams { - collection: QualifiedCollection::new(ctx.database_id(), collection), - field_name: fields.field_name.clone().unwrap_or_default(), - dim: fields.vector_dim.unwrap_or(0) as usize, - m, - ef_construction, - metric, - index_type, - pq_m: 0, - ivf_cells: 0, - ivf_nprobe: 0, - })) -} diff --git a/nodedb/src/control/server/native/dispatch/raw_dispatch.rs b/nodedb/src/control/server/native/dispatch/raw_dispatch.rs index 8a32fbc2d..665ee98f9 100644 --- a/nodedb/src/control/server/native/dispatch/raw_dispatch.rs +++ b/nodedb/src/control/server/native/dispatch/raw_dispatch.rs @@ -5,8 +5,8 @@ use crate::bridge::envelope::{Payload, PhysicalPlan, Response, Status}; use std::sync::Arc; -use crate::control::gateway::GatewayErrorMap; use crate::control::gateway::core::QueryContext as GatewayQueryContext; +use crate::control::gateway::router::is_task_vshard_scoped; use crate::control::server::shared::clone_write::CloneCheckedOutcome; use crate::types::{Lsn, RequestId, TenantId, TraceId, TxnId, VShardId}; @@ -57,7 +57,29 @@ pub(super) async fn dispatch_authorized_single_task( CloneCheckedOutcome::Handled(resp) => return Ok(resp), CloneCheckedOutcome::Proceed(checked) => checked, }; - match ctx.state.gateway.get() { + // A write whose RLS write policy is decided per row cannot be proposed + // bare: a follower has no writing identity to decide it against. It + // resolves to a concrete row set here, while the identity is live, the + // way the planned native and pgwire writes resolve. + if checked.txn_id().is_none() + && ctx.state.async_raft_proposer().is_some() + && let Some(resolver) = crate::control::write_resolve::resolver_for_plan(checked.plan()) + { + return crate::control::write_resolve::run_authorized_write_resolve( + ctx.state, + checked.into_authorized(), + resolver, + ) + .await; + } + // A staged write and the other transaction meta-ops run on the core of + // the task's own vShard. The gateway would route them to vShard 0. + let gateway = ctx + .state + .gateway + .get() + .filter(|_| !is_task_vshard_scoped(checked.plan())); + match gateway { Some(gateway) => { let query = GatewayQueryContext { tenant_id, @@ -65,14 +87,9 @@ pub(super) async fn dispatch_authorized_single_task( database_id: ctx.database_id(), txn_id, }; - gateway - .execute(&query, checked) - .await - .map(gateway_payloads_to_response) - .map_err(|error| { - let (_, detail) = GatewayErrorMap::to_native(&error); - crate::Error::Dispatch { detail } - }) + // The typed error passes through unchanged. The native frame + // renders its SQLSTATE and numeric code from it. + gateway.execute_response(&query, checked).await } None => dispatch_without_gateway(ctx, checked).await, } @@ -107,7 +124,8 @@ async fn dispatch_external_crdt_apply( .map_err(crate::Error::from)?; let task = nodedb_physical::physical_task::PhysicalTask { tenant_id, - vshard_id: VShardId::from_collection_in_database(ctx.database_id(), collection.as_str()), + vshard_id: nodedb_types::CollectionKey::from_qualified(ctx.database_id(), &collection)? + .vshard(), database_id: ctx.database_id(), plan, post_set_op: nodedb_physical::physical_task::PostSetOp::None, @@ -180,8 +198,7 @@ pub(super) async fn dispatch_without_gateway( if crate::control::crdt_admission::changes_crdt_frontier(op) ); let write = || async move { - dispatch_utils::dispatch_authorized_autocommit_write(ctx.state, checked, TraceId::ZERO) - .await + dispatch_utils::dispatch_authorized_durable_write(ctx.state, checked, TraceId::ZERO).await }; if frontier_mutation { ctx.state @@ -192,23 +209,3 @@ pub(super) async fn dispatch_without_gateway( write().await } } - -fn gateway_payloads_to_response(payloads: Vec>) -> Response { - let payload = payloads - .into_iter() - .next() - .map(Payload::from_vec) - .unwrap_or_else(Payload::empty); - Response { - request_id: RequestId::new(0), - status: Status::Ok, - attempt: 0, - partial: false, - payload, - watermark_lsn: Lsn::ZERO, - error_code: None, - read_set_valid: None, - read_version_lsn: Lsn::ZERO, - write_set: Vec::new(), - } -} diff --git a/nodedb/src/control/server/native/dispatch/single_task.rs b/nodedb/src/control/server/native/dispatch/single_task.rs index b417f29d3..d141355ee 100644 --- a/nodedb/src/control/server/native/dispatch/single_task.rs +++ b/nodedb/src/control/server/native/dispatch/single_task.rs @@ -32,9 +32,9 @@ use super::{ /// Routes through the same protocol-neutral in-transaction staging gate /// (`route_in_tx_write`) the SQL-planned dispatch loops (`sql_loop.rs`, /// pgwire's `execute_dml_hooks.rs`) already use. Outside a transaction block -/// this is a no-op passthrough (`InTxnRoute::Read` with the task unchanged), -/// so autocommit direct ops (including `KvBatchPut`) dispatch exactly as -/// before. Inside a transaction block, a stageable write (e.g. `KvBatchPut`) +/// the task comes back unchanged (`InTxnRoute::Read`, or `Autocommit` for a +/// write), and the gateway gives a write its durable route. Inside a +/// transaction block, a stageable write (e.g. `KvBatchPut`) /// is applied to the per-transaction overlay at statement time instead of /// hitting durable storage directly. Otherwise a native direct-op write /// inside `BEGIN...COMMIT` would commit immediately and survive `ROLLBACK`, @@ -60,8 +60,8 @@ pub(super) async fn dispatch_single_task( // Only when metering is enabled — the default is disabled, so this is a // no-op on the hot path for every deployment that hasn't turned it on. - // Covers the plain `Read` dispatch below (autocommit writes/reads, and - // in-transaction reads). `Staged` meters itself inside + // Covers the `Read` / `Autocommit` dispatch below (reads, and writes that + // apply now). `Staged` meters itself inside // `staging_gate::stage_write` — the single choke-point every `Staged` // route (this file, `sql_loop.rs`, the expander's per-op staging, and // pgwire's `execute_dml_hooks.rs`) dispatches through, so it is metered @@ -100,7 +100,9 @@ pub(super) async fn dispatch_single_task( ) .await { - Ok(InTxnRoute::Read(routed_task)) => *routed_task, + // A write here reaches the gateway, which proposes it through Raft or + // appends its redo record in the funnel. + Ok(InTxnRoute::Read(routed_task) | InTxnRoute::Autocommit(routed_task)) => *routed_task, // A buffered write applies at COMMIT: no count and no verb yet. Ok(InTxnRoute::Buffered) => return NativeResponse::ok(seq), Ok(InTxnRoute::Staged(outcome)) => { diff --git a/nodedb/src/control/server/native/dispatch/sorted_read_op.rs b/nodedb/src/control/server/native/dispatch/sorted_read_op.rs new file mode 100644 index 000000000..e38d5e755 --- /dev/null +++ b/nodedb/src/control/server/native/dispatch/sorted_read_op.rs @@ -0,0 +1,129 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Native sorted-index read opcodes. +//! +//! `KvSortedIndexRank`, `KvSortedIndexTopK`, `KvSortedIndexRange`, +//! `KvSortedIndexCount` and `KvSortedIndexScore` name only an index. They run +//! through the same gated read as the SQL functions `RANK`, `TOPK`, `RANGE` +//! and `SORTED_COUNT`: +//! +//! - The caller must hold `Read` on the collection the index covers. The +//! index registry names that collection. +//! - The read goes to the core that holds the collection's rows. +//! - Inside an explicit transaction the read sees the transaction's own +//! writes and an index the transaction created. +//! +//! The reply keeps the Data Plane shape each opcode has always answered with. + +use nodedb_physical::physical_plan::SortedIndexRead; +use nodedb_types::protocol::{NativeResponse, OpCode, TextFields}; + +use crate::control::server::shared::ddl::neutral::kv_sorted_index::run_read; +use crate::control::server::shared::session::DmlTxnCtx; + +use super::response::data_plane_response_to_native; +use super::{DispatchCtx, ddl_result_to_native, error_to_native, error_to_native_with_sqlstate}; + +/// Run the sorted-index read opcode `op`. +pub(crate) async fn handle_sorted_read_op( + ctx: &DispatchCtx<'_>, + seq: u64, + op: OpCode, + fields: &TextFields, +) -> NativeResponse { + // Per-operation caps (top_k and the like), as every direct read has. + if let Err(e) = super::limits::check_op_limits(ctx.state, fields) { + return error_to_native_with_sqlstate(seq, "0A000", &e); + } + if let Err(e) = ctx.state.check_tenant_quota(ctx.tenant_id()) { + return error_to_native(seq, &e); + } + let (index_name, read) = match sorted_read(op, fields) { + Ok(parsed) => parsed, + Err(e) => return error_to_native_with_sqlstate(seq, "42601", &e), + }; + let txn_ctx = DmlTxnCtx { + sessions: ctx.sessions, + session_id: ctx.peer_addr.into(), + }; + match run_read( + ctx.state, + ctx.identity, + ctx.database_id(), + &txn_ctx, + &index_name, + read, + ) + .await + { + Ok((plan, response)) => data_plane_response_to_native(ctx, seq, &plan, &response), + Err(error) => ddl_result_to_native(seq, Err(error)), + } +} + +/// The index an opcode names and the read it asks for. +fn sorted_read(op: OpCode, fields: &TextFields) -> crate::Result<(String, SortedIndexRead)> { + let index_name = required(fields.index_name.as_deref(), "index_name")?.to_string(); + let key = || required(fields.key.as_deref(), "key").map(|key| key.as_bytes().to_vec()); + let read = match op { + OpCode::KvSortedIndexRank => SortedIndexRead::Rank { + primary_key: key()?, + }, + OpCode::KvSortedIndexTopK => SortedIndexRead::TopK { + k: fields.top_k_count.unwrap_or(10), + }, + OpCode::KvSortedIndexRange => SortedIndexRead::Range { + score_min: fields.score_min.clone(), + score_max: fields.score_max.clone(), + }, + OpCode::KvSortedIndexCount => SortedIndexRead::Count, + OpCode::KvSortedIndexScore => SortedIndexRead::Score { + primary_key: key()?, + }, + other => { + return Err(crate::Error::BadRequest { + detail: format!("opcode {other:?} is not a sorted-index read"), + }); + } + }; + Ok((index_name, read)) +} + +fn required<'a>(value: Option<&'a str>, name: &str) -> crate::Result<&'a str> { + value.ok_or_else(|| crate::Error::BadRequest { + detail: format!("missing '{name}'"), + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn each_read_opcode_names_its_read() { + let fields = TextFields { + index_name: Some("lb".into()), + key: Some("p1".into()), + top_k_count: Some(3), + ..TextFields::default() + }; + assert_eq!( + sorted_read(OpCode::KvSortedIndexTopK, &fields).expect("top k"), + ("lb".to_string(), SortedIndexRead::TopK { k: 3 }) + ); + assert_eq!( + sorted_read(OpCode::KvSortedIndexScore, &fields) + .expect("score") + .1, + SortedIndexRead::Score { + primary_key: b"p1".to_vec() + } + ); + let no_key = TextFields { + index_name: Some("lb".into()), + ..TextFields::default() + }; + assert!(sorted_read(OpCode::KvSortedIndexRank, &no_key).is_err()); + assert!(sorted_read(OpCode::KvSortedIndexCount, &no_key).is_ok()); + } +} diff --git a/nodedb/src/control/server/native/dispatch/sql.rs b/nodedb/src/control/server/native/dispatch/sql.rs index fc2ec9bd7..d1f177a79 100644 --- a/nodedb/src/control/server/native/dispatch/sql.rs +++ b/nodedb/src/control/server/native/dispatch/sql.rs @@ -316,10 +316,12 @@ async fn execute_planned( // sequencer so it commits atomically. Single-shard (and best-effort) keep // the existing per-task gateway/SPSC dispatch loop below unchanged. // Autocommit single-statement dispatch: no session read-set to widen with. - match classify_dispatch( - &tasks, - &crate::control::planner::calvin::read_vshards_of(&sum_target_reads), - ) { + let sum_read_vshards = match crate::control::planner::calvin::read_vshards_of(&sum_target_reads) + { + Ok(vshards) => vshards, + Err(error) => return resp(error_to_native(seq, &error)), + }; + match classify_dispatch(&tasks, &sum_read_vshards) { DispatchClass::SingleShard { .. } => {} DispatchClass::MultiShard { .. } => { // Dispatching to Calvin here applies the statement durably at @@ -385,7 +387,7 @@ async fn execute_planned( let Some(scope) = lease_scope.take() else { return resp(sqlstate_error( seq, - "XX000", + nodedb_types::error::sqlstate::INTERNAL_ERROR, "internal error: query lease scope missing before SQL stream dispatch", )); }; @@ -406,7 +408,7 @@ async fn execute_planned( let Some(lease_scope) = lease_scope.take() else { return resp(sqlstate_error( seq, - "XX000", + nodedb_types::error::sqlstate::INTERNAL_ERROR, "internal error: query lease scope missing before materialized SQL dispatch", )); }; @@ -431,7 +433,7 @@ async fn execute_planned( // expects (a single, fully-resolved SQL string). // // Errors here surface as `42P02` (`undefined_parameter`) so the client -// gets a typed SQLSTATE rather than a generic `XX000` opaque failure. +// gets a typed SQLSTATE rather than an opaque internal error. /// Substitute `$N` placeholders in `sql` with canonical SQL literals. fn inline_params(sql: &str, params: &[Value]) -> String { diff --git a/nodedb/src/control/server/native/dispatch/sql_admin.rs b/nodedb/src/control/server/native/dispatch/sql_admin.rs index f290c8c13..731bbf6b9 100644 --- a/nodedb/src/control/server/native/dispatch/sql_admin.rs +++ b/nodedb/src/control/server/native/dispatch/sql_admin.rs @@ -101,7 +101,13 @@ pub(super) async fn handle_explain(ctx: &DispatchCtx<'_>, seq: u64, sql: &str) - }; } - let perm_cache = ctx.state.permission_cache.read().await; + let perm_cache = + match crate::control::security::auth_fence::permission_view(ctx.state, ctx.tenant_id()) + .await + { + Ok(view) => view, + Err(e) => return error_to_native(seq, &e), + }; let sec = crate::control::planner::context::PlanSecurityContext { identity: ctx.identity, auth: ctx.auth_context(), diff --git a/nodedb/src/control/server/native/dispatch/sql_gateway.rs b/nodedb/src/control/server/native/dispatch/sql_gateway.rs index bf38963d7..7731841a8 100644 --- a/nodedb/src/control/server/native/dispatch/sql_gateway.rs +++ b/nodedb/src/control/server/native/dispatch/sql_gateway.rs @@ -3,19 +3,19 @@ //! Gateway-based SQL task dispatch for the native protocol. //! //! When `SharedState.gateway` is `Some`, tasks are routed through -//! `Gateway::execute` which handles cluster-aware routing, typed `NotLeader` +//! `Gateway::execute_response` which handles cluster-aware routing, typed `NotLeader` //! retry, and plan caching. The `None` fallback retains the original //! `dispatch_to_data_plane` path for single-node boot before the gateway is //! wired. This is native's SQL-TEXT opcode path — distinct from //! `raw_dispatch.rs`, which serves only native's direct-op opcodes. -use crate::bridge::envelope::{Payload, Response, Status}; +use crate::bridge::envelope::Response; use std::sync::Arc; -use crate::control::gateway::GatewayErrorMap; use crate::control::gateway::core::QueryContext as GatewayQueryContext; +use crate::control::gateway::router::is_task_vshard_scoped; use crate::control::server::shared::clone_write::CloneCheckedOutcome; -use crate::types::{Lsn, RequestId, TraceId}; +use crate::types::TraceId; use nodedb_physical::physical_task::PhysicalTask; use super::DispatchCtx; @@ -48,8 +48,8 @@ pub(super) fn authorize_native_task( /// Dispatch a single `PhysicalTask` through the gateway when available, /// falling back to the local SPSC path. /// -/// Returns a synthetic `Response` shaped identically to the SPSC path so that -/// the calling code in `sql.rs` is unchanged. +/// Both paths return the Data-Plane `Response` shape, with a `NotFound` +/// verdict as an error status. pub(super) async fn dispatch_task_via_gateway( ctx: &DispatchCtx<'_>, task: PhysicalTask, @@ -77,7 +77,14 @@ pub(super) async fn dispatch_task_via_gateway( let database_id = checked.database_id(); let txn_id = checked.txn_id(); - match ctx.state.gateway.get() { + // A staged write and the other transaction meta-ops run on the core of + // the task's own vShard. The gateway would route them to vShard 0. + let gateway = ctx + .state + .gateway + .get() + .filter(|_| !is_task_vshard_scoped(checked.plan())); + match gateway { Some(gw) => { let gw_ctx = GatewayQueryContext { tenant_id, @@ -87,18 +94,13 @@ pub(super) async fn dispatch_task_via_gateway( // dispatch resolves the per-txn staging overlay. txn_id, }; - gw.execute(&gw_ctx, checked) - .await - .map_err(|e| { - let (code, msg) = GatewayErrorMap::to_native(&e); - crate::Error::Internal { - detail: format!("gateway error {code}: {msg}"), - } - }) - .map(payloads_to_response) + // The typed error passes through unchanged. The native frame + // renders its SQLSTATE and numeric code from it. + gw.execute_response(&gw_ctx, checked).await } + // A write takes the durable route, a read the read route. None => { - crate::control::server::dispatch_utils::dispatch_authorized_to_data_plane( + crate::control::server::dispatch_utils::dispatch_authorized_task_by_class( ctx.state, checked, TraceId::generate(), @@ -107,28 +109,3 @@ pub(super) async fn dispatch_task_via_gateway( } } } - -/// Convert gateway `Vec>` payloads into a synthetic `Response`. -/// -/// Mirrors the same conversion used in the RESP gateway_dispatch module: -/// the first payload is used as the response body; an empty `Vec` yields an -/// empty payload with `Status::Ok`. -fn payloads_to_response(payloads: Vec>) -> Response { - let payload = payloads - .into_iter() - .next() - .map(Payload::from_vec) - .unwrap_or_else(Payload::empty); - Response { - request_id: RequestId::new(0), - status: Status::Ok, - attempt: 0, - partial: false, - payload, - watermark_lsn: Lsn::new(0), - error_code: None, - read_set_valid: None, - read_version_lsn: crate::types::Lsn::ZERO, - write_set: Vec::new(), - } -} diff --git a/nodedb/src/control/server/native/dispatch/sql_loop.rs b/nodedb/src/control/server/native/dispatch/sql_loop.rs index 45797f8c4..f9c8b234e 100644 --- a/nodedb/src/control/server/native/dispatch/sql_loop.rs +++ b/nodedb/src/control/server/native/dispatch/sql_loop.rs @@ -120,8 +120,8 @@ pub(super) async fn run_dispatch_loop( // Extracted from the same clone above, before `task` is moved into // the routing call below — metering needs the collection/engine // shape after this task's dispatch succeeds. Only covers the direct - // dispatch below (`InTxnRoute::Read`, i.e. autocommit writes/reads - // and in-transaction reads); `Buffered`/`Staged` tasks `continue` + // dispatch below (`InTxnRoute::Read` / `Autocommit`: reads, and + // writes that apply now); `Buffered`/`Staged` tasks `continue` // before reaching the metering call and are not billed here — a // `Buffered` task performs no dispatch yet (replayed at COMMIT), and // a `Staged` task's dispatch happens inside `route_in_tx_write`'s @@ -145,8 +145,8 @@ pub(super) async fn run_dispatch_loop( // buffered for COMMIT-time replay; stageable writes are applied to // the per-transaction overlay immediately for a real affected count // and statement-time constraint errors. Outside a transaction block, - // `route_in_tx_write` always returns `Read(task)` unchanged, so the - // autocommit path is untouched. + // `route_in_tx_write` returns the task unchanged, as `Read` or as + // `Autocommit` for a write. // In-transaction `MERGE` and `UPDATE ... FROM` are resolved + staged at // STATEMENT time by the expander (read-your-own-writes for later // statements in the same txn); every other task falls through to the @@ -191,7 +191,7 @@ pub(super) async fn run_dispatch_loop( { return resp(sqlstate_error( seq, - "XX000", + nodedb_types::error::sqlstate::INTERNAL_ERROR, "internal error: failed to retain descriptor leases for buffered transaction tasks", )); } @@ -205,7 +205,9 @@ pub(super) async fn run_dispatch_loop( PlanKind::ReturningRows ); let task = match routed { - Ok(InTxnRoute::Read(routed_task)) => *routed_task, + // A write here reaches the gateway, which proposes it through + // Raft or appends its redo record in the funnel. + Ok(InTxnRoute::Read(routed_task) | InTxnRoute::Autocommit(routed_task)) => *routed_task, Ok(InTxnRoute::Buffered) => { if returns_rows { return resp(error_to_native( diff --git a/nodedb/src/control/server/native/dispatch/streaming.rs b/nodedb/src/control/server/native/dispatch/streaming.rs index 725a9b77e..d63bd3b1e 100644 --- a/nodedb/src/control/server/native/dispatch/streaming.rs +++ b/nodedb/src/control/server/native/dispatch/streaming.rs @@ -49,7 +49,7 @@ impl SqlOutcome { SqlOutcome::Response(r) => *r, SqlOutcome::Stream(s) => crate::control::server::native::sqlstate_code::sqlstate_error( s.seq, - "XX000", + nodedb_types::error::sqlstate::INTERNAL_ERROR, "internal error: SQL stream produced on a non-streaming path", ), } diff --git a/nodedb/src/control/server/native/dispatch/transaction.rs b/nodedb/src/control/server/native/dispatch/transaction.rs index 1e21eadb9..3dffd388d 100644 --- a/nodedb/src/control/server/native/dispatch/transaction.rs +++ b/nodedb/src/control/server/native/dispatch/transaction.rs @@ -21,7 +21,6 @@ use crate::control::server::shared::session::{ AbortReason, CommitOutcome, TxnDataPlane, commit, lifecycle, }; use crate::control::state::SharedState; -use crate::types::Lsn; use nodedb_physical::physical_task::PhysicalTask; use super::super::super::dispatch_utils; @@ -32,9 +31,9 @@ use super::DispatchCtx; /// Always dispatches through the direct SPSC write path using the task's /// pre-classified `vshard_id`, mirroring pgwire's `dispatch_task_no_wal`. /// The gateway must NOT be used here: commit-time tasks carry `MetaOp` plans -/// (`ResolveTxn`, `TransactionBatch`) with no named collection, so the +/// (`ResolveTxn`, `ApplyTransactionRedo`) with no named collection, so the /// gateway's router cannot derive a route for them and falls back to -/// vShard 0 — durably applying the commit batch on the wrong core. +/// vShard 0 — durably applying the commit on the wrong core. pub(crate) struct NativeTxnDp<'a> { pub(crate) state: &'a SharedState, } @@ -43,7 +42,6 @@ impl TxnDataPlane for NativeTxnDp<'_> { fn dispatch_no_wal<'a>( &'a self, task: PhysicalTask, - wal_lsn: Option, ) -> Pin> + Send + 'a>> { let state = self.state; Box::pin(async move { @@ -57,15 +55,20 @@ impl TxnDataPlane for NativeTxnDp<'_> { trace_id: TraceId::ZERO, event_source: crate::event::EventSource::User, txn_id: None, - wal_lsn, + wal_lsn: None, // Batch COMMIT record, not per-task WAL append — see // `dispatch_task_no_wal`'s equivalent limitation. resolved_now_ms: None, + minted: None, }, ) .await }) } + + fn event_source(&self) -> crate::event::EventSource { + crate::event::EventSource::User + } } pub(crate) fn handle_begin(ctx: &DispatchCtx<'_>, seq: u64) -> NativeResponse { @@ -120,9 +123,9 @@ pub(crate) async fn handle_rollback(ctx: &DispatchCtx<'_>, seq: u64) -> NativeRe NativeResponse::status_row(seq, "ROLLBACK") } -/// Map a neutral commit abort reason to the native error frame native emitted -/// before extraction (batch/dispatch failures collapse to `40001`, batch -/// rejections carry the Data-Plane SQLSTATE). +/// Map a neutral commit abort reason to the native error frame. A batch +/// rejection carries the Data-Plane SQLSTATE. A dispatch or DDL-propose error +/// keeps the SQLSTATE and code of its typed error, as on pgwire. fn commit_abort_to_native(seq: u64, reason: &AbortReason) -> NativeResponse { // The numeric NodeDB code rides alongside the SQLSTATE wherever the abort // was classified: a UNIQUE violation that only surfaces at COMMIT is the @@ -167,16 +170,78 @@ fn commit_abort_to_native(seq: u64, reason: &AbortReason) -> NativeResponse { format!("could not serialize access due to concurrent schema change: {detail}"), nodedb_types::error::ErrorCode::WRITE_CONFLICT.0, ), - AbortReason::Dispatch(e) => ( - "40001", - format!("transaction commit failed: {e}"), - nodedb_types::error::ErrorCode::WRITE_CONFLICT.0, - ), - AbortReason::DdlPropose(e) => ( - "XX000", - format!("{e}"), - nodedb_types::error::ErrorCode::INTERNAL.0, - ), + AbortReason::Dispatch(e) => { + let fields = super::native_error_fields(e); + ( + fields.sqlstate, + format!("transaction commit failed: {}", fields.message), + fields.code.0, + ) + } + AbortReason::DdlPropose(e) => { + let fields = super::native_error_fields(e); + (fields.sqlstate, fields.message, fields.code.0) + } }; NativeResponse::error_with_code(seq, code, message, ndb_code) } + +#[cfg(test)] +mod tests { + use nodedb_types::error::ErrorCode as PublicCode; + + use super::*; + + fn frame(reason: &AbortReason) -> (String, String, u16) { + let payload = commit_abort_to_native(1, reason) + .error + .expect("an aborted commit answers an error frame"); + (payload.code, payload.message, payload.ndb_code) + } + + /// A buffered DDL refused at COMMIT keeps the class of its typed error, + /// the same SQLSTATE pgwire renders for it. + #[test] + fn a_ddl_propose_abort_keeps_its_class() { + let in_use = crate::Error::RoleInUse { + role: "analyst".into(), + dependents: crate::control::security::role_assignment::RoleDependents::Users(vec![ + "bob".into(), + ]), + }; + let (code, _, ndb_code) = frame(&AbortReason::DdlPropose(in_use)); + assert_eq!(code, "2BP01"); + assert_eq!(ndb_code, PublicCode::DEPENDENT_OBJECTS_EXIST.0); + } + + /// A DDL step that failed at COMMIT answers the frame its own DDL + /// error carries: the exact SQLSTATE, code and message. + #[test] + fn a_ddl_error_abort_keeps_its_sqlstate_and_code() { + let ddl = crate::control::server::shared::ddl::DdlError::new( + "42710", + "index 'by_email' already exists", + ); + let (code, message, ndb_code) = frame(&AbortReason::DdlPropose(crate::Error::from(ddl))); + assert_eq!(code, "42710"); + assert_eq!(ndb_code, PublicCode::ALREADY_EXISTS.0); + assert_eq!(message, "index 'by_email' already exists"); + } + + /// A commit dispatch error keeps its class instead of reading as a + /// serialization failure. + #[test] + fn a_dispatch_abort_keeps_its_class() { + let missing = crate::Error::CollectionNotFound { + tenant_id: crate::types::TenantId::new(1), + collection: "orders".into(), + }; + let (code, message, ndb_code) = frame(&AbortReason::Dispatch(missing)); + assert_eq!(code, "42P01"); + assert_eq!(ndb_code, PublicCode::COLLECTION_NOT_FOUND.0); + assert!( + message.starts_with("transaction commit failed: "), + "{message}" + ); + } +} diff --git a/nodedb/src/control/server/native/session/auth.rs b/nodedb/src/control/server/native/session/auth.rs index a2ffdefde..1557077e9 100644 --- a/nodedb/src/control/server/native/session/auth.rs +++ b/nodedb/src/control/server/native/session/auth.rs @@ -12,7 +12,9 @@ use crate::control::server::shared::authorization::authorize_database; use super::NativeSession; use super::dispatch; -use crate::control::server::native::dispatch::error_to_native_with_sqlstate; +use crate::control::server::native::dispatch::{ + error_to_native_in_context, error_to_native_with_sqlstate, +}; use crate::control::server::native::sqlstate_code::sqlstate_error; impl NativeSession { @@ -88,8 +90,12 @@ impl NativeSession { "selected database does not exist", ); } - Err(_) => { - return sqlstate_error(seq, "XX000", "database catalog lookup failed"); + Err(e) => { + return error_to_native_in_context( + seq, + "database catalog lookup failed", + &e, + ); } }, None => identity @@ -105,8 +111,12 @@ impl NativeSession { Ok(None) => { return sqlstate_error(seq, "3D000", "selected database does not exist"); } - Err(_) => { - return sqlstate_error(seq, "XX000", "database catalog lookup failed"); + Err(e) => { + return error_to_native_in_context( + seq, + "database catalog lookup failed", + &e, + ); } } @@ -150,7 +160,7 @@ impl NativeSession { drop(scoped); return sqlstate_error( seq, - "XX000", + nodedb_types::error::sqlstate::INTERNAL_ERROR, "internal error: global admission permit missing during auth assembly", ); }; diff --git a/nodedb/src/control/server/native/session/request.rs b/nodedb/src/control/server/native/session/request.rs index 8a92723fd..0b228108a 100644 --- a/nodedb/src/control/server/native/session/request.rs +++ b/nodedb/src/control/server/native/session/request.rs @@ -52,13 +52,22 @@ impl NativeSession { // status surfaces must agree that this node cannot serve. // A halted sequencer is the same shape of after-boot degradation as // a wedged applier: the node still serves, but a whole class of - // writes no longer completes. Both surfaces must say so. + // writes no longer completes. Both surfaces must say so. A halted + // Calvin scheduler is the same shape, scoped to one vShard. // A stalled Data Plane core is a third after-boot degradation with // the same consequence: the gate reads Ok while work sent to that // core never completes. One atomic load, so it stays on this path. + // A fail-stopped core refuses its work outright, with the same + // consequence. let native_status = if self.state.metadata_apply_wedge.is_wedged() || self.state.sequencer_halt.is_halted() + || self.state.sequencer_halt.apply_halt().is_halted() || self.state.core_stall.is_stalled() + || self + .state + .system_metrics + .as_ref() + .is_some_and(|metrics| metrics.core_fail_stops.is_stopped()) { crate::control::startup::health::NativeStatus::Failed } else { @@ -316,6 +325,27 @@ impl NativeSession { dispatch::handle_sql(&ctx, seq, &format!("EXPLAIN {sql}"), None).await } + // Sorted-index reads name only an index: gated on its owning + // collection and run in the caller's transaction, like the SQL + // sorted-index functions. + OpCode::KvSortedIndexRank + | OpCode::KvSortedIndexTopK + | OpCode::KvSortedIndexRange + | OpCode::KvSortedIndexCount + | OpCode::KvSortedIndexScore => { + dispatch::handle_sorted_read_op(&ctx, seq, op, fields).await + } + + // Index DDL runs as the SQL statement it names, so it reaches the + // catalog and the transaction's DDL buffer. + OpCode::KvRegisterSortedIndex + | OpCode::KvDropSortedIndex + | OpCode::VectorSetParams + | OpCode::DocumentDropIndex + | OpCode::DocumentRegister + | OpCode::KvRegisterIndex + | OpCode::KvDropIndex => dispatch::handle_index_ddl_op(&ctx, seq, op, fields).await, + // Direct Data Plane operations. OpCode::PointGet | OpCode::PointPut @@ -360,23 +390,11 @@ impl NativeSession { | OpCode::DocumentTruncate | OpCode::DocumentEstimateCount | OpCode::DocumentInsertSelect - | OpCode::DocumentRegister - | OpCode::DocumentDropIndex - | OpCode::KvRegisterIndex - | OpCode::KvDropIndex | OpCode::KvTruncate - | OpCode::VectorSetParams | OpCode::KvIncr | OpCode::KvIncrFloat | OpCode::KvCas | OpCode::KvGetSet - | OpCode::KvRegisterSortedIndex - | OpCode::KvDropSortedIndex - | OpCode::KvSortedIndexRank - | OpCode::KvSortedIndexTopK - | OpCode::KvSortedIndexRange - | OpCode::KvSortedIndexCount - | OpCode::KvSortedIndexScore | OpCode::CrdtListInsert | OpCode::CrdtListDelete | OpCode::CrdtListMove => dispatch::handle_direct_op(&ctx, seq, op, fields).await, diff --git a/nodedb/src/control/server/native/sqlstate_code.rs b/nodedb/src/control/server/native/sqlstate_code.rs index 7c6c9d721..bd5fa99d6 100644 --- a/nodedb/src/control/server/native/sqlstate_code.rs +++ b/nodedb/src/control/server/native/sqlstate_code.rs @@ -1,6 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Numeric NodeDB codes for native error frames authored as a bare SQLSTATE. +//! Native error frames authored as a bare SQLSTATE. //! //! A native error frame carries both a SQLSTATE and the stable numeric NodeDB //! code, and the client rebuilds its typed error from the number: a frame that @@ -10,83 +10,25 @@ //! //! Most frames get their number from [`crate::error_classify::classify`], //! which is the one internal-`Error`-to-public mapping the crate owns. This -//! module exists for the frames that never held an `Error` to classify: a DDL -//! refusal ([`DdlError`](crate::control::server::shared::ddl::DdlError) is -//! authored as a SQLSTATE plus a message, in ~600 places, and has no numeric -//! code to carry), and the session/dispatch guards that reject a request with -//! a literal SQLSTATE and a static message. For those the SQLSTATE *is* the -//! only classification the server ever produced, so reading it back is a -//! lookup rather than a guess. +//! module serves the frames that never held an `Error` to classify: the +//! session and dispatch guards that reject a request with a literal SQLSTATE +//! and a static message. For those the SQLSTATE *is* the only classification +//! the server produced, so the number comes from the one SQLSTATE-to-code +//! table, [`code_for_sqlstate`], which DDL refusals read too. A bare `0A000` +//! therefore carries the same feature-not-supported class on native as on +//! pgwire. //! -//! This is the inverse of the client-side rule in -//! `NodeDbError::from_wire`, and deliberately so. There, every SQLSTATE the -//! server can emit arrives through one funnel, so a reverse mapping would have -//! to resolve `23505` into either a unique violation or a duplicate -//! idempotency key with no way to tell them apart. Here the lookup happens at -//! the site that chose the SQLSTATE, and the table only carries SQLSTATEs -//! whose NodeDB classification is unambiguous *whatever* site emitted them. -//! -//! Everything else maps to `0`, which is exactly the frame today's code ships, -//! so an unmapped SQLSTATE is never worse off than before this table existed. -//! Three groups stay unmapped on purpose: -//! -//! - **Overloaded SQLSTATEs.** `53400` is `QUOTA_OVERCOMMIT`, -//! `TENANT_QUOTA_EXCEEDED`, `DATABASE_QUOTA_EXCEEDED` and `SERVER_OVERLOAD`; -//! `0A000` is `SQL_NOT_ENABLED` and `CANNOT_CLONE_MIRROR` (the default- -//! database drop guard also sends `0A000` but has no numeric code at all). -//! A caller that knows which one it is passes the code explicitly instead -//! of routing through this table. -//! - **SQLSTATEs with no NodeDB variant.** `42P07` (duplicate table), `42704` -//! (undefined object), `25P02` (aborted transaction), `3B001` (no such -//! savepoint). These need new `ErrorCode`/`ErrorDetails` variants to type at -//! all, which is a public-API change tracked separately. -//! - **SQLSTATEs that are deliberately undistinguished.** Every credential -//! failure renders as `28P01` and every ILP auth failure as a single code -//! with one message, precisely so a caller cannot tell a wrong password from -//! an unknown user. Typing them would rebuild the oracle that collapsing -//! removed. -//! - **SQLSTATEs that mean different things to different emitters.** `57014` -//! is `query_canceled`, which this server sends both for a deadline and for -//! a cancellation that is not one — `COPY restore aborted` renders as `57014` -//! with no deadline anywhere near it. Mapping it to `DEADLINE_EXCEEDED` would -//! be the one entry here that fails the rule above, and it fails it in the -//! expensive direction: `DeadlineExceeded` is retriable, so a cancelled -//! operation would come back classified as worth retrying. A site that -//! cancels for a deadline holds the error and passes the code explicitly, as -//! the Calvin abort paths already do. +//! A SQLSTATE that more than one code shares (`0A000`, `55006`, `57014`, +//! `28000`, `XX000`, `02000` in their special meanings) has a typed constant +//! a `&str` parameter rejects, so a site that means one of those special +//! codes builds its frame from the code, not from this table. -use nodedb_types::error::{ErrorCode, sqlstate}; use nodedb_types::protocol::NativeResponse; -/// The numeric NodeDB code a bare `sqlstate` classifies to, or `0` when it -/// carries no unambiguous classification. -pub(crate) fn ndb_code_for_sqlstate(sqlstate_str: &str) -> u16 { - let code = match sqlstate_str { - sqlstate::UNDEFINED_TABLE => ErrorCode::COLLECTION_NOT_FOUND, - sqlstate::INVALID_CATALOG_NAME => ErrorCode::DATABASE_NOT_FOUND, - sqlstate::INSUFFICIENT_PRIVILEGE => ErrorCode::AUTHORIZATION_DENIED, - sqlstate::UNDEFINED_FUNCTION => ErrorCode::UNDEFINED_FUNCTION, - sqlstate::UNDEFINED_COLUMN => ErrorCode::UNDEFINED_COLUMN, - sqlstate::AMBIGUOUS_COLUMN => ErrorCode::AMBIGUOUS_COLUMN, - // Both a malformed request and a plan that cannot be built render as - // `42601`, so this cannot say which. It does not have to: the two - // differ in which side wrote the bad statement, not in how a client - // must react, and both `BadRequest` and `PlanError` are client errors - // that no caller should retry. - sqlstate::SYNTAX_ERROR => ErrorCode::BAD_REQUEST, - // A cross-shard OCC abort and a retryable refusal both mean "nothing - // applied, retry the whole thing" — the same contract `WriteConflict` - // states, and the classification a retry loop reads. - sqlstate::SERIALIZATION_FAILURE => ErrorCode::WRITE_CONFLICT, - sqlstate::TOO_MANY_CONNECTIONS => ErrorCode::RATE_EXCEEDED, - sqlstate::INTERNAL_ERROR => ErrorCode::INTERNAL, - _ => return 0, - }; - code.0 -} +use crate::control::server::shared::ddl::result::code_for_sqlstate; /// Build a native error frame from a bare SQLSTATE, classifying it through -/// [`ndb_code_for_sqlstate`]. +/// [`code_for_sqlstate`]. /// /// Use this wherever a site rejects a request with a literal SQLSTATE and no /// `Error` value. A site that holds an `Error` must use @@ -98,30 +40,30 @@ pub(crate) fn sqlstate_error( message: impl Into, ) -> NativeResponse { let sqlstate_str = sqlstate_str.into(); - let ndb_code = ndb_code_for_sqlstate(&sqlstate_str); + let ndb_code = code_for_sqlstate(&sqlstate_str).0; NativeResponse::error_with_code(seq, sqlstate_str, message, ndb_code) } #[cfg(test)] mod tests { + use nodedb_types::error::ErrorCode; + use super::*; + fn frame_code(sqlstate: &str) -> u16 { + sqlstate_error(1, sqlstate, "refused") + .error + .expect("error frames carry a payload") + .ndb_code + } + #[test] fn classified_sqlstates_carry_their_code() { - assert_eq!( - ndb_code_for_sqlstate("42P01"), - ErrorCode::COLLECTION_NOT_FOUND.0 - ); - assert_eq!( - ndb_code_for_sqlstate("42501"), - ErrorCode::AUTHORIZATION_DENIED.0 - ); - assert_eq!(ndb_code_for_sqlstate("42601"), ErrorCode::BAD_REQUEST.0); - assert_eq!( - ndb_code_for_sqlstate("3D000"), - ErrorCode::DATABASE_NOT_FOUND.0 - ); - assert_eq!(ndb_code_for_sqlstate("XX000"), ErrorCode::INTERNAL.0); + assert_eq!(frame_code("42P01"), ErrorCode::COLLECTION_NOT_FOUND.0); + assert_eq!(frame_code("42501"), ErrorCode::AUTHORIZATION_DENIED.0); + assert_eq!(frame_code("42601"), ErrorCode::BAD_REQUEST.0); + assert_eq!(frame_code("3D000"), ErrorCode::DATABASE_NOT_FOUND.0); + assert_eq!(frame_code("XX000"), ErrorCode::INTERNAL.0); } /// A retry loop reads the numeric code, so the SQLSTATE the server sends @@ -137,38 +79,44 @@ mod tests { ); } - /// An overloaded or unmapped SQLSTATE must fall through to `0` rather than - /// pick a side: `0` is what the frame ships today, so an unknown SQLSTATE - /// is no worse off, while a wrong guess would misreport retriability. + /// A bare `0A000` guard ("opcode not supported") carries the same class a + /// DDL `0A000` refusal does, not the internal class. + #[test] + fn feature_not_supported_matches_the_ddl_class() { + assert_eq!(frame_code("0A000"), ErrorCode::SQL_NOT_ENABLED.0); + assert_eq!( + frame_code("0A000"), + crate::control::server::shared::ddl::DdlError::new("0A000", "x") + .code + .0 + ); + } + + /// Credential failures stay undistinguished: every one gets the same + /// code, so a caller cannot tell a wrong password from an unknown user. + #[test] + fn credential_failures_share_one_code() { + assert_eq!(frame_code("28P01"), frame_code("28000")); + assert!( + !nodedb_types::NodeDbError::from_wire(ErrorCode(frame_code("28P01")), "x") + .is_retriable() + ); + } + + /// `57014` is sent both for a deadline and for a cancellation that is not + /// one, so it must not classify as the retriable deadline class. #[test] - fn ambiguous_and_unknown_sqlstates_stay_unclassified() { - // Overloaded across several NodeDB variants. - assert_eq!(ndb_code_for_sqlstate("53400"), 0); - assert_eq!(ndb_code_for_sqlstate("0A000"), 0); - // No NodeDB variant exists to map onto. - assert_eq!(ndb_code_for_sqlstate("42P07"), 0); - assert_eq!(ndb_code_for_sqlstate("42704"), 0); - // Deliberately undistinguished so credential failures stay opaque. - assert_eq!(ndb_code_for_sqlstate("28P01"), 0); - assert_eq!(ndb_code_for_sqlstate("28000"), 0); - // Sent both for a deadline and for a cancellation that is not one, so - // it cannot be typed here. `DeadlineExceeded` is retriable, and a - // cancelled operation classified as retriable is one this table told a - // client to run again. - assert_eq!(ndb_code_for_sqlstate("57014"), 0); - // Not a SQLSTATE this server emits. - assert_eq!(ndb_code_for_sqlstate("99999"), 0); + fn query_canceled_is_not_retriable() { + assert_ne!(frame_code("57014"), ErrorCode::DEADLINE_EXCEEDED.0); } - /// An unclassified frame must still reach the client exactly as it does - /// today — same SQLSTATE, same message, `ndb_code == 0` — so adding the - /// table cannot regress a path it does not cover. + /// The frame keeps the SQLSTATE and message the site chose. #[test] - fn unclassified_frame_is_unchanged() { + fn frame_keeps_sqlstate_and_message() { let frame = sqlstate_error(7, "42P07", "table 'repro_t' already exists"); let payload = frame.error.expect("error frames carry a payload"); assert_eq!(payload.code, "42P07"); assert_eq!(payload.message, "table 'repro_t' already exists"); - assert_eq!(payload.ndb_code, 0); + assert_eq!(payload.ndb_code, ErrorCode::ALREADY_EXISTS.0); } } diff --git a/nodedb/src/control/server/payload_merge.rs b/nodedb/src/control/server/payload_merge.rs index d301e90cf..6ce98a092 100644 --- a/nodedb/src/control/server/payload_merge.rs +++ b/nodedb/src/control/server/payload_merge.rs @@ -9,7 +9,7 @@ //! a separate trailing array that the decoder silently ignores (truncating the //! result to the first chunk). //! -//! Used by both `dispatch_utils::collect_bounded_response` (the per-request +//! Used by both `local_dispatch::collect_bounded_response` (the per-request //! bounded collector) and `exchange::gather` (the cross-core/vShard gather). use nodedb_query::msgpack_scan; diff --git a/nodedb/src/control/server/pgwire/connection.rs b/nodedb/src/control/server/pgwire/connection.rs index 3c181b06b..6561d41a9 100644 --- a/nodedb/src/control/server/pgwire/connection.rs +++ b/nodedb/src/control/server/pgwire/connection.rs @@ -27,7 +27,6 @@ use super::connection_identity::PgConnectionContext; use super::factory::NodeDbPgHandlerFactory; const STARTUP_TIMEOUT: Duration = Duration::from_secs(60); -const INTERNAL_ERROR_CODE: &str = "XX000"; const INTERNAL_ERROR_MESSAGE: &str = "internal server error"; /// The observable outcome of a single connection loop. @@ -63,7 +62,7 @@ fn materialize_handlers(build: impl FnOnce() -> T) -> Result { fn fixed_panic_response() -> PgWireBackendMessage { let error = ErrorInfo::new( "FATAL".to_owned(), - INTERNAL_ERROR_CODE.to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), INTERNAL_ERROR_MESSAGE.to_owned(), ); PgWireBackendMessage::ErrorResponse(error.into()) diff --git a/nodedb/src/control/server/pgwire/ddl/database/use_database.rs b/nodedb/src/control/server/pgwire/ddl/database/use_database.rs index 3b335f122..e2964a436 100644 --- a/nodedb/src/control/server/pgwire/ddl/database/use_database.rs +++ b/nodedb/src/control/server/pgwire/ddl/database/use_database.rs @@ -19,7 +19,7 @@ use crate::control::server::shared::session::{ }; use crate::control::state::SharedState; -use super::super::super::types::sqlstate_error; +use super::super::super::types::{error_to_pg_in_context, sqlstate_error}; /// Handle `USE DATABASE `. /// @@ -38,7 +38,7 @@ pub async fn handle_use_database( // Verify the named database exists. let db_id = catalog .get_database_id_by_name(name) - .map_err(|e| sqlstate_error("XX000", &format!("catalog lookup failed: {e}")))? + .map_err(|e| error_to_pg_in_context("catalog lookup failed", &e))? .ok_or_else(|| sqlstate_error("3D000", &format!("database '{name}' does not exist")))?; // Enforce `accessible_databases`: reject the switch if the identity does diff --git a/nodedb/src/control/server/pgwire/ddl_encode.rs b/nodedb/src/control/server/pgwire/ddl_encode.rs index ea11f4194..8b717da47 100644 --- a/nodedb/src/control/server/pgwire/ddl_encode.rs +++ b/nodedb/src/control/server/pgwire/ddl_encode.rs @@ -43,9 +43,23 @@ pub fn ddl_results_to_pgwire( sqlstate, code, message, + cause, + .. }) => { let mut info = ErrorInfo::new("ERROR".to_owned(), sqlstate, message); info.routine = Some(code.to_string()); + // The typed cause travels in `detail`: its SQLSTATE, its numeric + // code, and its message. + info.detail = cause.map(|cause| { + format!( + "caused by {} ({}): {}", + crate::control::server::pgwire::types::error_map::numeric_code_to_sqlstate( + cause.code() + ), + cause.code(), + cause.message() + ) + }); return Err(PgWireError::UserError(Box::new(info))); } }; @@ -175,6 +189,26 @@ mod tests { use super::*; + /// A phase failure keeps its SQLSTATE, and the typed cause travels in + /// `detail` with its own SQLSTATE and code. + #[test] + fn ddl_phase_failure_names_its_cause_in_detail() { + let phase = nodedb_types::NodeDbError::move_tenant_snapshot_failed("7", "dispatch") + .with_cause(nodedb_types::NodeDbError::division_by_zero()); + let result: Result, DdlError> = + Err(DdlError::move_tenant_snapshot_failed(phase.message()).with_cause_of(&phase)); + + let err = ddl_results_to_pgwire(result).expect_err("must map to a pgwire error"); + let PgWireError::UserError(info) = err else { + panic!("expected a UserError carrying ErrorInfo"); + }; + let info = *info; + assert_eq!(info.code, "XX000"); + let detail = info.detail.expect("the cause travels in detail"); + assert!(detail.contains("22012"), "{detail}"); + assert!(detail.contains("NDB-1204"), "{detail}"); + } + /// Round-trips through the actual PostgreSQL wire bytes `ErrorResponse` /// encodes and a client's `pgwire` codec decodes — proving the code /// reaches the wire, not just that the server set it. diff --git a/nodedb/src/control/server/pgwire/factory/startup.rs b/nodedb/src/control/server/pgwire/factory/startup.rs index f3e84e559..d7279c10e 100644 --- a/nodedb/src/control/server/pgwire/factory/startup.rs +++ b/nodedb/src/control/server/pgwire/factory/startup.rs @@ -22,7 +22,7 @@ use crate::control::server::session_auth::identity::stored_user_identity; use crate::control::state::SharedState; use super::super::handler::NodeDbPgHandler; -use super::super::types::sqlstate_error; +use super::super::types::{error_to_pg_in_context, sqlstate_error}; use super::provider::NodeDbParameterProvider; /// Enum dispatch for startup handler — avoids dyn trait object issues. @@ -68,7 +68,7 @@ fn bind_startup_database( .credentials .catalog() .get_database_id_by_name(&db_name) - .map_err(|e| sqlstate_error("XX000", &format!("catalog lookup failed: {e}")))? + .map_err(|e| error_to_pg_in_context("catalog lookup failed", &e))? .ok_or_else(|| sqlstate_error("3D000", &format!("database '{db_name}' does not exist")))?; handler.sessions.set_current_database(session_id, db_id); @@ -109,7 +109,7 @@ fn admit_connection( .set_admission_permit(handler.session_id, permit) { return Err(sqlstate_error( - "XX000", + nodedb_types::error::sqlstate::INTERNAL_ERROR, "internal error: connection session is not registered", )); } diff --git a/nodedb/src/control/server/pgwire/handler/copy_handler.rs b/nodedb/src/control/server/pgwire/handler/copy_handler.rs index 94a452b7e..6ef694a62 100644 --- a/nodedb/src/control/server/pgwire/handler/copy_handler.rs +++ b/nodedb/src/control/server/pgwire/handler/copy_handler.rs @@ -70,7 +70,7 @@ impl NodeDbPgHandler { .ok_or_else(|| { PgWireError::UserError(Box::new(ErrorInfo::new( "FATAL".to_owned(), - "XX000".to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), "connection session metadata is unavailable".to_owned(), ))) })?, @@ -127,10 +127,10 @@ impl NodeDbPgHandler { // byte. The charge below is on the success path and so can // never be where a cap blocks anything. admit_backup_restore_quota(&self.state, request.scope(), tenant_id) - .map_err(internal)?; + .map_err(typed)?; let bytes = backup::backup_tenant(&self.state, tenant_id) .await - .map_err(internal)?; + .map_err(typed)?; // Metered here, on the success path, before the response is // built below — there is no `PhysicalPlan` for a whole-tenant // backup, so the collection dimension is a synthetic @@ -262,7 +262,7 @@ impl CopyHandler for NodeDbCopyHandler { PgWireError: From<>::Error>, { cancel_restore(&self.restore_state, self.connection_id); - sqlstate(ss::QUERY_CANCELED, "COPY restore aborted") + sqlstate(ss::QUERY_CANCELED.0, "COPY restore aborted") } async fn on_copy_done(&self, client: &mut C, _done: CopyDone) -> PgWireResult<()> @@ -290,7 +290,7 @@ impl CopyHandler for NodeDbCopyHandler { self.state.auth_stores(), database_id, ); - admit_backup_restore_quota(&self.state, &scope, pending.tenant_id).map_err(internal)?; + admit_backup_restore_quota(&self.state, &scope, pending.tenant_id).map_err(typed)?; let stats = backup::restore_tenant( &self.state, @@ -300,7 +300,7 @@ impl CopyHandler for NodeDbCopyHandler { pending.force, ) .await - .map_err(internal)?; + .map_err(typed)?; // pgwire does not auto-send CommandComplete after `on_copy_done` // returns Ok — the trait contract leaves message construction to // the handler. Send a `RESTORE TENANT N ` tag so the @@ -343,11 +343,11 @@ fn sqlstate(code: &str, message: &str) -> PgWireError { ))) } -fn internal(e: crate::Error) -> PgWireError { - // Surface error string but never echo deserializer context — the - // restore orchestrator already scrubs envelope errors. We pass - // through everything else (RPC failures, dispatch errors). - sqlstate(ss::INTERNAL_ERROR, &e.to_string()) +/// Render a backup or restore error with its own SQLSTATE: a spent quota, +/// a tenant mismatch or a wrong key keeps its class. The restore +/// orchestrator already scrubs envelope errors before they reach here. +fn typed(e: crate::Error) -> PgWireError { + super::super::types::error_to_pg(&e) } #[cfg(test)] diff --git a/nodedb/src/control/server/pgwire/handler/cursor_query.rs b/nodedb/src/control/server/pgwire/handler/cursor_query.rs index 7d0310141..b190690e7 100644 --- a/nodedb/src/control/server/pgwire/handler/cursor_query.rs +++ b/nodedb/src/control/server/pgwire/handler/cursor_query.rs @@ -3,7 +3,7 @@ //! `DECLARE CURSOR` materialisation: plan a SELECT, dispatch it to the //! Data Plane, and collect JSON-encoded rows for cursor storage. -use pgwire::error::{ErrorInfo, PgWireError, PgWireResult}; +use pgwire::error::{PgWireError, PgWireResult}; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::retry::retry_on_schema_change; @@ -57,7 +57,10 @@ impl NodeDbPgHandler { // rejected cursor declaration consumes no descriptor lease. The scope // remains live while every cursor-materialization task is dispatched. let (tasks, _lease_scope) = retry_on_schema_change(move || async move { - let perm_cache = self.state.permission_cache.read().await; + let perm_cache = + crate::control::security::auth_fence::permission_view(&self.state, tenant_id) + .await + .map_err(StatementSetupError::from)?; let sec = crate::control::planner::context::PlanSecurityContext { identity, auth: auth_ctx, @@ -109,13 +112,7 @@ impl NodeDbPgHandler { TraceId::ZERO, ) .await - .map_err(|e| { - PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - "XX000".to_owned(), - e.to_string(), - ))) - })?; + .map_err(|e| super::super::types::error_to_pg(&e))?; if !resp.payload.is_empty() { let json = diff --git a/nodedb/src/control/server/pgwire/handler/dispatch.rs b/nodedb/src/control/server/pgwire/handler/dispatch.rs deleted file mode 100644 index 22941a6b7..000000000 --- a/nodedb/src/control/server/pgwire/handler/dispatch.rs +++ /dev/null @@ -1,485 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Core dispatch mechanics: single-task dispatch, Raft replication, and local Data Plane submission. - -use std::sync::Arc; - -use crate::bridge::envelope::Response; -use crate::control::security::identity::AuthenticatedIdentity; -use crate::control::server::dispatch_utils::{WalDurability, publish_origin_change_events}; -use crate::control::server::exchange::resolve::{ - DistributedReadCapture, Resolved, resolve_and_materialize, -}; -use crate::types::{Lsn, ReadConsistency, TraceId, VShardId}; -use nodedb_physical::physical_plan::{CrdtOp, PhysicalPlan}; -use nodedb_physical::physical_task::PhysicalTask; - -use super::core::NodeDbPgHandler; -use super::submit::SubmitArgs; - -/// Inputs for [`NodeDbPgHandler::dispatch_replicated_write`]: the entry to -/// propose, the proposer, and the identity + plan its origin CDC publish needs. -struct ReplicatedWrite<'a> { - entry: crate::control::wal_replication::ReplicatedEntry, - proposer: &'a Arc, - authorized: crate::control::server::shared::authorization::AuthorizedTask, -} - -impl NodeDbPgHandler { - fn authorize_for_dispatch( - &self, - identity: &AuthenticatedIdentity, - task: &PhysicalTask, - ) -> crate::Result { - let emitter = - crate::control::security::audit::ArcAuditEmitter(Arc::clone(&self.state.audit)); - crate::control::server::shared::authorization::authorize_task_set( - identity, - std::slice::from_ref(task), - &self.state.permissions, - &self.state.roles, - &emitter, - ) - .map_err(crate::Error::from)? - .into_tasks() - .into_iter() - .next() - .ok_or_else(|| crate::Error::Internal { - detail: "pgwire authorization returned no capability".into(), - }) - } - - /// Dispatch a single physical task and wait for the response. - /// - /// In cluster mode, writes propose to Raft first and execute only after - /// quorum commit; reads bypass Raft. `identity` must be passed for every - /// externally derived task. - pub(super) async fn dispatch_authorized_task( - &self, - task: PhysicalTask, - user_id: Option>, - identity: &AuthenticatedIdentity, - ) -> crate::Result { - let mut shard_watermarks = Vec::new(); - let mut distributed_reads = Vec::new(); - self.dispatch_task_hlc( - task, - user_id, - identity, - &mut shard_watermarks, - &mut distributed_reads, - ) - .await - } - - /// Dispatch a task and return the response, per-shard watermark LSNs a fan - /// gather observed, and per-side read captures a shuffle JOIN produced. - /// Used by the transactional read-recording seam. - pub(super) async fn dispatch_authorized_task_with_watermarks( - &self, - task: PhysicalTask, - user_id: Option>, - identity: &AuthenticatedIdentity, - ) -> crate::Result<(Response, Vec<(VShardId, Lsn)>, Vec)> { - let mut shard_watermarks = Vec::new(); - let mut distributed_reads = Vec::new(); - let resp = self - .dispatch_task_hlc( - task, - user_id, - identity, - &mut shard_watermarks, - &mut distributed_reads, - ) - .await?; - Ok((resp, shard_watermarks, distributed_reads)) - } - - async fn dispatch_task_hlc( - &self, - task: PhysicalTask, - user_id: Option>, - identity: &AuthenticatedIdentity, - shard_watermarks: &mut Vec<(VShardId, Lsn)>, - distributed_reads: &mut Vec, - ) -> crate::Result { - let tenant_id = task.tenant_id; - let result = self - .dispatch_task_inner(task, user_id, identity, shard_watermarks, distributed_reads) - .await; - // Advances per-tenant write-HLC on any successful dispatch; used by RESTORE's - // staleness gate. Backup captures its watermark after fan-out, so it dominates. - if let Ok(ref resp) = result - && resp.status == crate::bridge::envelope::Status::Ok - { - self.state.advance_tenant_write_hlc(tenant_id.as_u64()); - } - result - } - - async fn dispatch_task_inner( - &self, - mut task: PhysicalTask, - user_id: Option>, - identity: &AuthenticatedIdentity, - shard_watermarks: &mut Vec<(VShardId, Lsn)>, - distributed_reads: &mut Vec, - ) -> crate::Result { - // Reject user writes against a database frozen by a clone materializer sweep. - // Reads/DDL pass through. - use crate::control::security::identity::{Permission, required_permission}; - let perm = required_permission(&task.plan); - if matches!(perm, Permission::Write | Permission::Admin) - && self.state.materialize_freeze.is_frozen(task.database_id) - { - return Err(crate::Error::SourceFrozen { - database_id: task.database_id, - }); - } - - // Mirror enforcement: writes reject on non-promoted mirrors; reads gate by - // ReadConsistency. Catalog lookup skipped for db id=0 to stay allocation-free. - let catalog = self.state.credentials.catalog(); - if task.database_id.as_u64() != 0 - && let Ok(Some(descriptor)) = catalog.get_database(task.database_id) - && let Some(origin) = descriptor.mirror_origin.as_ref() - && !matches!(origin.status, nodedb_types::MirrorStatus::Promoted) - { - if matches!(perm, Permission::Write | Permission::Admin) { - return Err(crate::Error::MirrorReadOnly { - database: descriptor.name.clone(), - }); - } - - use crate::control::server::pgwire::ddl::database::{ - MirrorReadOutcome, check_mirror_read_consistency, - }; - // Defaults to Strong: mirrors aren't the source leader, so reads reject - // unless the session opted into BoundedStaleness or Eventual. - let now_ms = std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap_or(std::time::Duration::ZERO) - .as_millis() as u64; - let outcome = check_mirror_read_consistency( - catalog, - task.database_id, - origin, - ReadConsistency::Strong, - now_ms, - ); - if let MirrorReadOutcome::Reject { message, .. } = outcome { - return Err(crate::Error::StaleReadNotLeader { - database: descriptor.name.clone(), - source_cluster: origin.source_cluster.clone(), - detail: message, - }); - } - } - - if matches!( - &task.plan, - crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::InsertSelect { .. } - ) - ) { - let authorized = self.authorize_for_dispatch(identity, &task)?; - return crate::control::insert_select::run_authorized_insert_select( - &self.state, - authorized, - ) - .await; - } - - // Autocommit `MERGE` orchestrates on the Control Plane (`control::merge_orchestrator`). - // In-transaction MERGE buffers for COMMIT replay and never reaches this method. - if matches!( - &task.plan, - crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::Merge { - resolved_inserts: None, - .. - } - ) - ) { - let authorized = self.authorize_for_dispatch(identity, &task)?; - return crate::control::merge_orchestrator::run_authorized_merge( - &self.state, - authorized, - ) - .await; - } - - // Scans the source on its own core and ships raw rows into the plan (source's - // vShard can differ). In-transaction buffers for COMMIT replay instead. - if matches!( - &task.plan, - crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::UpdateFromJoin { - source_rows: None, - .. - } - ) - ) { - let authorized = self.authorize_for_dispatch(identity, &task)?; - return crate::control::update_from_join_orchestrator::run_authorized_update_from_join( - &self.state, - authorized, - ) - .await; - } - - // Can't replicate bare over Raft — a follower has no writing identity to decide - // `$auth.*` against. `write_resolve` resolves it while the identity is live. - if let Some(resolver) = crate::control::write_resolve::resolver_for_plan(&task.plan) - && self.state.async_raft_proposer().is_some() - { - let authorized = self.authorize_for_dispatch(identity, &task)?; - return crate::control::write_resolve::run_authorized_write_resolve( - &self.state, - authorized, - resolver, - ) - .await; - } - - // `DROP ARRAY` reaches every core so each releases its store and segment dir — - // otherwise a follow-up `CREATE ARRAY` carries stale state. - if matches!( - task.plan, - crate::bridge::envelope::PhysicalPlan::Array( - nodedb_physical::physical_plan::ArrayOp::DropArray { .. } - ) - ) { - // Broadcast bypasses the write funnel, so a denied DROP must not - // delete catalog rows or surrogate bindings. - let authorized = self.authorize_for_dispatch(identity, &task)?; - let task = authorized.into_physical_task(); - return crate::control::array_catalog::ddl::run_authorized_drop( - &self.state, - task.tenant_id, - task.database_id, - task.plan, - TraceId::ZERO, - ) - .await; - } - - // Clone-read must run first: resolving derived Exchange plans below - // dispatches straight to the Data Plane, bypassing the clone check. - if let Some(resp) = self - .maybe_intercept_clone_read_early(&task, identity, perm) - .await? - { - return Ok(resp); - } - - // Resolve derived Exchange plans before authorizing the dispatched task. - match resolve_and_materialize( - &self.state, - identity, - task.database_id, - task.tenant_id, - task.plan, - TraceId::ZERO, - task.txn_id, - ) - .await? - { - Resolved::Gathered(resp, wms, caps) => { - *shard_watermarks = wms; - *distributed_reads = caps; - return Ok(resp); - } - Resolved::Plan(resolved_plan) => { - let resolved_plan = *resolved_plan; - task.plan = resolved_plan; - } - Resolved::Stream(stream) => { - return crate::control::server::exchange::gather::stream_to_response(stream).await; - } - } - - reject_unadmitted_crdt_apply(&task.plan)?; - let checked = self - .intercept_and_authorize_for_dispatch(identity, task) - .await?; - let checked = match checked { - crate::control::server::shared::clone_write::CloneCheckedOutcome::Handled(resp) => { - return Ok(resp); - } - crate::control::server::shared::clone_write::CloneCheckedOutcome::Proceed(t) => t, - }; - if let Some(async_proposer) = self.state.async_raft_proposer() - && let Some(entry) = crate::control::wal_replication::to_replicated_entry( - checked.tenant_id(), - checked.database_id(), - checked.vshard_id(), - &crate::control::wal_replication::ReplicableWrite::decide_for_replication( - checked.plan(), - )?, - )? - { - return self - .dispatch_replicated_write(ReplicatedWrite { - entry, - proposer: async_proposer, - authorized: checked.into_authorized(), - }) - .await; - } - self.dispatch_local(checked, user_id).await - } - - /// Dispatch a write through Raft: propose → register waiter → await apply. - /// `ProposeTracker` is race-safe against an entry applying before register. - /// - /// Also the origin CDC publish site; replicas publish nothing (`ChangeFeedOwner::Unowned`). - async fn dispatch_replicated_write( - &self, - args: ReplicatedWrite<'_>, - ) -> crate::Result { - let ReplicatedWrite { - entry, - proposer, - authorized, - } = args; - let task = authorized.into_physical_task(); - let tenant_id = task.tenant_id; - let database_id = task.database_id; - let plan = task.plan; - let request_id = self.next_request_id(); - - // `write_version` is the post-write `coll_write_lsn`, surfaced so the session - // can floor a later read-set at it (read-your-writes for cross-shard OCC). - let (payload, write_version) = - crate::control::wal_replication::propose_replicated_entry(&self.state, proposer, entry) - .await?; - - let response = Response { - request_id, - status: crate::bridge::envelope::Status::Ok, - attempt: 1, - partial: false, - payload: payload.into(), - // Authoritative participant WAL LSN — CDC ordering must use it, not zero. - watermark_lsn: write_version, - error_code: None, - read_set_valid: None, - read_version_lsn: write_version, - write_set: Vec::new(), - }; - - // Propose returned: entry is committed and applied. Publish once, from this plan. - publish_origin_change_events(&self.state, tenant_id, database_id, &plan, &response); - - Ok(response) - } - - /// Dispatch a task directly to the local Data Plane (single-node or reads). - /// - /// WAL append happens inside the write funnel, under the admission guard just - /// before enqueue, so LSN order equals apply order. Reads bypass the WAL entirely. - async fn dispatch_local( - &self, - checked: crate::control::server::shared::clone_write::CloneCheckedTask, - user_id: Option>, - ) -> crate::Result { - self.submit_authorized_to_data_plane( - checked, - user_id, - WalDurability::AppendHere { now_override: None }, - ) - .await - } - - /// Dispatch a task to the Data Plane WITHOUT individual WAL append. - /// - /// Used by COMMIT after the transaction is written as one `RecordType::Transaction` - /// record — per-task WAL would double-write. - pub(super) async fn dispatch_task_no_wal( - &self, - task: PhysicalTask, - user_id: Option>, - wal_lsn: Option, - ) -> crate::Result { - // Without this, a transaction begun before the freeze could COMMIT mid-scan and - // break the as-of contract. - use crate::control::security::identity::{Permission, required_permission}; - let perm = required_permission(&task.plan); - if matches!(perm, Permission::Write | Permission::Admin) - && self.state.materialize_freeze.is_frozen(task.database_id) - { - return Err(crate::Error::SourceFrozen { - database_id: task.database_id, - }); - } - reject_unadmitted_crdt_apply(&task.plan)?; - let txn_id = task.txn_id; - // Writes were durably recorded under one `RecordType::Transaction` record at - // COMMIT; per-task WAL append is skipped. `wal_lsn` stamps that record's LSN. - self.submit_to_data_plane(SubmitArgs { - tenant_id: task.tenant_id, - vshard_id: task.vshard_id, - database_id: task.database_id, - plan: task.plan, - user_id, - txn_id, - // No per-task TTL instant (see `flush_transaction_buffer`), so a TTL-bearing - // KV write falls back to `epoch_system_ms` at apply time. - durability: WalDurability::CallerSupplied { - wal_lsn, - resolved_now_ms: None, - }, - }) - .await - } -} - -fn reject_unadmitted_crdt_apply(plan: &PhysicalPlan) -> crate::Result<()> { - if matches!( - plan, - PhysicalPlan::Crdt( - CrdtOp::Apply { .. } - | CrdtOp::ApplyAuthenticated { .. } - | CrdtOp::ImportSnapshot { .. } - ) - ) { - return Err(crate::Error::CrdtApplyRequiresAdmission); - } - Ok(()) -} - -#[cfg(test)] -mod tests { - use nodedb_types::Surrogate; - - use super::*; - - #[test] - fn generic_pgwire_dispatch_rejects_unadmitted_apply() { - let plan = PhysicalPlan::Crdt(CrdtOp::Apply { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "docs", - ), - document_id: "doc-1".into(), - delta: Vec::new(), - peer_id: 1, - mutation_id: 1, - surrogate: Surrogate::ZERO, - provenance: None, - constraint_version_required: 0, - expected_frontier_digest: None, - }); - assert!(matches!( - reject_unadmitted_crdt_apply(&plan), - Err(crate::Error::CrdtApplyRequiresAdmission) - )); - } - - #[test] - fn dispatch_task_compile_check() { - // Confirms the dispatch module compiles. - let _: () = (); - } -} diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/authorize.rs b/nodedb/src/control/server/pgwire/handler/dispatch/authorize.rs new file mode 100644 index 000000000..594bc47b6 --- /dev/null +++ b/nodedb/src/control/server/pgwire/handler/dispatch/authorize.rs @@ -0,0 +1,87 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Task authorization and the CRDT admission-gate check shared by every +//! dispatch routing path. + +use std::sync::Arc; + +use nodedb_physical::physical_plan::{CrdtOp, PhysicalPlan}; +use nodedb_physical::physical_task::PhysicalTask; + +use crate::control::security::identity::AuthenticatedIdentity; + +use super::super::core::NodeDbPgHandler; + +impl NodeDbPgHandler { + pub(super) fn authorize_for_dispatch( + &self, + identity: &AuthenticatedIdentity, + task: &PhysicalTask, + ) -> crate::Result { + let emitter = + crate::control::security::audit::ArcAuditEmitter(Arc::clone(&self.state.audit)); + crate::control::server::shared::authorization::authorize_task_set( + identity, + std::slice::from_ref(task), + &self.state.permissions, + &self.state.roles, + &emitter, + ) + .map_err(crate::Error::from)? + .into_tasks() + .into_iter() + .next() + .ok_or_else(|| crate::Error::Internal { + detail: "pgwire authorization returned no capability".into(), + }) + } +} + +pub(super) fn reject_unadmitted_crdt_apply(plan: &PhysicalPlan) -> crate::Result<()> { + if matches!( + plan, + PhysicalPlan::Crdt( + CrdtOp::Apply { .. } + | CrdtOp::ApplyAuthenticated { .. } + | CrdtOp::ImportSnapshot { .. } + ) + ) { + return Err(crate::Error::CrdtApplyRequiresAdmission); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use nodedb_types::Surrogate; + + use super::*; + + #[test] + fn generic_pgwire_dispatch_rejects_unadmitted_apply() { + let plan = PhysicalPlan::Crdt(CrdtOp::Apply { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "docs", + ), + document_id: "doc-1".into(), + delta: Vec::new(), + peer_id: 1, + mutation_id: 1, + surrogate: Surrogate::ZERO, + provenance: None, + constraint_version_required: 0, + expected_frontier_digest: None, + }); + assert!(matches!( + reject_unadmitted_crdt_apply(&plan), + Err(crate::Error::CrdtApplyRequiresAdmission) + )); + } + + #[test] + fn dispatch_task_compile_check() { + // Confirms the dispatch module compiles. + let _: () = (); + } +} diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/entry.rs b/nodedb/src/control/server/pgwire/handler/dispatch/entry.rs new file mode 100644 index 000000000..b8ab69fa0 --- /dev/null +++ b/nodedb/src/control/server/pgwire/handler/dispatch/entry.rs @@ -0,0 +1,65 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Public dispatch entry points. +//! +//! A write records its commit HLC on the tenant's observed high-water where it +//! commits: the write funnel for a local append, the Raft proposer and each +//! replica's apply for a replicated entry. + +use std::sync::Arc; + +use crate::bridge::envelope::Response; +use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::exchange::resolve::DistributedReadCapture; +use crate::types::{Lsn, VShardId}; +use nodedb_physical::physical_task::PhysicalTask; + +use super::super::core::NodeDbPgHandler; + +impl NodeDbPgHandler { + /// Dispatch a single physical task and wait for the response. + /// + /// In cluster mode, writes propose to Raft first and execute only after + /// quorum commit; reads bypass Raft. `identity` must be passed for every + /// externally derived task. + pub(in crate::control::server::pgwire::handler) async fn dispatch_authorized_task( + &self, + task: PhysicalTask, + user_id: Option>, + identity: &AuthenticatedIdentity, + ) -> crate::Result { + let mut shard_watermarks = Vec::new(); + let mut distributed_reads = Vec::new(); + self.dispatch_task_inner( + task, + user_id, + identity, + &mut shard_watermarks, + &mut distributed_reads, + ) + .await + } + + /// Dispatch a task and return the response, per-shard watermark LSNs a fan + /// gather observed, and per-side read captures a shuffle JOIN produced. + /// Used by the transactional read-recording seam. + pub(in crate::control::server::pgwire::handler) async fn dispatch_authorized_task_with_watermarks( + &self, + task: PhysicalTask, + user_id: Option>, + identity: &AuthenticatedIdentity, + ) -> crate::Result<(Response, Vec<(VShardId, Lsn)>, Vec)> { + let mut shard_watermarks = Vec::new(); + let mut distributed_reads = Vec::new(); + let resp = self + .dispatch_task_inner( + task, + user_id, + identity, + &mut shard_watermarks, + &mut distributed_reads, + ) + .await?; + Ok((resp, shard_watermarks, distributed_reads)) + } +} diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/local.rs b/nodedb/src/control/server/pgwire/handler/dispatch/local.rs new file mode 100644 index 000000000..553ef4a97 --- /dev/null +++ b/nodedb/src/control/server/pgwire/handler/dispatch/local.rs @@ -0,0 +1,78 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Dispatch a task directly to the local Data Plane, with or without a +//! per-task WAL append. + +use std::sync::Arc; + +use crate::bridge::envelope::Response; +use crate::control::server::dispatch_utils::WalDurability; +use nodedb_physical::physical_task::PhysicalTask; + +use super::super::core::NodeDbPgHandler; +use super::super::submit::SubmitArgs; +use super::authorize::reject_unadmitted_crdt_apply; + +impl NodeDbPgHandler { + /// Dispatch a task directly to the local Data Plane (single-node or reads). + /// + /// WAL append happens inside the write funnel, under the admission guard just + /// before enqueue, so LSN order equals apply order. Reads bypass the WAL entirely. + pub(super) async fn dispatch_local( + &self, + checked: crate::control::server::shared::clone_write::CloneCheckedTask, + user_id: Option>, + ) -> crate::Result { + self.submit_authorized_to_data_plane( + checked, + user_id, + WalDurability::AppendHere { + now_override: None, + apply_key: 0, + commit_hlc: None, + }, + ) + .await + } + + /// Dispatch a task to the Data Plane WITHOUT individual WAL append. + /// + /// Used by COMMIT after the transaction is written as one `RecordType::Transaction` + /// record — per-task WAL would double-write. + pub(in crate::control::server::pgwire::handler) async fn dispatch_task_no_wal( + &self, + task: PhysicalTask, + ) -> crate::Result { + // Without this, a transaction begun before the freeze could COMMIT mid-scan and + // break the as-of contract. + use crate::control::security::identity::{Permission, required_permission}; + let perm = required_permission(&task.plan); + if matches!(perm, Permission::Write | Permission::Admin) + && self.state.materialize_freeze.is_frozen(task.database_id) + { + return Err(crate::Error::SourceFrozen { + database_id: task.database_id, + }); + } + reject_unadmitted_crdt_apply(&task.plan)?; + let txn_id = task.txn_id; + // The caller owns the transaction's durability, so the task carries no + // WAL record of its own. + self.submit_to_data_plane(SubmitArgs { + tenant_id: task.tenant_id, + vshard_id: task.vshard_id, + database_id: task.database_id, + plan: task.plan, + user_id: None, + txn_id, + // No per-task TTL instant (see `flush_transaction_buffer`), so a TTL-bearing + // KV write falls back to `epoch_system_ms` at apply time. + durability: WalDurability::CallerSupplied { + wal_lsn: None, + resolved_now_ms: None, + minted: None, + }, + }) + .await + } +} diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/mod.rs b/nodedb/src/control/server/pgwire/handler/dispatch/mod.rs new file mode 100644 index 000000000..dd3957b21 --- /dev/null +++ b/nodedb/src/control/server/pgwire/handler/dispatch/mod.rs @@ -0,0 +1,22 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Core dispatch mechanics: single-task dispatch, Raft replication, and local Data Plane submission. +//! +//! Split by concern: +//! - [`entry`]: the public dispatch entry points and the write-HLC +//! bookkeeping wrapper around them. +//! - [`routing`]: the per-task routing decision — freeze/mirror checks, +//! orchestrated DML, exchange resolution, and the replicated-vs-local +//! choice. +//! - [`replicated`]: proposing a write to Raft and shaping the response once +//! it applies. +//! - [`local`]: dispatching a task directly to the local Data Plane, with or +//! without a per-task WAL append. +//! - [`authorize`]: the task-authorization helper and the CRDT +//! admission-gate check shared by every routing path. + +mod authorize; +mod entry; +mod local; +mod replicated; +mod routing; diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/replicated.rs b/nodedb/src/control/server/pgwire/handler/dispatch/replicated.rs new file mode 100644 index 000000000..8b98e7479 --- /dev/null +++ b/nodedb/src/control/server/pgwire/handler/dispatch/replicated.rs @@ -0,0 +1,65 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Propose a write to Raft and shape the response once it applies. + +use std::sync::Arc; + +use crate::bridge::envelope::Response; +use crate::control::server::dispatch_utils::publish_origin_change_events; + +use super::super::core::NodeDbPgHandler; + +/// Inputs for [`NodeDbPgHandler::dispatch_replicated_write`]: the entry to +/// propose, the proposer, and the identity + plan its origin CDC publish needs. +pub(super) struct ReplicatedWrite<'a> { + pub(super) entry: crate::control::wal_replication::ReplicatedEntry, + pub(super) proposer: &'a Arc, + pub(super) authorized: crate::control::server::shared::authorization::AuthorizedTask, +} + +impl NodeDbPgHandler { + /// Dispatch a write through Raft: propose → register waiter → await apply. + /// `ProposeTracker` is race-safe against an entry applying before register. + /// + /// Also the origin CDC publish site; replicas publish nothing (`ChangeFeedOwner::Unowned`). + pub(super) async fn dispatch_replicated_write( + &self, + args: ReplicatedWrite<'_>, + ) -> crate::Result { + let ReplicatedWrite { + entry, + proposer, + authorized, + } = args; + let task = authorized.into_physical_task(); + let tenant_id = task.tenant_id; + let database_id = task.database_id; + let plan = task.plan; + let request_id = self.next_request_id(); + + // `write_version` is the post-write `coll_write_lsn`, surfaced so the session + // can floor a later read-set at it (read-your-writes for cross-shard OCC). + let (payload, write_version) = + crate::control::wal_replication::propose_replicated_entry(&self.state, proposer, entry) + .await?; + + let response = Response { + request_id, + status: crate::bridge::envelope::Status::Ok, + attempt: 1, + partial: false, + payload: payload.into(), + // Authoritative participant WAL LSN — CDC ordering must use it, not zero. + watermark_lsn: write_version, + error_code: None, + read_set_valid: None, + read_version_lsn: write_version, + write_set: Vec::new(), + }; + + // Propose returned: entry is committed and applied. Publish once, from this plan. + publish_origin_change_events(&self.state, tenant_id, database_id, &plan, &response); + + Ok(response) + } +} diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/routing.rs b/nodedb/src/control/server/pgwire/handler/dispatch/routing.rs new file mode 100644 index 000000000..33fbb2be2 --- /dev/null +++ b/nodedb/src/control/server/pgwire/handler/dispatch/routing.rs @@ -0,0 +1,233 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-task routing: freeze/mirror checks, orchestrated DML, exchange +//! resolution, and the replicated-vs-local dispatch choice. + +use std::sync::Arc; + +use crate::bridge::envelope::Response; +use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::exchange::resolve::{ + DistributedReadCapture, Resolved, resolve_and_materialize, +}; +use crate::types::{Lsn, ReadConsistency, TraceId, VShardId}; +use nodedb_physical::physical_task::PhysicalTask; + +use super::super::core::NodeDbPgHandler; +use super::authorize::reject_unadmitted_crdt_apply; +use super::replicated::ReplicatedWrite; + +impl NodeDbPgHandler { + pub(super) async fn dispatch_task_inner( + &self, + mut task: PhysicalTask, + user_id: Option>, + identity: &AuthenticatedIdentity, + shard_watermarks: &mut Vec<(VShardId, Lsn)>, + distributed_reads: &mut Vec, + ) -> crate::Result { + // Reject user writes against a database frozen by a clone materializer sweep. + // Reads/DDL pass through. + use crate::control::security::identity::{Permission, required_permission}; + let perm = required_permission(&task.plan); + if matches!(perm, Permission::Write | Permission::Admin) + && self.state.materialize_freeze.is_frozen(task.database_id) + { + return Err(crate::Error::SourceFrozen { + database_id: task.database_id, + }); + } + + // Mirror enforcement: writes reject on non-promoted mirrors; reads gate by + // ReadConsistency. Catalog lookup skipped for db id=0 to stay allocation-free. + let catalog = self.state.credentials.catalog(); + if task.database_id.as_u64() != 0 + && let Ok(Some(descriptor)) = catalog.get_database(task.database_id) + && let Some(origin) = descriptor.mirror_origin.as_ref() + && !matches!(origin.status, nodedb_types::MirrorStatus::Promoted) + { + if matches!(perm, Permission::Write | Permission::Admin) { + return Err(crate::Error::MirrorReadOnly { + database: descriptor.name.clone(), + }); + } + + use crate::control::server::pgwire::ddl::database::{ + MirrorReadOutcome, check_mirror_read_consistency, + }; + // Defaults to Strong: mirrors aren't the source leader, so reads reject + // unless the session opted into BoundedStaleness or Eventual. + let now_ms = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap_or(std::time::Duration::ZERO) + .as_millis() as u64; + let outcome = check_mirror_read_consistency( + catalog, + task.database_id, + origin, + ReadConsistency::Strong, + now_ms, + ); + if let MirrorReadOutcome::Reject { message, .. } = outcome { + return Err(crate::Error::StaleReadNotLeader { + database: descriptor.name.clone(), + source_cluster: origin.source_cluster.clone(), + detail: message, + }); + } + } + + if matches!( + &task.plan, + crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::InsertSelect { .. } + ) + ) { + let authorized = self.authorize_for_dispatch(identity, &task)?; + return crate::control::insert_select::run_authorized_insert_select( + &self.state, + authorized, + ) + .await; + } + + // Autocommit `MERGE` orchestrates on the Control Plane (`control::merge_orchestrator`). + // In-transaction MERGE buffers for COMMIT replay and never reaches this method. + if matches!( + &task.plan, + crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::Merge { + resolved_inserts: None, + .. + } + ) + ) { + let authorized = self.authorize_for_dispatch(identity, &task)?; + return crate::control::merge_orchestrator::run_authorized_merge( + &self.state, + authorized, + ) + .await; + } + + // Scans the source on its own core and ships raw rows into the plan (source's + // vShard can differ). In-transaction buffers for COMMIT replay instead. + if matches!( + &task.plan, + crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::UpdateFromJoin { + source_rows: None, + .. + } + ) + ) { + let authorized = self.authorize_for_dispatch(identity, &task)?; + return crate::control::update_from_join_orchestrator::run_authorized_update_from_join( + &self.state, + authorized, + ) + .await; + } + + // Can't replicate bare over Raft — a follower has no writing identity to decide + // `$auth.*` against. `write_resolve` resolves it while the identity is live. + if let Some(resolver) = crate::control::write_resolve::resolver_for_plan(&task.plan) + && self.state.async_raft_proposer().is_some() + { + let authorized = self.authorize_for_dispatch(identity, &task)?; + return crate::control::write_resolve::run_authorized_write_resolve( + &self.state, + authorized, + resolver, + ) + .await; + } + + // `DROP ARRAY` reaches every core so each releases its store and segment dir — + // otherwise a follow-up `CREATE ARRAY` carries stale state. + if matches!( + task.plan, + crate::bridge::envelope::PhysicalPlan::Array( + nodedb_physical::physical_plan::ArrayOp::DropArray { .. } + ) + ) { + // Broadcast bypasses the write funnel, so a denied DROP must not + // delete catalog rows or surrogate bindings. + let authorized = self.authorize_for_dispatch(identity, &task)?; + let task = authorized.into_physical_task(); + return crate::control::array_catalog::ddl::run_authorized_drop( + &self.state, + task.tenant_id, + task.database_id, + task.plan, + TraceId::ZERO, + ) + .await; + } + + // Clone-read must run first: resolving derived Exchange plans below + // dispatches straight to the Data Plane, bypassing the clone check. + if let Some(resp) = self + .maybe_intercept_clone_read_early(&task, identity, perm) + .await? + { + return Ok(resp); + } + + // Resolve derived Exchange plans before authorizing the dispatched task. + match resolve_and_materialize( + &self.state, + identity, + task.database_id, + task.tenant_id, + task.plan, + TraceId::ZERO, + task.txn_id, + ) + .await? + { + Resolved::Gathered(resp, wms, caps) => { + *shard_watermarks = wms; + *distributed_reads = caps; + return Ok(resp); + } + Resolved::Plan(resolved_plan) => { + let resolved_plan = *resolved_plan; + task.plan = resolved_plan; + } + Resolved::Stream(stream) => { + return crate::control::server::exchange::gather::stream_to_response(stream).await; + } + } + + reject_unadmitted_crdt_apply(&task.plan)?; + let checked = self + .intercept_and_authorize_for_dispatch(identity, task) + .await?; + let checked = match checked { + crate::control::server::shared::clone_write::CloneCheckedOutcome::Handled(resp) => { + return Ok(resp); + } + crate::control::server::shared::clone_write::CloneCheckedOutcome::Proceed(t) => t, + }; + if let Some(async_proposer) = self.state.async_raft_proposer() + && let Some(entry) = crate::control::wal_replication::to_replicated_entry( + checked.tenant_id(), + checked.database_id(), + checked.vshard_id(), + &crate::control::wal_replication::ReplicableWrite::decide_for_replication( + checked.plan(), + )?, + )? + { + return self + .dispatch_replicated_write(ReplicatedWrite { + entry, + proposer: async_proposer, + authorized: checked.into_authorized(), + }) + .await; + } + self.dispatch_local(checked, user_id).await + } +} diff --git a/nodedb/src/control/server/pgwire/handler/facet.rs b/nodedb/src/control/server/pgwire/handler/facet.rs index dbd7b1d4c..d96c806d8 100644 --- a/nodedb/src/control/server/pgwire/handler/facet.rs +++ b/nodedb/src/control/server/pgwire/handler/facet.rs @@ -15,7 +15,7 @@ use sonic_rs; use crate::bridge::envelope::PhysicalPlan; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::session::SessionId; -use crate::types::{DatabaseId, VShardId}; +use crate::types::DatabaseId; use nodedb_physical::physical_plan::QueryOp; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; @@ -37,7 +37,7 @@ pub(super) async fn execute_facet_counts_sql( .sessions .get_current_database(session_id) .unwrap_or(DatabaseId::DEFAULT); - let vshard = VShardId::from_collection_in_database(database_id, &parsed.collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &parsed.collection).vshard(); // Convert filter text to ScanFilter predicates. let filter_bytes = if parsed.filter.is_empty() { @@ -101,7 +101,7 @@ pub(super) async fn execute_search_with_facets_sql( .sessions .get_current_database(session_id) .unwrap_or(DatabaseId::DEFAULT); - let vshard = VShardId::from_collection_in_database(database_id, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(); let filter_bytes = if filter_text.is_empty() { Vec::new() @@ -305,7 +305,7 @@ fn build_filter_bytes(filter_text: &str) -> PgWireResult> { zerompk::to_msgpack_vec(&filters).map_err(|e| { PgWireError::UserError(Box::new(ErrorInfo::new( "ERROR".to_owned(), - "XX000".to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), format!("filter serialization failed: {e}"), ))) }) diff --git a/nodedb/src/control/server/pgwire/handler/plan.rs b/nodedb/src/control/server/pgwire/handler/plan.rs index a0275f0cf..db3a0e90d 100644 --- a/nodedb/src/control/server/pgwire/handler/plan.rs +++ b/nodedb/src/control/server/pgwire/handler/plan.rs @@ -97,6 +97,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let outcome = calvin_tag_for_plan(&plan).expect("an upsert folds without a round-trip"); let tag: pgwire::messages::response::CommandComplete = render(outcome).into(); diff --git a/nodedb/src/control/server/pgwire/handler/prepared/parser.rs b/nodedb/src/control/server/pgwire/handler/prepared/parser.rs index 11c044fba..e53bb7318 100644 --- a/nodedb/src/control/server/pgwire/handler/prepared/parser.rs +++ b/nodedb/src/control/server/pgwire/handler/prepared/parser.rs @@ -209,7 +209,12 @@ impl NodeDbQueryParser { self.state.auth_stores(), database_id, ); - let permission_cache = self.state.permission_cache.read().await; + // Parse plans against the same authorization state as every other + // planning path. A refusal is an error, not "not plannable". + let permission_cache = + crate::control::security::auth_fence::permission_view(&self.state, identity.tenant_id) + .await + .map_err(|e| crate::control::server::pgwire::types::error_map::error_to_pg(&e))?; let security = crate::control::planner::context::PlanSecurityContext { identity, auth: scope.auth(), diff --git a/nodedb/src/control/server/pgwire/handler/routing/calvin_dispatch.rs b/nodedb/src/control/server/pgwire/handler/routing/calvin_dispatch.rs index 9c694b685..58cd7b32b 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/calvin_dispatch.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/calvin_dispatch.rs @@ -196,7 +196,7 @@ impl NodeDbPgHandler { // invariant is ever broken by a future refactor. PgWireError::UserError(Box::new(ErrorInfo::new( "ERROR".to_owned(), - "XX000".to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), "internal: static Calvin path reached the OLLP dispatch branch".to_owned(), ))) })? diff --git a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs index 29246e846..c665af5cb 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs @@ -167,7 +167,7 @@ impl NodeDbPgHandler { .ok_or_else(|| { PgWireError::UserError(Box::new(ErrorInfo::new( "ERROR".to_owned(), - "XX000".to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), "ClusterArray authorization returned no capability".to_owned(), ))) })?; diff --git a/nodedb/src/control/server/pgwire/handler/routing/execute.rs b/nodedb/src/control/server/pgwire/handler/routing/execute.rs index 6070bce06..0d890968f 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/execute.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/execute.rs @@ -203,7 +203,18 @@ impl NodeDbPgHandler { // Autocommit statement routing: the only reads to widen with are the // ones the materialized-sum settlement stamped on the source rows its // shipped balances were folded from. - let sum_read_vshards = crate::control::planner::calvin::read_vshards_of(&sum_target_reads); + let sum_read_vshards = + match crate::control::planner::calvin::read_vshards_of(&sum_target_reads) { + Ok(vshards) => vshards, + Err(error) => { + let (severity, code, message) = error_to_sqlstate(&error); + return Err(PgWireError::UserError(Box::new(ErrorInfo::new( + severity.to_owned(), + code.to_owned(), + message, + )))); + } + }; match classify_dispatch(&tasks, &sum_read_vshards) { DispatchClass::SingleShard { .. } => { // A single-shard dependent-predicate write (e.g. `DELETE ... diff --git a/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs b/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs index 0b57d5855..a4fd1dfd8 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs @@ -129,14 +129,18 @@ impl NodeDbPgHandler { { return Err(PgWireError::UserError(Box::new(ErrorInfo::new( "ERROR".to_owned(), - "XX000".to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), "internal error: failed to retain descriptor leases for buffered transaction tasks" .to_owned(), )))); } match routed { - Ok(InTxnRoute::Read(routed_task)) => Ok(TxnRouteOutcome::Proceed(routed_task)), + // The pgwire dispatch proposes a write through Raft or appends its + // redo record in the funnel. + Ok(InTxnRoute::Read(routed_task) | InTxnRoute::Autocommit(routed_task)) => { + Ok(TxnRouteOutcome::Proceed(routed_task)) + } Ok(InTxnRoute::Buffered) => Ok(TxnRouteOutcome::Handled(HandledWrite::Opaque)), Ok(InTxnRoute::Staged(outcome)) => Ok(TxnRouteOutcome::Handled(HandledWrite::Dml( staged_dml_outcome(outcome.kind, outcome.affected), @@ -154,7 +158,11 @@ impl NodeDbPgHandler { Some(code) => { crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate(&code) } - None => ("ERROR", "XX000", "unknown data plane error".to_owned()), + None => ( + "ERROR", + nodedb_types::error::sqlstate::INTERNAL_ERROR, + "unknown data plane error".to_owned(), + ), }; Err(PgWireError::UserError(Box::new(ErrorInfo::new( severity.to_owned(), diff --git a/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs b/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs index 97b4b21a7..11067e398 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs @@ -26,7 +26,7 @@ use nodedb_physical::physical_task::PhysicalTask; use super::super::core::NodeDbPgHandler; use super::super::plan::describe_plan; -use super::gateway_fold::{GatewayFold, GatewayShaping}; +use super::gateway_fold::{GatewayFold, GatewayShaping, plan_produces_rows}; /// Meter one gateway-forwarded task, once its response has already shaped /// successfully — mirrors `calvin_dispatch::meter_calvin_task`, the sibling @@ -99,6 +99,8 @@ impl NodeDbPgHandler { // Resolved once for the whole forwarded task set, before the loop. let redaction = QueryRedaction::for_plans(tenant_id, auth, tasks.iter().map(|t| &t.plan)); let shaping = GatewayShaping { + tenant_id, + database_id, projection, result_formats, redaction: &redaction, @@ -127,6 +129,9 @@ impl NodeDbPgHandler { let mut fold = GatewayFold::with_capacity(tasks.len()); for task in tasks { let plan_kind = describe_plan(&task.plan); + // The task moves into authorization below; a row-producing task + // keeps its plan for shaping the rows it answers with. + let shape_plan = plan_produces_rows(plan_kind).then(|| task.plan.clone()); let counts_toward_tag = plan_counts_toward_statement_tag(&task.plan, has_user_write); let metering_info = PlanMeteringInfo::extract(&task.plan); let emitter = crate::control::security::audit::ArcAuditEmitter(std::sync::Arc::clone( @@ -164,6 +169,7 @@ impl NodeDbPgHandler { &mut fold, resp.payload.as_ref(), plan_kind, + shape_plan.as_ref(), counts_toward_tag, &shaping, )?; @@ -201,6 +207,7 @@ impl NodeDbPgHandler { &mut fold, &[], plan_kind, + shape_plan.as_ref(), counts_toward_tag, &shaping, )?; @@ -210,6 +217,7 @@ impl NodeDbPgHandler { &mut fold, payload, plan_kind, + shape_plan.as_ref(), counts_toward_tag, &shaping, )? { diff --git a/nodedb/src/control/server/pgwire/handler/routing/gateway_fold.rs b/nodedb/src/control/server/pgwire/handler/routing/gateway_fold.rs index b84981b01..a897a9f2e 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/gateway_fold.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/gateway_fold.rs @@ -8,8 +8,10 @@ use pgwire::api::results::{FieldFormat, Response}; use pgwire::error::PgWireResult; +use crate::bridge::envelope::PhysicalPlan; use crate::control::server::response_shape::compose::{self, ShapeOutcome}; use crate::control::server::response_shape::redaction::QueryRedaction; +use crate::control::server::response_shape::request::MaterializedShapeRequest; use crate::control::server::response_shape::schema::OutputSchema; use crate::control::server::response_shape::types::{ ShapedRows, StatementTag, payload_to_dml_outcome, @@ -24,6 +26,8 @@ use super::super::shape_encode; /// How forwarded rows shape back to the client. pub(super) struct GatewayShaping<'a> { + pub(super) tenant_id: crate::types::TenantId, + pub(super) database_id: crate::types::DatabaseId, pub(super) projection: Option<&'a OutputSchema>, pub(super) result_formats: &'a [FieldFormat], /// Resolved once over the whole forwarded task set. @@ -85,6 +89,15 @@ impl NodeDbPgHandler { /// not answer the statement (`counts_toward_tag` false: a derived /// implicit-edge write beside the user's own) folds as opaque. /// + /// A row-producing payload shapes exactly as a locally dispatched one + /// does, through `shape_response_materialized` with the task's plan. A + /// forwarded Data-Plane payload has the shape a local core produces, and + /// the plan-dependent steps (the KV point-get `{key, value}` row wrap, the + /// vector surrogate-to-PK translation) must run on it too: shaped without + /// them, a KV point read's stored value decodes as a scalar and yields no + /// row. `shape_plan` is the task's plan, which a row-producing kind always + /// carries (see [`plan_produces_rows`]). + /// /// Returns the rows the payload decoded to, `None` for a passthrough /// payload with no row count. pub(super) fn fold_gateway_payload( @@ -92,21 +105,36 @@ impl NodeDbPgHandler { fold: &mut GatewayFold, payload: &[u8], plan_kind: PlanKind, + shape_plan: Option<&PhysicalPlan>, counts_toward_tag: bool, shaping: &GatewayShaping<'_>, ) -> PgWireResult> { - // Gateway forwarding carries no sequence access: a projection with - // Control-Plane computed columns is refused by the shaper rather - // than NULL-filled. - match compose::shape_payload_no_plan( - payload, - plan_kind, - shaping.projection, - Some(shaping.redaction.ctx(&self.state.redaction)), - None, - ) - .map_err(|e| shape_error_to_pg(&e))? - { + let outcome = match (plan_produces_rows(plan_kind), shape_plan) { + (false, _) => ShapeOutcome::Passthrough, + // Gateway forwarding carries no sequence access: a projection + // with Control-Plane computed columns is refused by the shaper + // rather than NULL-filled. + (true, Some(plan)) => compose::shape_response_materialized(MaterializedShapeRequest { + payload, + plan, + plan_kind, + projection: shaping.projection, + state: &self.state, + database_id: shaping.database_id, + tenant_id: shaping.tenant_id, + redaction: Some(shaping.redaction.ctx(&self.state.redaction)), + sequences: None, + }) + .map_err(|e| shape_error_to_pg(&e))?, + (true, None) => { + return Err(error_to_pg(&crate::Error::Internal { + detail: format!( + "gateway fold: a {plan_kind:?} task reached shaping without its plan" + ), + })); + } + }; + match outcome { ShapeOutcome::Rows(shaped) => { let rows = shaped.rows.len() as u64; if matches!(plan_kind, PlanKind::ReturningRows) { @@ -143,3 +171,15 @@ impl NodeDbPgHandler { } } } + +/// Whether a plan of `kind` answers with rows, which shape through the +/// task's plan. Every other kind folds into the command tag. +pub(super) fn plan_produces_rows(kind: PlanKind) -> bool { + match kind { + PlanKind::SingleDocument + | PlanKind::MultiRow + | PlanKind::ArraySlice + | PlanKind::ReturningRows => true, + PlanKind::Execution | PlanKind::DmlResult(_) | PlanKind::DmlResultByOp => false, + } +} diff --git a/nodedb/src/control/server/pgwire/handler/routing/planning.rs b/nodedb/src/control/server/pgwire/handler/routing/planning.rs index 769a92c39..3b566bac9 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/planning.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/planning.rs @@ -38,7 +38,7 @@ impl NodeDbPgHandler { .ok_or_else(|| { PgWireError::UserError(Box::new(ErrorInfo::new( "FATAL".to_owned(), - "XX000".to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), "connection session metadata is unavailable".to_owned(), ))) })?, @@ -89,7 +89,7 @@ impl NodeDbPgHandler { .ok_or_else(|| { StatementSetupError::protocol( "FATAL", - "XX000", + nodedb_types::error::sqlstate::INTERNAL_ERROR, "connection session metadata is unavailable", ) })?, @@ -205,10 +205,15 @@ impl NodeDbPgHandler { // bumped on every mutation; a cache hit re-validates the stamped // versions against these live values so a revoked grant or dropped // policy evicts the entry instead of replaying a frozen filter. - let current_permission_tree_version = { - let perm_cache = self.state.permission_cache.read().await; - perm_cache.tenant_version(tenant_id.as_u64()) - }; + // + // The checked read refuses unless this node holds every + // authorization change acknowledged before this statement. The plain + // reads below see that state or newer. + let current_permission_tree_version = + crate::control::security::auth_fence::permission_view(&self.state, tenant_id) + .await + .map_err(StatementSetupError::from)? + .tenant_version(tenant_id.as_u64()); let current_rls_version = self.state.rls.tenant_version(tenant_id.as_u64()); let cached_tasks = if bypass_cache { diff --git a/nodedb/src/control/server/pgwire/handler/session_cmds.rs b/nodedb/src/control/server/pgwire/handler/session_cmds.rs index ce70d756d..d4010d887 100644 --- a/nodedb/src/control/server/pgwire/handler/session_cmds.rs +++ b/nodedb/src/control/server/pgwire/handler/session_cmds.rs @@ -202,7 +202,7 @@ impl NodeDbPgHandler { .ok_or_else(|| { PgWireError::UserError(Box::new(ErrorInfo::new( "FATAL".to_owned(), - "XX000".to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), "connection metadata is missing".to_owned(), ))) })?; diff --git a/nodedb/src/control/server/pgwire/handler/session_explain.rs b/nodedb/src/control/server/pgwire/handler/session_explain.rs index 842900300..68988a315 100644 --- a/nodedb/src/control/server/pgwire/handler/session_explain.rs +++ b/nodedb/src/control/server/pgwire/handler/session_explain.rs @@ -48,15 +48,18 @@ impl NodeDbPgHandler { ))]); } Some(Err(error)) => { - let sqlstate = match error { - nodedb_sql::SqlError::UnsupportedConstraint { .. } - | nodedb_sql::SqlError::ConflictingEngineClause { .. } => "0A000", - _ => "42601", - }; + // The SQLSTATE the planner path renders for the same error. + let message = error.to_string(); + let (_, sqlstate, _) = crate::control::server::pgwire::types::error_to_sqlstate( + &crate::control::planner::plan_error_map::map_plan_error( + error, + identity.tenant_id, + ), + ); return Err(PgWireError::UserError(Box::new(ErrorInfo::new( "ERROR".to_owned(), sqlstate.to_owned(), - error.to_string(), + message, )))); } None => {} @@ -75,7 +78,10 @@ impl NodeDbPgHandler { self.state.auth_stores(), database_id, ); - let perm_cache = self.state.permission_cache.read().await; + let perm_cache = + crate::control::security::auth_fence::permission_view(&self.state, tenant_id) + .await + .map_err(|e| crate::control::server::pgwire::types::error_map::error_to_pg(&e))?; let sec = crate::control::planner::context::PlanSecurityContext { identity, auth: scope.auth(), diff --git a/nodedb/src/control/server/pgwire/handler/stream_response.rs b/nodedb/src/control/server/pgwire/handler/stream_response.rs index a63f7c5ff..f839ca70b 100644 --- a/nodedb/src/control/server/pgwire/handler/stream_response.rs +++ b/nodedb/src/control/server/pgwire/handler/stream_response.rs @@ -22,7 +22,7 @@ use crate::control::state::SharedState; use crate::data::executor::response_codec::{decode_payload_to_json, decode_payload_value}; use super::super::ddl_encode::col_type_to_field_with_format; -use super::super::types::{error_to_sqlstate, text_field}; +use super::super::types::{error_to_pg_in_context, error_to_sqlstate, text_field}; use super::shape_encode::{encode_shaped_row, shaped_query_response}; /// The per-request plumbing a lazily-streamed pgwire response owns for its @@ -115,7 +115,7 @@ pub(crate) fn streaming_multirow_response( encoder.encode_field(&item.to_string()).map_err(|e| { PgWireError::UserError(Box::new(ErrorInfo::new( "ERROR".to_owned(), - "XX000".to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), format!("failed to encode streamed row: {e}"), ))) })?; @@ -214,13 +214,8 @@ pub(crate) fn streaming_shaped_response( break; } - let value = decode_payload_value(&batch.payload).map_err(|e| { - PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - "XX000".to_owned(), - format!("failed to decode streamed batch: {e}"), - ))) - })?; + let value = decode_payload_value(&batch.payload) + .map_err(|e| error_to_pg_in_context("failed to decode streamed batch", &e))?; // Resolved once before the first batch was pulled; this only // re-borrows it, so no batch can slip out ahead of the policy. // A streamed plan never carries Control-Plane computed columns: @@ -232,13 +227,7 @@ pub(crate) fn streaming_shaped_response( redaction.as_ref().map(|r| r.ctx(&state.redaction)), None, ) - .map_err(|e| { - PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - "XX000".to_owned(), - format!("failed to shape streamed batch: {e}"), - ))) - })?; + .map_err(|e| error_to_pg_in_context("failed to shape streamed batch", &e))?; for row in &shaped.rows { if emitted >= limit { break; @@ -313,16 +302,15 @@ pub(crate) async fn streaming_star_response( Ok(_) => { return single_pgwire_error(PgWireError::UserError(Box::new(ErrorInfo::new( "ERROR".to_owned(), - "XX000".to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), "streamed batch payload was not a row array".to_owned(), )))); } Err(e) => { - return single_pgwire_error(PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - "XX000".to_owned(), - format!("failed to decode streamed batch: {e}"), - )))); + return single_pgwire_error(error_to_pg_in_context( + "failed to decode streamed batch", + &e, + )); } } } @@ -349,11 +337,10 @@ pub(crate) async fn streaming_star_response( ) { Ok(s) => s, Err(e) => { - return single_pgwire_error(PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - "XX000".to_owned(), - format!("failed to shape streamed batch: {e}"), - )))); + return single_pgwire_error(error_to_pg_in_context( + "failed to shape streamed batch", + &e, + )); } }; // `SELECT *` derives its columns from the rows and has no client-requested diff --git a/nodedb/src/control/server/pgwire/handler/tenant_session.rs b/nodedb/src/control/server/pgwire/handler/tenant_session.rs index 88d2214b9..2af4e91eb 100644 --- a/nodedb/src/control/server/pgwire/handler/tenant_session.rs +++ b/nodedb/src/control/server/pgwire/handler/tenant_session.rs @@ -8,7 +8,7 @@ use pgwire::error::{ErrorInfo, PgWireError, PgWireResult}; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::session::{SessionId, TransactionState}; -use super::super::types::sqlstate_error; +use super::super::types::{error_to_pg_in_context, sqlstate_error}; use super::core::NodeDbPgHandler; impl NodeDbPgHandler { @@ -78,7 +78,7 @@ impl NodeDbPgHandler { let catalog = self.state.credentials.catalog(); let stored = catalog .find_tenant_by_name(value) - .map_err(|error| sqlstate_error("XX000", &format!("catalog read: {error}")))? + .map_err(|error| error_to_pg_in_context("catalog read", &error))? .ok_or_else(|| sqlstate_error("42704", &format!("tenant '{value}' not found")))?; crate::types::TenantId::new(stored.tenant_id) }; diff --git a/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs b/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs index 241402b11..7a937d568 100644 --- a/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs +++ b/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs @@ -38,9 +38,12 @@ impl TxnDataPlane for PgwireTxnDp<'_> { fn dispatch_no_wal<'a>( &'a self, task: PhysicalTask, - wal_lsn: Option, ) -> Pin> + Send + 'a>> { - Box::pin(self.handler.dispatch_task_no_wal(task, None, wal_lsn)) + Box::pin(self.handler.dispatch_task_no_wal(task)) + } + + fn event_source(&self) -> crate::event::EventSource { + crate::event::EventSource::User } } diff --git a/nodedb/src/control/server/pgwire/listener.rs b/nodedb/src/control/server/pgwire/listener.rs index 0228c684f..5dc8393e7 100644 --- a/nodedb/src/control/server/pgwire/listener.rs +++ b/nodedb/src/control/server/pgwire/listener.rs @@ -39,7 +39,11 @@ fn forced_drain_cleanup_ids( impl PgListener { pub async fn bind(addr: SocketAddr) -> crate::Result { - let tcp = TcpListener::bind(addr).await?; + Self::from_listener(TcpListener::bind(addr).await?) + } + + /// Serve on a socket that already listens. + pub fn from_listener(tcp: TcpListener) -> crate::Result { let local_addr = tcp.local_addr()?; info!(%local_addr, "pgwire listener bound"); Ok(Self { diff --git a/nodedb/src/control/server/pgwire/types/error_map.rs b/nodedb/src/control/server/pgwire/types/error_map.rs index fbd48e481..b29a8d3b6 100644 --- a/nodedb/src/control/server/pgwire/types/error_map.rs +++ b/nodedb/src/control/server/pgwire/types/error_map.rs @@ -9,6 +9,8 @@ use crate::OllpExhaustedCause; use crate::bridge::envelope::{ErrorCode, Status}; use crate::control::server::response_shape::types::DmlFoldError; +pub(crate) use super::numeric_sqlstate::numeric_code_to_sqlstate; + /// Create a pgwire ErrorResponse with a SQLSTATE code. pub fn sqlstate_error(code: &str, message: &str) -> PgWireError { PgWireError::UserError(Box::new(ErrorInfo::new( @@ -22,7 +24,7 @@ pub fn sqlstate_error(code: &str, message: &str) -> PgWireError { /// Two tasks of one statement disagreeing on their verb is a planner bug, /// so it surfaces as an internal error. pub fn dml_fold_error_to_pg(e: &DmlFoldError) -> PgWireError { - sqlstate_error("XX000", &e.to_string()) + sqlstate_error(sqlstate::INTERNAL_ERROR, &e.to_string()) } /// Map a NodeDB `Error` to the pgwire error the client reads, through the @@ -36,6 +38,17 @@ pub fn error_to_pg(err: &crate::Error) -> PgWireError { ))) } +/// Map a NodeDB `Error` to the pgwire error the client reads, with `context` +/// before its message. The SQLSTATE stays the error's own. +pub fn error_to_pg_in_context(context: &str, err: &crate::Error) -> PgWireError { + let (severity, code, message) = error_to_sqlstate(err); + PgWireError::UserError(Box::new(ErrorInfo::new( + severity.to_owned(), + code.to_owned(), + format!("{context}: {message}"), + ))) +} + /// Map an error raised while shaping a response to the pgwire error the /// client reads, with the SQLSTATE its numeric code maps to. A per-row /// sequence accessor refusal (`42704`, `55000`) or a division by zero @@ -52,7 +65,7 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str ("ERROR", sqlstate::BACKUP_TENANT_MISMATCH, err.to_string()) } crate::Error::BackupKeyMismatch => { - ("ERROR", sqlstate::BACKUP_KEY_MISMATCH, err.to_string()) + ("ERROR", sqlstate::BACKUP_KEY_MISMATCH.0, err.to_string()) } crate::Error::PlanError { detail } => ("ERROR", sqlstate::SYNTAX_ERROR, detail.clone()), crate::Error::CollectionNotFound { collection, .. } => ( @@ -75,6 +88,9 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str crate::Error::FeatureNotSupported { detail } => { ("ERROR", sqlstate::FEATURE_NOT_SUPPORTED, detail.clone()) } + crate::Error::NotInTransactionBlock { .. } => { + ("ERROR", sqlstate::ACTIVE_SQL_TRANSACTION, err.to_string()) + } crate::Error::UndefinedFunction { name } => ( "ERROR", sqlstate::UNDEFINED_FUNCTION, @@ -102,6 +118,9 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str ("ERROR", sqlstate::UNDEFINED_COLUMN, err.to_string()) } crate::Error::DivisionByZero => ("ERROR", sqlstate::DIVISION_BY_ZERO, err.to_string()), + crate::Error::DataException { detail } => { + ("ERROR", sqlstate::DATA_EXCEPTION, detail.clone()) + } crate::Error::InvalidLimitValue { .. } => { ("ERROR", sqlstate::INVALID_LIMIT_VALUE, err.to_string()) } @@ -115,19 +134,71 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str ), crate::Error::RejectedConstraint { constraint, detail, .. - } => { - let code = if constraint == "not_null" { - sqlstate::NOT_NULL_VIOLATION - } else { - sqlstate::UNIQUE_VIOLATION - }; - ("ERROR", code, detail.clone()) - } + } => ( + "ERROR", + crate::control::server::shared::ddl::sqlstate::constraint_sqlstate(constraint), + detail.clone(), + ), crate::Error::TxnOverlayMemoryExceeded { .. } => { ("ERROR", sqlstate::PROGRAM_LIMIT_EXCEEDED, err.to_string()) } + // Control-Plane twins of Data-Plane codes take the SQLSTATE their + // Data-Plane code has, so one condition answers one class wherever + // it is detected. + crate::Error::RejectedPrevalidation { .. } | crate::Error::InsufficientBalance { .. } => { + ("ERROR", sqlstate::CHECK_VIOLATION, err.to_string()) + } + crate::Error::RetryableRefusal { .. } => { + ("ERROR", sqlstate::SERIALIZATION_FAILURE, err.to_string()) + } + crate::Error::AppendOnlyViolation { .. } => { + ("ERROR", sqlstate::APPEND_ONLY_VIOLATION, err.to_string()) + } + crate::Error::BalanceViolation { .. } => { + ("ERROR", sqlstate::BALANCE_VIOLATION, err.to_string()) + } + crate::Error::PeriodLocked { .. } => ("ERROR", sqlstate::PERIOD_LOCKED, err.to_string()), + crate::Error::PeriodLockMisconfigured { .. } => ( + "ERROR", + sqlstate::PERIOD_LOCK_MISCONFIGURED, + err.to_string(), + ), + crate::Error::RetentionViolation { .. } => { + ("ERROR", sqlstate::RETENTION_VIOLATION, err.to_string()) + } + crate::Error::LegalHoldActive { .. } => { + ("ERROR", sqlstate::LEGAL_HOLD_ACTIVE, err.to_string()) + } + crate::Error::StateTransitionViolation { .. } => ( + "ERROR", + sqlstate::STATE_TRANSITION_VIOLATION, + err.to_string(), + ), + crate::Error::TransitionCheckViolation { .. } => ( + "ERROR", + sqlstate::TRANSITION_CHECK_VIOLATION, + err.to_string(), + ), + crate::Error::TypeGuardViolation { .. } => { + ("ERROR", sqlstate::TYPE_GUARD_VIOLATION, err.to_string()) + } + crate::Error::TypeMismatch { .. } => ("ERROR", sqlstate::CANNOT_COERCE, err.to_string()), crate::Error::DeadlineExceeded { .. } => { - ("ERROR", sqlstate::QUERY_CANCELED, err.to_string()) + ("ERROR", sqlstate::QUERY_CANCELED.0, err.to_string()) + } + // Nothing ran, and a retry plans against caught-up state. + crate::Error::AuthorizationStateBehind { .. } => { + ("ERROR", sqlstate::STALE_READ_NOT_LEADER, err.to_string()) + } + // Nothing was applied, and a retry succeeds once the group's majority + // is reachable again. + crate::Error::GroupQuorumUnavailable { .. } => { + ("ERROR", sqlstate::LOCK_NOT_AVAILABLE, err.to_string()) + } + // Nothing was restored, and a retry succeeds once a replica of the + // group answers. + crate::Error::GroupMarksUnavailable { .. } => { + ("ERROR", sqlstate::LOCK_NOT_AVAILABLE, err.to_string()) } crate::Error::ConflictRetry { .. } => { ("ERROR", sqlstate::SERIALIZATION_FAILURE, err.to_string()) @@ -146,6 +217,16 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str crate::Error::SourceFrozen { .. } => { ("ERROR", sqlstate::SERIALIZATION_FAILURE, err.to_string()) } + // A descriptor changed under the statement and the server's own + // retries ran out. The client retries the statement, so it takes + // SERIALIZATION_FAILURE (40001), the SQLSTATE drivers retry on. + crate::Error::RetryableSchemaChanged { .. } => { + ("ERROR", sqlstate::SERIALIZATION_FAILURE, err.to_string()) + } + // The session's bearer token expired. The client re-authenticates. + crate::Error::SessionTokenExpired => { + ("ERROR", sqlstate::AUTH_TOKEN_EXPIRED.0, err.to_string()) + } crate::Error::CloneWriteRequiresMaterialize { .. } => ( "ERROR", sqlstate::CLONE_WRITE_REQUIRES_MATERIALIZE.0, @@ -160,6 +241,11 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str crate::Error::RateExceeded { .. } => { ("ERROR", sqlstate::TOO_MANY_CONNECTIONS, err.to_string()) } + // A dispatcher capacity refusal enqueued nothing. SERVER_OVERLOAD + // (57P03) is transient: the client retries after a backoff. + crate::Error::DispatchCapacity { .. } => { + ("ERROR", sqlstate::SERVER_OVERLOAD, err.to_string()) + } crate::Error::MemoryExhausted { .. } => ("ERROR", sqlstate::OUT_OF_MEMORY, err.to_string()), crate::Error::Backpressure { .. } => ("ERROR", sqlstate::OUT_OF_MEMORY, err.to_string()), crate::Error::FanOutExceeded { .. } => { @@ -218,62 +304,99 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str numeric_code_to_sqlstate(e.code()), e.message().to_string(), ), - _ => ("ERROR", sqlstate::INTERNAL_ERROR, err.to_string()), - } -} - -/// Map a numeric `ErrorCode` received from a remote node back to a SQLSTATE. -/// Local errors map by variant identity above; a remote error arrives as a bare -/// numeric code, so this recovers the classification. Each bucket mirrors the -/// sqlstate the corresponding local variant arm chooses above for the same -/// numeric code, so a constraint violation (say) maps to the same SQLSTATE -/// whether it happened locally or on a remote node. Unmapped/unknown codes -/// fall back to INTERNAL_ERROR — the behaviour before codes were preserved. -pub(crate) fn numeric_code_to_sqlstate(code: nodedb_types::error::ErrorCode) -> &'static str { - use nodedb_types::error::ErrorCode as Ec; - match code { - // Mirrors the `RejectedConstraint` arm. - Ec::CONSTRAINT_VIOLATION => sqlstate::UNIQUE_VIOLATION, - // Mirrors the `ConflictRetry` / `CalvinSerializationConflict` / - // `SourceFrozen` arms, and `OllpExhausted` when it exhausted on drift. - Ec::WRITE_CONFLICT => sqlstate::SERIALIZATION_FAILURE, - // Mirrors the `DeadlineExceeded` arm. - Ec::DEADLINE_EXCEEDED => sqlstate::QUERY_CANCELED, - // Mirrors the `CollectionNotFound` / `CollectionDeactivated` arms. - Ec::COLLECTION_NOT_FOUND | Ec::COLLECTION_DEACTIVATED => sqlstate::UNDEFINED_TABLE, - // Mirrors the `DocumentNotFound` arm. - Ec::DOCUMENT_NOT_FOUND => sqlstate::NO_DATA, - // Mirrors the `BadRequest` / `PlanError` arms. - Ec::BAD_REQUEST | Ec::PLAN_ERROR => sqlstate::SYNTAX_ERROR, - // Mirrors the `UndefinedFunction` arm. - Ec::UNDEFINED_FUNCTION => sqlstate::UNDEFINED_FUNCTION, - // Mirrors the `UndefinedObject` arm. - Ec::UNDEFINED_OBJECT => sqlstate::UNDEFINED_OBJECT, - // Mirrors the `ObjectNotInPrerequisiteState` arm. - Ec::OBJECT_NOT_READY => sqlstate::OBJECT_NOT_IN_PREREQUISITE_STATE, - // Mirrors the `UndefinedColumn` arm. - Ec::UNDEFINED_COLUMN => sqlstate::UNDEFINED_COLUMN, - // Mirrors the `AmbiguousColumn` arm. - Ec::AMBIGUOUS_COLUMN => sqlstate::AMBIGUOUS_COLUMN, - // Mirrors the `DivisionByZero` arm. - Ec::DIVISION_BY_ZERO => sqlstate::DIVISION_BY_ZERO, - // Mirrors the `InvalidLimitValue` arm. - Ec::INVALID_LIMIT_VALUE => sqlstate::INVALID_LIMIT_VALUE, - // Mirrors the `FanOutExceeded` arm. - Ec::FAN_OUT_EXCEEDED => sqlstate::STATEMENT_TOO_COMPLEX, - // Mirrors the `RejectedAuthz` arm. - Ec::AUTHORIZATION_DENIED => sqlstate::INSUFFICIENT_PRIVILEGE, - // Mirrors the `RateExceeded` arm. - Ec::RATE_EXCEEDED => sqlstate::TOO_MANY_CONNECTIONS, - // Mirrors the `MemoryExhausted` / `Backpressure` arms. - Ec::MEMORY_EXHAUSTED => sqlstate::OUT_OF_MEMORY, - // Mirrors the `NoLeader` arm. - Ec::NO_LEADER => sqlstate::LOCK_NOT_AVAILABLE, - // Mirrors the `NotLeader` arm. - Ec::NOT_LEADER => sqlstate::DATABASE_DROPPED, - // Mirrors the `CloneWriteRequiresMaterialize` arm. - Ec::CLONE_WRITE_REQUIRES_MATERIALIZE => sqlstate::CLONE_WRITE_REQUIRES_MATERIALIZE.0, - _ => sqlstate::INTERNAL_ERROR, + // A DDL error keeps the exact SQLSTATE its statement reports. + crate::Error::Ddl(ddl) => ( + "ERROR", + crate::control::server::shared::ddl::static_sqlstate::static_sqlstate(&ddl.sqlstate), + ddl.message.clone(), + ), + // The DDL path renders a regressed consumer offset as an invalid + // parameter value, so the typed error takes that class too. + crate::Error::OffsetRegression { .. } => { + ("ERROR", sqlstate::INVALID_PARAMETER_VALUE, err.to_string()) + } + // A full admission queue is a rate refusal, its public code's class. + crate::Error::VShardAdmissionCapacityExceeded { .. } => { + ("ERROR", sqlstate::TOO_MANY_CONNECTIONS, err.to_string()) + } + // The CRDT frontier kept moving. The client retries the write. + crate::Error::CrdtAdmissionRetriesExhausted { .. } => { + ("ERROR", sqlstate::SERIALIZATION_FAILURE, err.to_string()) + } + crate::Error::CrdtAdmissionTimeout { .. } => { + ("ERROR", sqlstate::QUERY_CANCELED.0, err.to_string()) + } + // Statements refused inside an explicit transaction block share + // the class of `NotInTransactionBlock`. + crate::Error::CrdtApplyForbiddenInTransaction + | crate::Error::CrossShardInExplicitTransaction => { + ("ERROR", sqlstate::ACTIVE_SQL_TRANSACTION, err.to_string()) + } + // Each variant here has the public code `BAD_REQUEST`, so pgwire + // renders the class that code renders on native and across nodes. + crate::Error::CrdtAdmissionInvalidPlan { .. } + | crate::Error::CrdtAdmissionCallerFence + | crate::Error::CrdtApplyRequiresAdmission + | crate::Error::ExecutionLimitExceeded { .. } + | crate::Error::LimitExceeded { .. } + | crate::Error::Promql(_) + | crate::Error::SequencerUnavailable + | crate::Error::SessionCapExceeded { .. } + | crate::Error::SessionIdleTimeout + | crate::Error::SessionKilledByAdmin + | crate::Error::SessionUserDropped + | crate::Error::OidcProviderTenantUnbound + | crate::Error::OidcProviderTenantUnavailable { .. } + | crate::Error::ExternalRoleUndefined { .. } + | crate::Error::OidcNoDefaultDatabase { .. } + | crate::Error::RoleInheritanceCycle { .. } + | crate::Error::RoleInheritanceDepthExceeded { .. } => { + ("ERROR", sqlstate::SYNTAX_ERROR, err.to_string()) + } + crate::Error::DependentObjectsExist { .. } | crate::Error::RoleInUse { .. } => ( + "ERROR", + sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, + err.to_string(), + ), + crate::Error::QuotaOvercommit { .. } => { + ("ERROR", sqlstate::QUOTA_OVERCOMMIT, err.to_string()) + } + crate::Error::TenantVectorDimExceeded { .. } + | crate::Error::TenantGraphDepthExceeded { .. } => { + ("ERROR", sqlstate::QUOTA_EXCEEDED, err.to_string()) + } + crate::Error::MirrorReadOnly { .. } => ( + "ERROR", + sqlstate::READ_ONLY_SQL_TRANSACTION, + err.to_string(), + ), + // The client redirects the strong read to the source cluster. + crate::Error::StaleReadNotLeader { .. } => { + ("ERROR", sqlstate::STALE_READ_NOT_LEADER, err.to_string()) + } + // Server-side faults and system defects. The client can act on none + // of them, and their public codes are internal classes. + crate::Error::MaterializedSumResolutionMissing { .. } + | crate::Error::RetryableLeaderChange { .. } + | crate::Error::MetadataLeaderUnavailable + | crate::Error::Wal(_) + | crate::Error::Dispatch { .. } + | crate::Error::Storage { .. } + | crate::Error::ColdStorage { .. } + | crate::Error::Serialization { .. } + | crate::Error::Codec { .. } + | crate::Error::SegmentCorrupted { .. } + | crate::Error::Crdt(_) + | crate::Error::Io(_) + | crate::Error::Config { .. } + | crate::Error::Encryption { .. } + | crate::Error::Bridge { .. } + | crate::Error::VersionCompat { .. } + | crate::Error::Internal { .. } + | crate::Error::DescriptorVersionAnomaly { .. } + | crate::Error::CatalogIntegrityViolation { .. } + | crate::Error::CollectionPurgeRowMissing { .. } + | crate::Error::CascadeCycle { .. } => ("ERROR", sqlstate::INTERNAL_ERROR, err.to_string()), } } @@ -297,8 +420,37 @@ pub fn response_status_to_sqlstate( if let Some(code) = error_code { Some(crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate(code)) } else { - Some(("ERROR", "XX000", "unknown data plane error".into())) + Some(( + "ERROR", + sqlstate::INTERNAL_ERROR, + "unknown data plane error".into(), + )) + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// A typed error behind a context prefix keeps its own SQLSTATE. + #[test] + fn an_error_in_context_keeps_its_sqlstate() { + let missing = crate::Error::CollectionNotFound { + tenant_id: crate::types::TenantId::new(1), + collection: "orders".into(), + }; + match error_to_pg_in_context("catalog read", &missing) { + PgWireError::UserError(info) => { + assert_eq!(info.code, sqlstate::UNDEFINED_TABLE); + assert!( + info.message.starts_with("catalog read: "), + "{}", + info.message + ); } + other => panic!("expected a user error, got {other:?}"), } } } diff --git a/nodedb/src/control/server/pgwire/types/mod.rs b/nodedb/src/control/server/pgwire/types/mod.rs index b15b76920..b7aa1ef45 100644 --- a/nodedb/src/control/server/pgwire/types/mod.rs +++ b/nodedb/src/control/server/pgwire/types/mod.rs @@ -7,11 +7,12 @@ pub mod error_map; pub mod field; +pub mod numeric_sqlstate; pub mod parse; pub mod privilege; pub use error_map::{ - dml_fold_error_to_pg, error_to_pg, error_to_sqlstate, notice_warning, + dml_fold_error_to_pg, error_to_pg, error_to_pg_in_context, error_to_sqlstate, notice_warning, response_status_to_sqlstate, shape_error_to_pg, sqlstate_error, }; pub use field::{ diff --git a/nodedb/src/control/server/pgwire/types/numeric_sqlstate.rs b/nodedb/src/control/server/pgwire/types/numeric_sqlstate.rs new file mode 100644 index 000000000..112c6fe9a --- /dev/null +++ b/nodedb/src/control/server/pgwire/types/numeric_sqlstate.rs @@ -0,0 +1,302 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Public numeric `ErrorCode` to PostgreSQL SQLSTATE mapping. + +use nodedb_types::error::sqlstate; + +/// Map a numeric `ErrorCode` received from a remote node back to a SQLSTATE. +/// Local errors map by variant identity in `error_to_sqlstate`. A remote +/// error arrives as a bare numeric code, so this recovers the class. Each +/// bucket mirrors the SQLSTATE the local variant arm chooses for the same +/// numeric code, so a constraint violation (say) maps to the same SQLSTATE +/// whether it happened locally or on a remote node. `ErrorCode` is an open +/// numeric newtype, so an unmapped or unknown code renders `INTERNAL_ERROR`. +pub(crate) fn numeric_code_to_sqlstate(code: nodedb_types::error::ErrorCode) -> &'static str { + use nodedb_types::error::ErrorCode as Ec; + match code { + // Mirrors the `RejectedConstraint` arm. + Ec::CONSTRAINT_VIOLATION => sqlstate::UNIQUE_VIOLATION, + // Mirrors the `ConflictRetry` / `CalvinSerializationConflict` / + // `SourceFrozen` / `RetryableSchemaChanged` arms, and `OllpExhausted` + // when it exhausted on drift. + Ec::WRITE_CONFLICT => sqlstate::SERIALIZATION_FAILURE, + // Mirrors the `CalvinParticipantError` arm. + Ec::TRANSACTION_ROLLBACK => sqlstate::TRANSACTION_ROLLBACK, + // Mirrors the `NotInTransactionBlock` / `CrdtApplyForbiddenInTransaction` + // / `CrossShardInExplicitTransaction` arms. + Ec::ACTIVE_SQL_TRANSACTION => sqlstate::ACTIVE_SQL_TRANSACTION, + // Mirrors the `DependentObjectsExist` arm. + Ec::DEPENDENT_OBJECTS_EXIST => sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, + // Mirrors the `DeadlineExceeded` arm. + Ec::DEADLINE_EXCEEDED => sqlstate::QUERY_CANCELED.0, + // Mirrors the `CollectionNotFound` / `CollectionDeactivated` arms. + Ec::COLLECTION_NOT_FOUND | Ec::COLLECTION_DEACTIVATED => sqlstate::UNDEFINED_TABLE, + // Mirrors the `DocumentNotFound` arm. + Ec::DOCUMENT_NOT_FOUND => sqlstate::NO_DATA, + // Mirrors the `BadRequest` / `PlanError` arms. + Ec::BAD_REQUEST | Ec::PLAN_ERROR => sqlstate::SYNTAX_ERROR, + // Mirrors the `UndefinedFunction` arm. + Ec::UNDEFINED_FUNCTION => sqlstate::UNDEFINED_FUNCTION, + // Mirrors the `UndefinedObject` arm. + Ec::UNDEFINED_OBJECT => sqlstate::UNDEFINED_OBJECT, + // Mirrors the `ObjectNotInPrerequisiteState` arm. + Ec::OBJECT_NOT_READY => sqlstate::OBJECT_NOT_IN_PREREQUISITE_STATE, + // Mirrors the `UndefinedColumn` arm. + Ec::UNDEFINED_COLUMN => sqlstate::UNDEFINED_COLUMN, + // Mirrors the `AmbiguousColumn` arm. + Ec::AMBIGUOUS_COLUMN => sqlstate::AMBIGUOUS_COLUMN, + // Mirrors the `DivisionByZero` arm. + Ec::DIVISION_BY_ZERO => sqlstate::DIVISION_BY_ZERO, + // Mirrors the `DataException` arm. + Ec::DATA_EXCEPTION => sqlstate::DATA_EXCEPTION, + // Mirrors the `InvalidLimitValue` arm. + Ec::INVALID_LIMIT_VALUE => sqlstate::INVALID_LIMIT_VALUE, + // Mirrors the `FanOutExceeded` arm. + Ec::FAN_OUT_EXCEEDED => sqlstate::STATEMENT_TOO_COMPLEX, + // Mirrors the `RejectedAuthz` arm. + Ec::AUTHORIZATION_DENIED => sqlstate::INSUFFICIENT_PRIVILEGE, + // Mirrors the `SessionTokenExpired` arm. + Ec::AUTH_EXPIRED => sqlstate::AUTH_TOKEN_EXPIRED.0, + // Mirrors the credential-failure frames of pgwire and native login. + Ec::AUTHENTICATION_FAILED => sqlstate::INVALID_AUTHORIZATION, + // Mirrors the `RateExceeded` arm. + Ec::RATE_EXCEEDED => sqlstate::TOO_MANY_CONNECTIONS, + // Mirrors the `MemoryExhausted` / `Backpressure` arms. + Ec::MEMORY_EXHAUSTED => sqlstate::OUT_OF_MEMORY, + // Mirrors the `DispatchCapacity` arm. + Ec::SERVER_OVERLOAD => sqlstate::SERVER_OVERLOAD, + // Mirrors the `NoLeader` arm. + Ec::NO_LEADER => sqlstate::LOCK_NOT_AVAILABLE, + // Mirrors the `NotLeader` arm. + Ec::NOT_LEADER => sqlstate::DATABASE_DROPPED, + // Mirrors the `CloneWriteRequiresMaterialize` arm. + Ec::CLONE_WRITE_REQUIRES_MATERIALIZE => sqlstate::CLONE_WRITE_REQUIRES_MATERIALIZE.0, + // Mirrors the `BackupTenantMismatch` arm. + Ec::BACKUP_TENANT_MISMATCH => sqlstate::BACKUP_TENANT_MISMATCH, + // Mirrors the `BackupKeyMismatch` arm. + Ec::BACKUP_KEY_MISMATCH => sqlstate::BACKUP_KEY_MISMATCH.0, + // Mirrors the `QuotaOvercommit` arm. + Ec::QUOTA_OVERCOMMIT => sqlstate::QUOTA_OVERCOMMIT, + // Mirrors the `TenantVectorDimExceeded` / `TenantGraphDepthExceeded` + // arms. + Ec::TENANT_VECTOR_DIM_EXCEEDED | Ec::TENANT_GRAPH_DEPTH_EXCEEDED => { + sqlstate::QUOTA_EXCEEDED + } + // Mirrors the `MirrorReadOnly` arm. + Ec::MIRROR_READ_ONLY => sqlstate::READ_ONLY_SQL_TRANSACTION, + // Mirrors the `StaleReadNotLeader` arm. + Ec::STALE_READ_NOT_LEADER => sqlstate::STALE_READ_NOT_LEADER, + // The codes below mirror the Data-Plane code table + // (`error_code_to_sqlstate`) for the public code each Data-Plane code + // classifies to, so a verdict that crossed a node as a numeric code + // renders in the class it has locally. + Ec::PREVALIDATION_REJECTED | Ec::INSUFFICIENT_BALANCE => sqlstate::CHECK_VIOLATION, + Ec::APPEND_ONLY_VIOLATION => sqlstate::APPEND_ONLY_VIOLATION, + Ec::BALANCE_VIOLATION => sqlstate::BALANCE_VIOLATION, + Ec::PERIOD_LOCKED => sqlstate::PERIOD_LOCKED, + Ec::PERIOD_LOCK_MISCONFIGURED => sqlstate::PERIOD_LOCK_MISCONFIGURED, + Ec::STATE_TRANSITION_VIOLATION => sqlstate::STATE_TRANSITION_VIOLATION, + Ec::TRANSITION_CHECK_VIOLATION => sqlstate::TRANSITION_CHECK_VIOLATION, + Ec::RETENTION_VIOLATION => sqlstate::RETENTION_VIOLATION, + Ec::LEGAL_HOLD_ACTIVE => sqlstate::LEGAL_HOLD_ACTIVE, + Ec::TYPE_GUARD_VIOLATION => sqlstate::TYPE_GUARD_VIOLATION, + Ec::TYPE_MISMATCH => sqlstate::CANNOT_COERCE, + Ec::OVERFLOW => sqlstate::NUMERIC_VALUE_OUT_OF_RANGE, + Ec::COLLECTION_DRAINING => sqlstate::CANNOT_CONNECT_NOW, + Ec::SQL_NOT_ENABLED => sqlstate::FEATURE_NOT_SUPPORTED, + Ec::PROGRAM_LIMIT_EXCEEDED => sqlstate::PROGRAM_LIMIT_EXCEEDED, + // The codes below mirror the SQLSTATE the DDL layer sends with each + // code, and the SQLSTATE `code_for_sqlstate` reads back into it. + Ec::DATABASE_NOT_FOUND => sqlstate::INVALID_CATALOG_NAME, + Ec::ALREADY_EXISTS => sqlstate::DUPLICATE_OBJECT, + Ec::NOT_FOUND | Ec::MOVE_TENANT_ALREADY_AT_TARGET => sqlstate::NO_DATA, + Ec::TENANT_QUOTA_EXCEEDED | Ec::DATABASE_QUOTA_EXCEEDED => sqlstate::QUOTA_EXCEEDED, + Ec::CLONE_DEPTH_EXCEEDED => sqlstate::CLONE_DEPTH_EXCEEDED, + Ec::CANNOT_CLONE_MIRROR => sqlstate::CANNOT_CLONE_MIRROR.0, + Ec::CANNOT_DROP_DEFAULT_DATABASE => sqlstate::CANNOT_DROP_DEFAULT_DATABASE.0, + Ec::CLONE_DEPENDENCY => sqlstate::CLONE_DEPENDENCY.0, + Ec::CLONE_PREDATES_QUERY_TIME => sqlstate::CLONE_PREDATES_QUERY_TIME, + Ec::MIRROR_NOT_PROMOTED => sqlstate::OBJECT_NOT_IN_PREREQUISITE_STATE, + Ec::MOVE_TENANT_DRAIN_TIMEOUT => sqlstate::MOVE_TENANT_DRAIN_TIMEOUT.0, + Ec::MOVE_TENANT_PREFLIGHT_FAILED => sqlstate::MOVE_TENANT_PREFLIGHT_FAILED, + // The codes below have no server-side `Error` variant. Each takes the + // SQLSTATE class of the condition it names. + Ec::HANDSHAKE_FAILED => sqlstate::PROTOCOL_VIOLATION, + Ec::SYNC_CONNECTION_FAILED | Ec::NODE_UNREACHABLE => sqlstate::CONNECTION_FAILURE, + Ec::SHAPE_SUBSCRIPTION_FAILED => sqlstate::SERVER_REJECTED_ESTABLISHMENT, + // Mirrors the Data-Plane `SyncRejected` code. + Ec::SYNC_DELTA_REJECTED => sqlstate::CHECK_VIOLATION, + Ec::MIGRATION_IN_PROGRESS => sqlstate::CANNOT_CONNECT_NOW, + // Internal faults: pgwire sends `XX000` for every `Error` variant + // that classifies to one of these. `INTERNAL_CODES` lists them. + _ => sqlstate::INTERNAL_ERROR, + } +} + +/// The codes whose correct SQLSTATE is `XX000`: each names a server-side +/// fault the client cannot act on. +#[cfg(test)] +const INTERNAL_CODES: &[(nodedb_types::error::ErrorCode, &str)] = { + use nodedb_types::error::ErrorCode as Ec; + &[ + ( + Ec::ARRAY, + "array-catalog registry invariant at bootstrap or open", + ), + ( + Ec::MOVE_TENANT_SNAPSHOT_FAILED, + "server-side phase failure; DDL sends XX000", + ), + ( + Ec::MOVE_TENANT_CUTOVER_FAILED, + "server-side phase failure; DDL sends XX000", + ), + (Ec::STORAGE, "storage engine or I/O fault"), + (Ec::SEGMENT_CORRUPTED, "on-disk segment corruption"), + (Ec::COLD_STORAGE, "cold-tier backend fault"), + (Ec::WAL, "WAL append or fsync fault"), + (Ec::SERIALIZATION, "internal payload encode or decode fault"), + (Ec::CODEC, "compression codec fault"), + (Ec::CONFIG, "server configuration fault"), + (Ec::CLUSTER, "cluster version-compatibility fault"), + (Ec::ENCRYPTION, "at-rest encryption fault"), + (Ec::INTERNAL, "generic internal fault"), + (Ec::BRIDGE, "SPSC bridge fault"), + (Ec::DISPATCH, "Control-to-Data-Plane dispatch fault"), + ] +}; + +/// Codes whose SQLSTATE another code owns: `code_for_sqlstate` reads that +/// SQLSTATE back as the owner. Each pair is `(code, owner)`. +#[cfg(test)] +const SHARED_SQLSTATE: &[( + nodedb_types::error::ErrorCode, + nodedb_types::error::ErrorCode, +)] = { + use nodedb_types::error::ErrorCode as Ec; + &[ + (Ec::PREVALIDATION_REJECTED, Ec::CONSTRAINT_VIOLATION), + (Ec::INSUFFICIENT_BALANCE, Ec::CONSTRAINT_VIOLATION), + (Ec::SYNC_DELTA_REJECTED, Ec::CONSTRAINT_VIOLATION), + (Ec::DOCUMENT_NOT_FOUND, Ec::NOT_FOUND), + (Ec::MOVE_TENANT_ALREADY_AT_TARGET, Ec::NOT_FOUND), + (Ec::COLLECTION_DEACTIVATED, Ec::COLLECTION_NOT_FOUND), + (Ec::COLLECTION_DRAINING, Ec::SERVER_OVERLOAD), + (Ec::MIGRATION_IN_PROGRESS, Ec::SERVER_OVERLOAD), + (Ec::PLAN_ERROR, Ec::BAD_REQUEST), + (Ec::TENANT_QUOTA_EXCEEDED, Ec::QUOTA_OVERCOMMIT), + (Ec::DATABASE_QUOTA_EXCEEDED, Ec::QUOTA_OVERCOMMIT), + (Ec::TENANT_VECTOR_DIM_EXCEEDED, Ec::QUOTA_OVERCOMMIT), + (Ec::TENANT_GRAPH_DEPTH_EXCEEDED, Ec::QUOTA_OVERCOMMIT), + (Ec::CANNOT_CLONE_MIRROR, Ec::SQL_NOT_ENABLED), + (Ec::CANNOT_DROP_DEFAULT_DATABASE, Ec::SQL_NOT_ENABLED), + (Ec::CLONE_DEPENDENCY, Ec::OBJECT_NOT_READY), + (Ec::CLONE_WRITE_REQUIRES_MATERIALIZE, Ec::OBJECT_NOT_READY), + (Ec::MIRROR_NOT_PROMOTED, Ec::OBJECT_NOT_READY), + (Ec::CLONE_PREDATES_QUERY_TIME, Ec::DATA_EXCEPTION), + (Ec::BACKUP_TENANT_MISMATCH, Ec::DATA_EXCEPTION), + (Ec::BACKUP_KEY_MISMATCH, Ec::AUTHENTICATION_FAILED), + (Ec::AUTH_EXPIRED, Ec::AUTHENTICATION_FAILED), + (Ec::STALE_READ_NOT_LEADER, Ec::NO_LEADER), + (Ec::SYNC_CONNECTION_FAILED, Ec::NODE_UNREACHABLE), + ] +}; + +/// Codes whose SQLSTATE has no default meaning, so `code_for_sqlstate` +/// reads it back as `INTERNAL` rather than guess. `57014` is sent for a +/// deadline and for a cancellation that is not one. +#[cfg(test)] +const NO_DEFAULT_SQLSTATE: &[nodedb_types::error::ErrorCode] = &[ + nodedb_types::error::ErrorCode::DEADLINE_EXCEEDED, + nodedb_types::error::ErrorCode::MOVE_TENANT_DRAIN_TIMEOUT, +]; + +#[cfg(test)] +mod tests { + use nodedb_types::error::ErrorCode; + + use super::*; + use crate::control::server::shared::ddl::result::code_for_sqlstate; + + fn class(state: &str) -> &str { + state.get(..2).unwrap_or(state) + } + + fn is_internal(code: ErrorCode) -> bool { + INTERNAL_CODES.iter().any(|(internal, _)| *internal == code) + } + + /// Every public code renders a class of its own, unless it names a + /// server-side fault. + #[test] + fn every_client_class_code_has_a_sqlstate() { + for &code in ErrorCode::ALL { + let state = numeric_code_to_sqlstate(code); + if is_internal(code) { + assert_eq!(state, sqlstate::INTERNAL_ERROR, "{code} is internal"); + } else { + assert_ne!(state, sqlstate::INTERNAL_ERROR, "{code} has no SQLSTATE"); + } + } + } + + /// Native code to SQLSTATE and back returns the same code. A code whose + /// SQLSTATE another code owns returns the owner, which renders the same + /// class. + #[test] + fn every_client_class_code_round_trips_through_its_sqlstate() { + for &code in ErrorCode::ALL { + if is_internal(code) { + continue; + } + let state = numeric_code_to_sqlstate(code); + let back = code_for_sqlstate(state); + if NO_DEFAULT_SQLSTATE.contains(&code) { + assert_eq!( + back, + ErrorCode::INTERNAL, + "{code} renders {state}, which has no default" + ); + continue; + } + match SHARED_SQLSTATE.iter().find(|(shared, _)| *shared == code) { + Some((_, owner)) => { + assert_eq!(back, *owner, "{code} renders {state}, owned by {owner}"); + assert_eq!( + class(numeric_code_to_sqlstate(*owner)), + class(state), + "{code} and its owner {owner} render different classes" + ); + } + None => assert_eq!(back, code, "{code} renders {state}, read back as {back}"), + } + } + } + + /// Both lists name only codes that exist, and no code twice. + #[test] + fn the_allowlists_are_well_formed() { + for (code, reason) in INTERNAL_CODES { + assert!(ErrorCode::ALL.contains(code), "{code}"); + assert!(!reason.is_empty(), "{code} has no reason"); + } + for (code, owner) in SHARED_SQLSTATE { + assert!(!is_internal(*code), "{code} is both internal and shared"); + assert!( + !is_internal(*owner), + "{owner} owns a SQLSTATE but is internal" + ); + assert_ne!(code, owner, "{code} owns its own SQLSTATE"); + assert!( + !NO_DEFAULT_SQLSTATE.contains(code), + "{code} is both shared and no-default" + ); + } + for code in NO_DEFAULT_SQLSTATE { + assert!( + !is_internal(*code), + "{code} is both internal and no-default" + ); + } + } +} diff --git a/nodedb/src/control/server/reserved_socket.rs b/nodedb/src/control/server/reserved_socket.rs new file mode 100644 index 000000000..e11d061bb --- /dev/null +++ b/nodedb/src/control/server/reserved_socket.rs @@ -0,0 +1,112 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A TCP address bound early and opened for connections late. +//! +//! Boot binds every protocol address before it waits for the node to become +//! ready, so a port conflict fails boot while nothing is exposed. A bound +//! socket that is not listening refuses each connection attempt at once. A +//! listening socket completes the TCP handshake in the kernel before any +//! accept runs, so a client would wait out the whole boot. Boot therefore +//! calls [`ReservedSocket::listen`] only once the node can serve. + +use std::net::SocketAddr; + +use tokio::net::{TcpListener, TcpSocket}; + +/// Pending-connection queue length, the same value +/// `tokio::net::TcpListener::bind` uses. +const LISTEN_BACKLOG: u32 = 1024; + +/// A TCP socket bound to its address but not yet listening. +#[derive(Debug)] +pub struct ReservedSocket { + socket: TcpSocket, + addr: SocketAddr, +} + +impl ReservedSocket { + /// Bind `addr` without listening. + /// + /// The bind fails when another socket listens on `addr`. `SO_REUSEADDR` + /// is set on Unix, as `TcpListener::bind` does, so a restart can rebind + /// while old connections sit in `TIME_WAIT`. + pub fn bind(addr: SocketAddr) -> crate::Result { + let bind_error = |e: std::io::Error| crate::Error::Config { + detail: format!("bind {addr}: {e}"), + }; + let socket = if addr.is_ipv4() { + TcpSocket::new_v4() + } else { + TcpSocket::new_v6() + } + .map_err(bind_error)?; + #[cfg(not(windows))] + socket.set_reuseaddr(true).map_err(bind_error)?; + socket.bind(addr).map_err(bind_error)?; + let addr = socket.local_addr().map_err(bind_error)?; + Ok(Self { socket, addr }) + } + + /// The bound address. For port `0` it holds the port the OS assigned. + pub fn local_addr(&self) -> SocketAddr { + self.addr + } + + /// Start listening. Connection attempts are accepted from here on. + /// + /// Fails when another socket started listening on the same address after + /// this one was bound. + pub fn listen(self) -> crate::Result { + let addr = self.addr; + self.socket + .listen(LISTEN_BACKLOG) + .map_err(|e| crate::Error::Config { + detail: format!("listen on {addr}: {e}"), + }) + } +} + +#[cfg(test)] +mod tests { + use std::time::Duration; + + use super::*; + + fn loopback_any_port() -> SocketAddr { + SocketAddr::from(([127, 0, 0, 1], 0)) + } + + /// A reserved socket refuses connections at once, so a client connecting + /// during boot never waits in the kernel's accept queue. + #[tokio::test] + async fn a_reserved_socket_refuses_connections_until_it_listens() { + let reserved = ReservedSocket::bind(loopback_any_port()).expect("bind"); + let addr = reserved.local_addr(); + assert_ne!(addr.port(), 0); + + let refused = + tokio::time::timeout(Duration::from_secs(5), tokio::net::TcpStream::connect(addr)) + .await + .expect("a refusal is immediate"); + assert!( + refused.is_err(), + "a reserved socket must refuse connections" + ); + + let listener = reserved.listen().expect("listen"); + let (connected, accepted) = + tokio::join!(tokio::net::TcpStream::connect(addr), listener.accept()); + connected.expect("connect after listen"); + accepted.expect("accept after listen"); + } + + /// An address another socket listens on cannot be reserved. + #[tokio::test] + async fn an_address_in_use_cannot_be_reserved() { + let occupied = TcpListener::bind(loopback_any_port()) + .await + .expect("occupy a port"); + let addr = occupied.local_addr().expect("occupied addr"); + assert!(ReservedSocket::bind(addr).is_err()); + } +} diff --git a/nodedb/src/control/server/resp/gateway_dispatch.rs b/nodedb/src/control/server/resp/gateway_dispatch.rs index cc33c82a5..bea9b39b5 100644 --- a/nodedb/src/control/server/resp/gateway_dispatch.rs +++ b/nodedb/src/control/server/resp/gateway_dispatch.rs @@ -2,7 +2,7 @@ //! RESP gateway dispatch helpers. //! -//! Routes KV operations through `Gateway::execute` when the gateway is +//! Routes KV operations through `Gateway::execute_response` when the gateway is //! available (cluster-aware routing), falling back to direct local SPSC //! dispatch on single-node boot. //! @@ -11,7 +11,7 @@ use std::sync::Arc; -use crate::bridge::envelope::{Payload, PhysicalPlan, Response, Status}; +use crate::bridge::envelope::{PhysicalPlan, Response}; use crate::control::gateway::GatewayErrorMap; use crate::control::gateway::core::QueryContext; use crate::control::security::identity::AuthenticatedIdentity; @@ -21,7 +21,7 @@ use crate::control::server::shared::clone_write::CloneCheckedOutcome; use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; use crate::control::server::shared::quota_admission::admit_quota_for_dispatch; use crate::control::state::SharedState; -use crate::types::{DatabaseId, Lsn, RequestId, TraceId, VShardId}; +use crate::types::{DatabaseId, TraceId, VShardId}; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; use super::session::RespSession; @@ -44,7 +44,7 @@ pub(super) async fn dispatch_kv( // `RequestAuthScope::builder` so the dispatched task and `$auth.database_id` // resolve from the same value and cannot drift apart. let database_id = DatabaseId::DEFAULT; - let vshard = VShardId::from_collection_in_database(database_id, &session.collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &session.collection).vshard(); // Extracted before `plan` is moved into `authorize_resp_task`, which // consumes it for RLS injection and task construction — metering needs // the collection/engine shape after dispatch succeeds below, and by then @@ -74,12 +74,11 @@ pub(super) async fn dispatch_kv( database_id: checked.database_id(), txn_id: None, }; - gw.execute(&gw_ctx, checked) + gw.execute_response(&gw_ctx, checked) .await .map_err(|e| crate::Error::Bridge { detail: GatewayErrorMap::to_resp(&e), }) - .map(gateway_payloads_to_response) } None => dispatch_utils::dispatch_authorized_to_data_plane(state, checked, TraceId::ZERO) .await @@ -109,7 +108,7 @@ pub(super) async fn dispatch_kv_write( // DatabaseId::DEFAULT is deliberate here, resolved once and threaded // through `authorize_resp_task` via `RequestAuthScope::builder`. let database_id = DatabaseId::DEFAULT; - let vshard = VShardId::from_collection_in_database(database_id, &session.collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &session.collection).vshard(); // See `dispatch_kv` above: extracted before `authorize_resp_task` moves // `plan`, since metering needs the plan shape after dispatch succeeds. let plan_metering_info = state @@ -136,14 +135,13 @@ pub(super) async fn dispatch_kv_write( database_id: checked.database_id(), txn_id: None, }; - gw.execute(&gw_ctx, checked) + gw.execute_response(&gw_ctx, checked) .await .map_err(|e| crate::Error::Bridge { detail: GatewayErrorMap::to_resp(&e), }) - .map(gateway_payloads_to_response) } - None => dispatch_utils::dispatch_authorized_autocommit_write(state, checked, TraceId::ZERO) + None => dispatch_utils::dispatch_authorized_durable_write(state, checked, TraceId::ZERO) .await .map_err(map_busy_error), }; @@ -341,31 +339,6 @@ fn resp_auth_scope<'a, 'p>( ClientRequestScope::for_database(identity, stores, database_id, peer_addr) } -/// Convert gateway `Vec>` payloads into a synthetic `Response`. -/// -/// The RESP sub-handlers inspect `resp.status` and `resp.payload`; we -/// synthesise a `Status::Ok` response carrying the first payload so that all -/// existing sub-handler logic continues to work without modification. -fn gateway_payloads_to_response(payloads: Vec>) -> Response { - let payload = payloads - .into_iter() - .next() - .map(Payload::from_vec) - .unwrap_or_else(Payload::empty); - Response { - request_id: RequestId::new(0), - status: Status::Ok, - attempt: 0, - partial: false, - payload, - watermark_lsn: Lsn::new(0), - error_code: None, - read_set_valid: None, - read_version_lsn: crate::types::Lsn::ZERO, - write_set: Vec::new(), - } -} - /// Map bridge/dispatch errors to a BUSY error for Redis client compatibility. /// /// When the SPSC ring buffer is full or the Data Plane core is overloaded, @@ -373,7 +346,9 @@ fn gateway_payloads_to_response(payloads: Vec>) -> Response { /// which Redis clients handle with automatic retry (same as Redis Cluster BUSY). fn map_busy_error(e: crate::Error) -> crate::Error { match &e { - crate::Error::Bridge { .. } | crate::Error::Dispatch { .. } => crate::Error::Bridge { + crate::Error::Bridge { .. } + | crate::Error::Dispatch { .. } + | crate::Error::DispatchCapacity { .. } => crate::Error::Bridge { detail: "BUSY NodeDB is processing requests, retry later".into(), }, _ => e, @@ -382,11 +357,12 @@ fn map_busy_error(e: crate::Error) -> crate::Error { #[cfg(test)] mod tests { + use crate::bridge::envelope::{Payload, Status}; use crate::control::security::identity::{AuthMethod, DatabaseSet, Role}; use crate::control::security::metering::quota::QuotaManager; use crate::control::security::request_scope::AuthStores; use crate::control::security::scope::grant::ScopeGrantStore; - use crate::types::TenantId; + use crate::types::{Lsn, TenantId}; use super::*; @@ -478,7 +454,8 @@ mod tests { surrogate_ceiling: None, }); let vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, &session.collection); + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &session.collection) + .vshard(); let result = authorize_resp_task( &state, diff --git a/nodedb/src/control/server/resp/handler_hash.rs b/nodedb/src/control/server/resp/handler_hash.rs index e33d19ea7..0c4dce8ca 100644 --- a/nodedb/src/control/server/resp/handler_hash.rs +++ b/nodedb/src/control/server/resp/handler_hash.rs @@ -130,9 +130,11 @@ pub(super) async fn handle_hset( // Content-addressed cross-engine identity so the merged row keeps the // surrogate its original insert assigned. let surrogate = match state.surrogate_assigner.assign( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare( + nodedb_types::DatabaseId::DEFAULT, + &session.collection, + ), session.tenant_id, - &session.collection, &key, ) { Ok(s) => s, diff --git a/nodedb/src/control/server/resp/handler_kv/counters.rs b/nodedb/src/control/server/resp/handler_kv/counters.rs index 45444f7c9..d4712460b 100644 --- a/nodedb/src/control/server/resp/handler_kv/counters.rs +++ b/nodedb/src/control/server/resp/handler_kv/counters.rs @@ -4,7 +4,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::state::SharedState; -use nodedb_physical::physical_plan::KvOp; +use nodedb_physical::physical_plan::{KvCounterShape, KvOp}; use nodedb_types::{DatabaseId, QualifiedCollection}; use super::super::codec::RespValue; @@ -105,9 +105,12 @@ async fn dispatch_incr( surrogate, // Filled by the RLS injection pass `dispatch_kv_write` runs. rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // RESP treats every value as a byte string, as `SET` and `GET` do, so + // an absent key starts as decimal text in any collection. + shape: KvCounterShape::Raw, }); - match dispatch_kv_write(state, session, plan).await { + match dispatch_counter(state, session, plan).await { Ok(resp) => match payload_field_i64(&resp.payload, "value") { Some(new_val) => RespValue::integer(new_val), // The counter did change; a response we cannot read means we do @@ -130,13 +133,11 @@ pub(in crate::control::server::resp) async fn handle_incrbyfloat( } let key = cmd.args[0].clone(); - let delta_str = match cmd.arg_str(1) { - Some(s) => s, - None => return RespValue::err("ERR value is not a valid float"), - }; - let delta: f64 = match delta_str.parse() { - Ok(v) => v, - Err(_) => return RespValue::err("ERR value is not a valid float"), + // The delta stays the client's decimal text, so the engine adds every + // digit the client sent. + let delta = match cmd.arg_str(1) { + Some(s) if nodedb_physical::kv_atomic::float_text::is_decimal_number(s) => s.to_string(), + _ => return RespValue::err("ERR value is not a valid float"), }; if let Some(refusal) = refuse_if_counter_is_redacted(state, session) { @@ -154,16 +155,40 @@ pub(in crate::control::server::resp) async fn handle_incrbyfloat( surrogate, // Filled by the RLS injection pass `dispatch_kv_write` runs. rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // See `dispatch_incr`. + shape: KvCounterShape::Raw, }); - match dispatch_kv_write(state, session, plan).await { + match dispatch_counter(state, session, plan).await { + // The reply is a bulk string. A raw body answers with the exact text + // it stored, the bytes `GET` returns. A typed row answers with the + // column's new number. Ok(resp) => { - // Return the new value as a bulk string (Redis convention). - match payload_json(&resp.payload).get("value") { - Some(v) => RespValue::bulk(v.to_string().into_bytes()), + let reply = payload_json(&resp.payload); + let text = match reply.get("text").and_then(serde_json::Value::as_str) { + Some(text) => Some(text.to_string()), + None => reply + .get("value") + .and_then(serde_json::Value::as_f64) + .map(|v| v.to_string()), + }; + match text { + Some(text) => RespValue::bulk(text.into_bytes()), None => RespValue::err("ERR counter response could not be decoded"), } } Err(e) => RespValue::from_error(&e), } } + +/// Dispatch a counter write and turn a Data-Plane error status into a typed +/// error, so a counter fault reaches the client as its own message. +async fn dispatch_counter( + state: &SharedState, + session: &RespSession, + plan: PhysicalPlan, +) -> crate::Result { + let resp = dispatch_kv_write(state, session, plan).await?; + crate::control::local_dispatch::reject_data_plane_error(&resp)?; + Ok(resp) +} diff --git a/nodedb/src/control/server/resp/handler_kv/strings.rs b/nodedb/src/control/server/resp/handler_kv/strings.rs index bfa7aa09b..44f9290d9 100644 --- a/nodedb/src/control/server/resp/handler_kv/strings.rs +++ b/nodedb/src/control/server/resp/handler_kv/strings.rs @@ -140,6 +140,7 @@ pub(in crate::control::server::resp) async fn handle_set( surrogate, returning: None, rls_filters: Vec::new(), + provenance: None, }); // A rejected write surfaces as the error it is, never as `OK`. @@ -170,6 +171,7 @@ pub(in crate::control::server::resp) async fn handle_del( // RESP has no RETURNING clause. returning: None, rls_filters: Vec::new(), + provenance: None, }); // A rejected delete surfaces as the error it is, never as `0` deleted. diff --git a/nodedb/src/control/server/resp/handler_kv/surrogate.rs b/nodedb/src/control/server/resp/handler_kv/surrogate.rs index 34624523c..7fa6544d1 100644 --- a/nodedb/src/control/server/resp/handler_kv/surrogate.rs +++ b/nodedb/src/control/server/resp/handler_kv/surrogate.rs @@ -19,9 +19,11 @@ pub(super) fn resp_kv_surrogate( state .surrogate_assigner .assign( - crate::types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare( + crate::types::DatabaseId::DEFAULT, + &session.collection, + ), session.tenant_id, - &session.collection, key, ) .map_err(|e| RespValue::err(format!("ERR {e}"))) diff --git a/nodedb/src/control/server/resp/handler_sorted.rs b/nodedb/src/control/server/resp/handler_sorted.rs index db4f1d748..ce2141017 100644 --- a/nodedb/src/control/server/resp/handler_sorted.rs +++ b/nodedb/src/control/server/resp/handler_sorted.rs @@ -65,9 +65,8 @@ pub(super) async fn handle_zadd( .unwrap_or_default(); let surrogate = match state.surrogate_assigner.assign( - crate::types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(crate::types::DatabaseId::DEFAULT, &index_name), session.tenant_id, - &index_name, &member, ) { Ok(s) => s, @@ -81,6 +80,7 @@ pub(super) async fn handle_zadd( surrogate, returning: None, rls_filters: Vec::new(), + provenance: None, }); match dispatch_kv_write(state, session, plan).await { @@ -116,6 +116,7 @@ pub(super) async fn handle_zrem( // RESP has no RETURNING clause. returning: None, rls_filters: Vec::new(), + provenance: None, }); match dispatch_kv_write(state, session, plan).await { diff --git a/nodedb/src/control/server/resp/listener.rs b/nodedb/src/control/server/resp/listener.rs index 9bc3327f4..502681f94 100644 --- a/nodedb/src/control/server/resp/listener.rs +++ b/nodedb/src/control/server/resp/listener.rs @@ -39,6 +39,11 @@ impl RespListener { .map_err(|e| crate::Error::Config { detail: format!("failed to bind RESP listener on {addr}: {e}"), })?; + Self::from_listener(tcp) + } + + /// Serve on a socket that already listens. + pub fn from_listener(tcp: TcpListener) -> crate::Result { let local_addr = tcp.local_addr().map_err(|e| crate::Error::Config { detail: format!("failed to get RESP local address: {e}"), })?; diff --git a/nodedb/src/control/server/response_shape/calvin_fold.rs b/nodedb/src/control/server/response_shape/calvin_fold.rs index acbff10fc..9cbfd05de 100644 --- a/nodedb/src/control/server/response_shape/calvin_fold.rs +++ b/nodedb/src/control/server/response_shape/calvin_fold.rs @@ -243,6 +243,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); assert_eq!( calvin_tag_for_plan(&plan), @@ -264,6 +265,7 @@ mod tests { rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: None, rls_filters: Vec::new(), + provenance: None, }); assert!(calvin_tag_for_plan(&delete).is_none()); diff --git a/nodedb/src/control/server/response_shape/compose/materialized.rs b/nodedb/src/control/server/response_shape/compose/materialized.rs index ce00ae8fd..9f4c97ed4 100644 --- a/nodedb/src/control/server/response_shape/compose/materialized.rs +++ b/nodedb/src/control/server/response_shape/compose/materialized.rs @@ -12,8 +12,10 @@ //! encoder; each protocol then encodes those rows in its own wire format //! (pgwire's RowDescription/DataRow, native's MessagePack, http's JSON). //! -//! Producers with no `PhysicalPlan` in scope (ClusterArray, set-op merges, -//! gateway forwarding, clone merges) call [`shape_payload_no_plan`], which +//! Gateway forwarding shapes through this same call with the forwarded task's +//! plan, so a forwarded payload yields the rows a local one does. Producers +//! with no `PhysicalPlan` in scope (ClusterArray, set-op merges, clone merges) +//! call [`shape_payload_no_plan`], which //! skips the plan-dependent `apply_kv_wrap` / `translate_search_response` //! transforms those callers never ran. The pure kernel `shape_decoded_rows` //! is shared with per-batch lazy streaming callers, which have an @@ -102,7 +104,7 @@ pub fn shape_response_materialized( /// Shape a Data-Plane payload with no `PhysicalPlan` in scope. /// /// Producers that never had a plan to KV-wrap or vector-translate -/// (ClusterArray, set-op merges, gateway forwarding, clone merges) call this +/// (ClusterArray, set-op merges, clone merges) call this /// instead of [`shape_response_materialized`]: it applies only the decode + /// scan-envelope unwrap + optional SELECT-list projection steps, skipping the /// plan-dependent `apply_kv_wrap` / `translate_search_response` transforms those diff --git a/nodedb/src/control/server/response_shape/types/plan_kind/describe.rs b/nodedb/src/control/server/response_shape/types/plan_kind/describe.rs index 7c51c000e..cf9b6fad7 100644 --- a/nodedb/src/control/server/response_shape/types/plan_kind/describe.rs +++ b/nodedb/src/control/server/response_shape/types/plan_kind/describe.rs @@ -205,6 +205,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: spec(), rls_filters: Vec::new(), + provenance: None, }), PhysicalPlan::Kv(KvOp::BatchPut { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), @@ -242,6 +243,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }) } diff --git a/nodedb/src/control/server/response_shape/types/plan_kind/kv.rs b/nodedb/src/control/server/response_shape/types/plan_kind/kv.rs index 12a3baf03..ee4cedf51 100644 --- a/nodedb/src/control/server/response_shape/types/plan_kind/kv.rs +++ b/nodedb/src/control/server/response_shape/types/plan_kind/kv.rs @@ -2,7 +2,7 @@ //! `KvOp` classification. -use nodedb_physical::physical_plan::KvOp; +use nodedb_physical::physical_plan::{KvOp, SortedIndexRead}; use super::kind::PlanKind; @@ -72,6 +72,14 @@ pub(super) fn describe_kv(op: &KvOp) -> PlanKind { // One row per sorted-index entry. KvOp::SortedIndexTopK { .. } | KvOp::SortedIndexRange { .. } => PlanKind::MultiRow, + // Shaped like the autocommit read it stands for. + KvOp::SortedIndexTxnRead { read, .. } => match read { + SortedIndexRead::Rank { .. } + | SortedIndexRead::Count + | SortedIndexRead::Score { .. } => PlanKind::SingleDocument, + SortedIndexRead::TopK { .. } | SortedIndexRead::Range { .. } => PlanKind::MultiRow, + }, + // TTL metadata mutations: no row count. KvOp::Expire { .. } | KvOp::Persist { .. } diff --git a/nodedb/src/control/server/response_translate/vector.rs b/nodedb/src/control/server/response_translate/vector.rs index 10eac7e97..f12731050 100644 --- a/nodedb/src/control/server/response_translate/vector.rs +++ b/nodedb/src/control/server/response_translate/vector.rs @@ -88,7 +88,9 @@ fn apply_rls_filter(hits: &mut Vec, rls_filters: &[u8], top_k: usize) { /// user's PK through this single catalog call. Returns `None` when the /// catalog has no PK mapping for the surrogate (headless row, or a document /// that was never written) — callers must leave the row's identifier -/// untouched in that case rather than fabricate a value. +/// untouched in that case rather than fabricate a value. `collection` is the +/// plan's database-qualified name, de-qualified into the canonical key; a +/// name that does not de-qualify also resolves to `None`. pub(crate) fn resolve_surrogate_pk( state: &SharedState, database_id: DatabaseId, @@ -97,8 +99,9 @@ pub(crate) fn resolve_surrogate_pk( surrogate: Surrogate, ) -> Option { let catalog = state.credentials.catalog(); + let key = nodedb_types::CollectionKey::from_qualified_str(database_id, collection).ok()?; let pk_bytes = catalog - .get_pk_for_surrogate(database_id, tenant_id, collection, surrogate) + .get_pk_for_surrogate(key, tenant_id, surrogate) .ok()??; String::from_utf8(pk_bytes).ok() } diff --git a/nodedb/src/control/server/result_stream.rs b/nodedb/src/control/server/result_stream.rs index fc58612fc..1f94856b9 100644 --- a/nodedb/src/control/server/result_stream.rs +++ b/nodedb/src/control/server/result_stream.rs @@ -3,7 +3,7 @@ //! Durable streaming result abstraction. //! //! A dispatched scan returns its rows as a sequence of `Response` frames over a -//! `tokio::sync::mpsc::Receiver` (see `RequestTracker::register`): +//! `ResponseReceiver` (see `RequestTracker::register`): //! several `partial: true` frames followed by one terminal (`partial: false`) //! frame, each carrying a standalone msgpack-array payload of rows //! (`encode_raw_document_rows`). @@ -15,8 +15,8 @@ //! merged msgpack array for byte-demanding consumers that still need the //! fully-collected result. -use crate::bridge::envelope::{Response, Status}; -use crate::control::server::dispatch_utils::reject_data_plane_error; +use crate::bridge::envelope::Status; +use crate::control::local_dispatch::reject_data_plane_error; use crate::control::server::payload_merge::merge_msgpack_arrays; use crate::types::Lsn; @@ -51,7 +51,7 @@ pub type ResultStream = /// ends after the terminal (`!partial`) frame is yielded, or when the channel /// closes. pub(crate) fn stream_response_channel( - mut rx: tokio::sync::mpsc::Receiver, + mut rx: crate::control::ResponseReceiver, max_result_bytes: usize, tolerate_not_found: bool, ) -> ResultStream { @@ -78,10 +78,11 @@ pub(crate) fn stream_response_channel( Err(error) => error, // `NotFound` is the one code that conversion reads as an // empty observation rather than an error. This stream - // declined to tolerate it, so it stops here. - Ok(()) => crate::Error::Dispatch { - detail: "data plane error: NotFound".to_string(), - }, + // declined to tolerate it, so it stops here with the + // code's own class. + Ok(()) => crate::Error::DataPlane( + crate::bridge::envelope::ErrorCode::NotFound, + ), }; Err(error)?; return; @@ -134,7 +135,7 @@ pub(crate) async fn materialize(mut stream: ResultStream) -> crate::Result<(Vec< #[cfg(test)] mod tests { use super::*; - use crate::bridge::envelope::{ErrorCode, Payload}; + use crate::bridge::envelope::{ErrorCode, Payload, Response}; use crate::control::server::payload_merge::{encode_msgpack_array, extract_msgpack_elements}; use crate::types::RequestId; use tokio::sync::mpsc; @@ -213,7 +214,11 @@ mod tests { tx.send(partial(1000)).await.unwrap(); tx.send(final_frame(500)).await.unwrap(); drop(tx); - let stream = stream_response_channel(rx, 1 << 20, false); + let stream = stream_response_channel( + crate::control::ResponseReceiver::from_channel(rx), + 1 << 20, + false, + ); let (merged, _lsn) = materialize(stream).await.unwrap(); assert_eq!( extract_msgpack_elements(&merged).len(), @@ -228,7 +233,11 @@ mod tests { tx.send(raw_partial(600)).await.unwrap(); tx.send(raw_partial(600)).await.unwrap(); drop(tx); - let stream = stream_response_channel(rx, 1000, false); + let stream = stream_response_channel( + crate::control::ResponseReceiver::from_channel(rx), + 1000, + false, + ); let err = materialize(stream).await.unwrap_err(); assert!(matches!(err, crate::Error::ExecutionLimitExceeded { .. })); } @@ -244,7 +253,11 @@ mod tests { .await .unwrap(); drop(tx); - let stream = stream_response_channel(rx, 1 << 20, false); + let stream = stream_response_channel( + crate::control::ResponseReceiver::from_channel(rx), + 1 << 20, + false, + ); match materialize(stream).await { Err(crate::Error::DataPlane(ErrorCode::ResourcesExhausted)) => {} other => panic!("expected the shard's own code, got {other:?}"), @@ -262,7 +275,11 @@ mod tests { .await .unwrap(); drop(tx); - let stream = stream_response_channel(rx, 1 << 20, false); + let stream = stream_response_channel( + crate::control::ResponseReceiver::from_channel(rx), + 1 << 20, + false, + ); match materialize(stream).await { Err(crate::Error::DeadlineExceeded { request_id }) => { assert_eq!(request_id, RequestId::new(1)); @@ -272,14 +289,21 @@ mod tests { } /// An untolerated `NotFound` still stops the stream rather than reading as - /// an empty success. + /// an empty success, and keeps its typed code. #[tokio::test] async fn untolerated_not_found_errors() { let (tx, rx) = mpsc::channel(8); tx.send(error_frame(ErrorCode::NotFound)).await.unwrap(); drop(tx); - let stream = stream_response_channel(rx, 1 << 20, false); - assert!(materialize(stream).await.is_err()); + let stream = stream_response_channel( + crate::control::ResponseReceiver::from_channel(rx), + 1 << 20, + false, + ); + match materialize(stream).await { + Err(crate::Error::DataPlane(ErrorCode::NotFound)) => {} + other => panic!("expected the typed NotFound refusal, got {other:?}"), + } } #[tokio::test] @@ -287,7 +311,11 @@ mod tests { let (tx, rx) = mpsc::channel(8); tx.send(error_frame(ErrorCode::NotFound)).await.unwrap(); drop(tx); - let stream = stream_response_channel(rx, 1 << 20, true); + let stream = stream_response_channel( + crate::control::ResponseReceiver::from_channel(rx), + 1 << 20, + true, + ); let (merged, _lsn) = materialize(stream).await.unwrap(); assert_eq!( extract_msgpack_elements(&merged).len(), diff --git a/nodedb/src/control/server/session_auth/bearer_jwt.rs b/nodedb/src/control/server/session_auth/bearer_jwt.rs index d396ff695..afef1e7b9 100644 --- a/nodedb/src/control/server/session_auth/bearer_jwt.rs +++ b/nodedb/src/control/server/session_auth/bearer_jwt.rs @@ -44,7 +44,7 @@ pub async fn authenticate_bearer_jwt( if let Err(error) = crate::control::security::jwt_policy::enforce_stateful_jwt_policy( state, verified.claims(), - identity.tenant_id, + &identity, ) { debug!(%error, "bearer token refused by auth.jwt policy"); return None; diff --git a/nodedb/src/control/server/shared/authorization/requirements/collect.rs b/nodedb/src/control/server/shared/authorization/requirements/collect.rs index 1ef47efbe..656e3001c 100644 --- a/nodedb/src/control/server/shared/authorization/requirements/collect.rs +++ b/nodedb/src/control/server/shared/authorization/requirements/collect.rs @@ -224,6 +224,7 @@ fn collect_requirements(plan: &PhysicalPlan, out: &mut Vec Vec<&str> { | KvOp::SortedIndexRange { .. } | KvOp::SortedIndexCount { .. } | KvOp::SortedIndexScore { .. } + | KvOp::SortedIndexTxnRead { .. } | KvOp::PredicateUpdate { .. } | KvOp::PredicateDelete { .. } | KvOp::MaterializeScan { .. }) => other.collection().into_iter().collect(), diff --git a/nodedb/src/control/server/shared/clone_read/dispatch.rs b/nodedb/src/control/server/shared/clone_read/dispatch.rs index a1582bd8c..240c4291d 100644 --- a/nodedb/src/control/server/shared/clone_read/dispatch.rs +++ b/nodedb/src/control/server/shared/clone_read/dispatch.rs @@ -8,11 +8,12 @@ //! filtered out. use crate::bridge::envelope::Response; +use crate::control::local_dispatch::reject_data_plane_error; use crate::control::security::audit::AuditEmitter; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::security::permission::PermissionStore; use crate::control::security::role::RoleStore; -use crate::control::server::dispatch_utils::{dispatch_to_data_plane, reject_data_plane_error}; +use crate::control::server::dispatch_utils::dispatch_to_data_plane; use crate::control::server::response_shape::kv::apply_kv_wrap; use crate::control::server::response_shape::types::{PlanKind, describe_plan}; use crate::control::server::shared::authorization::authorize_task_set; diff --git a/nodedb/src/control/server/shared/clone_write/document.rs b/nodedb/src/control/server/shared/clone_write/document.rs index 095cc4b6c..3469da3f1 100644 --- a/nodedb/src/control/server/shared/clone_write/document.rs +++ b/nodedb/src/control/server/shared/clone_write/document.rs @@ -121,9 +121,8 @@ fn source_surrogate( state .surrogate_assigner .lookup( - source_db_id, + nodedb_types::CollectionKey::from_qualified_str(source_db_id, source_coll_qualified)?, tenant_id, - source_coll_qualified, document_id.as_bytes(), ) .map_err(|e| write_err(format!("clone write source surrogate lookup: {e}"))) diff --git a/nodedb/src/control/server/shared/clone_write/kv.rs b/nodedb/src/control/server/shared/clone_write/kv.rs index d84093828..1fefc79da 100644 --- a/nodedb/src/control/server/shared/clone_write/kv.rs +++ b/nodedb/src/control/server/shared/clone_write/kv.rs @@ -13,7 +13,6 @@ use crate::control::security::identity::{AuthenticatedIdentity, Permission}; use crate::control::server::shared::authorization::authorize_collection; use crate::control::server::shared::sql::staging_predicates::require_affected_count; use crate::control::state::SharedState; -use crate::types::VShardId; use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; use nodedb_physical::physical_task::PhysicalTask; @@ -38,6 +37,10 @@ pub(super) async fn intercept_kv_clone_write( rls_write_check, returning, rls_filters, + // The tombstone path answers a count, not a sync ack. A Lite KV + // push that lands here moves its stream mark on its own, after + // this returns `Handled`. + provenance: _, }) => { // Delete may have multiple keys; handle each. We serialize here // (one tombstone per key) and return Handled with synthetic OK. @@ -167,8 +170,11 @@ pub(super) async fn intercept_kv_clone_write( // Same statement, same projection and read gate. returning: returning.clone(), rls_filters: rls_filters.clone(), + provenance: None, }); - let vshard_id = VShardId::from_collection_in_database(db_id, collection_qualified); + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(db_id, collection_qualified)? + .vshard(); let resp = dispatch_data_plane_raw(state, tenant_id, vshard_id, db_id, delete_plan) .await .map_err(|e| write_err(format!("clone kv delete dispatch: {e}")))?; diff --git a/nodedb/src/control/server/shared/clone_write/probes.rs b/nodedb/src/control/server/shared/clone_write/probes.rs index 6f7d6385b..071feb778 100644 --- a/nodedb/src/control/server/shared/clone_write/probes.rs +++ b/nodedb/src/control/server/shared/clone_write/probes.rs @@ -38,7 +38,8 @@ pub(super) async fn probe_row_in_target( valid_at_ms: None, }); let plan = with_caller_rls(state, identity, tenant_id, db_id, plan)?; - let vshard_id = VShardId::from_collection_in_database(db_id, collection_qualified); + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(db_id, collection_qualified)?.vshard(); let resp = dispatch_data_plane_raw(state, tenant_id, vshard_id, db_id, plan).await?; Ok(!resp.payload.is_empty() && resp.status == Status::Ok) } @@ -65,7 +66,9 @@ pub(super) async fn fetch_source_row( valid_at_ms: None, }); let plan = with_caller_rls(state, identity, tenant_id, source_db_id, plan)?; - let vshard_id = VShardId::from_collection_in_database(source_db_id, source_coll_qualified); + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(source_db_id, source_coll_qualified)? + .vshard(); let resp = dispatch_data_plane_raw(state, tenant_id, vshard_id, source_db_id, plan).await?; if resp.payload.is_empty() || resp.status != Status::Ok { return Ok(None); @@ -94,7 +97,8 @@ pub(super) async fn probe_kv_key_in_target( surrogate_ceiling: None, }); let plan = with_caller_rls(state, identity, tenant_id, db_id, plan)?; - let vshard_id = VShardId::from_collection_in_database(db_id, collection_qualified); + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(db_id, collection_qualified)?.vshard(); let resp = dispatch_data_plane_raw(state, tenant_id, vshard_id, db_id, plan).await?; Ok(!resp.payload.is_empty() && resp.status == Status::Ok) } @@ -120,7 +124,9 @@ pub(super) async fn fetch_kv_source_value( surrogate_ceiling: None, }); let plan = with_caller_rls(state, identity, tenant_id, source_db_id, plan)?; - let vshard_id = VShardId::from_collection_in_database(source_db_id, source_coll_qualified); + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(source_db_id, source_coll_qualified)? + .vshard(); let resp = dispatch_data_plane_raw(state, tenant_id, vshard_id, source_db_id, plan).await?; if resp.payload.is_empty() || resp.status != Status::Ok { return Ok(None); diff --git a/nodedb/src/control/server/shared/ddl/catalog.rs b/nodedb/src/control/server/shared/ddl/catalog.rs index 1d41aaf75..2eecf15bb 100644 --- a/nodedb/src/control/server/shared/ddl/catalog.rs +++ b/nodedb/src/control/server/shared/ddl/catalog.rs @@ -34,7 +34,7 @@ pub fn propose_and_apply( entry: &CatalogEntry, ) -> Result { let outcome = propose_catalog_entry(state, entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; apply_locally_if_needed(state, entry, outcome); Ok(outcome) } diff --git a/nodedb/src/control/server/shared/ddl/engine_apply.rs b/nodedb/src/control/server/shared/ddl/engine_apply.rs index 3c39f3752..6aaef0f89 100644 --- a/nodedb/src/control/server/shared/ddl/engine_apply.rs +++ b/nodedb/src/control/server/shared/ddl/engine_apply.rs @@ -9,7 +9,7 @@ use std::time::Duration; -use crate::bridge::envelope::PhysicalPlan; +use crate::bridge::envelope::{ErrorCode, PhysicalPlan}; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId}; @@ -17,14 +17,16 @@ use super::result::DdlError; use super::sync_dispatch::{SystemReason, SystemTask, dispatch_system}; /// Dispatch `plan` for `collection` and translate any Data-Plane refusal into -/// a [`DdlError`] carrying `sqlstate` and `context`. +/// a [`DdlError`] with `context` before its message. +/// +/// A typed refusal keeps its own SQLSTATE and code. A refusal with no code is +/// an internal error. pub(crate) async fn apply_in_engine( state: &SharedState, tenant_id: TenantId, database_id: DatabaseId, collection: &str, plan: PhysicalPlan, - sqlstate: &str, context: &str, ) -> Result<(), DdlError> { let timeout = Duration::from_secs(state.tuning.network.default_deadline_secs); @@ -33,13 +35,68 @@ pub(crate) async fn apply_in_engine( SystemTask::new( SystemReason::DdlApply, tenant_id, - database_id, - collection, + nodedb_types::CollectionKey::from_bare(database_id, collection), plan, ), timeout, ) .await .map(|_| ()) - .map_err(|e| DdlError::new(sqlstate, format!("{context}: {e}"))) + .map_err(|e| DdlError::from_error_in_context(context, &e)) +} + +/// Refuse a vector index definition the engine would refuse, without +/// changing engine state. +/// +/// `VectorOp::SetParams` refuses a core whose index already materialized with +/// `ErrorCode::Unsupported`. Inside an explicit transaction the parameters +/// install at COMMIT, so the statement probes with the read-only +/// `VectorOp::QueryStats` instead: an index that answers has materialized, and +/// `NotFound` means the parameters will install. A materialized index gets the +/// same `Unsupported` verdict the engine gives, so both paths answer one +/// SQLSTATE. +pub(crate) async fn refuse_materialized_vector_index( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection: &str, + field_name: &str, + context: &str, +) -> Result<(), DdlError> { + let timeout = Duration::from_secs(state.tuning.network.default_deadline_secs); + let plan = PhysicalPlan::Vector(nodedb_physical::physical_plan::VectorOp::QueryStats { + collection: nodedb_types::QualifiedCollection::new(database_id, collection), + field_name: field_name.to_string(), + }); + let response = super::sync_dispatch::dispatch_system_response_with_source( + state, + SystemTask::new( + SystemReason::DdlApply, + tenant_id, + nodedb_types::CollectionKey::from_bare(database_id, collection), + plan, + ), + timeout, + crate::event::EventSource::User, + ) + .await + .map_err(|e| DdlError::from_error_in_context(context, &e))?; + match (response.status, response.error_code.as_deref()) { + (crate::bridge::envelope::Status::Ok, _) => Err(DdlError::from_error_in_context( + context, + &crate::Error::DataPlane(ErrorCode::Unsupported { + detail: "changing vector index params after the index holds vectors is not \ + supported; drop and recreate the collection" + .into(), + }), + )), + (_, Some(ErrorCode::NotFound)) => Ok(()), + (_, Some(code)) => Err(DdlError::from_error_in_context( + &format!("{context}: vector index probe"), + &crate::Error::DataPlane(code.clone()), + )), + (_, None) => Err(DdlError::internal(format!( + "{context}: vector index probe failed with no error code" + ))), + } } diff --git a/nodedb/src/control/server/shared/ddl/index_registry.rs b/nodedb/src/control/server/shared/ddl/index_registry.rs index 7cbdb9c6c..473581153 100644 --- a/nodedb/src/control/server/shared/ddl/index_registry.rs +++ b/nodedb/src/control/server/shared/ddl/index_registry.rs @@ -15,10 +15,6 @@ use crate::types::{DatabaseId, TenantId}; use super::result::DdlError; -fn registry_err(message: String) -> DdlError { - DdlError::new("XX000", message) -} - /// The identity of one index, as its creating statement declared it. pub struct IndexRegistration<'a> { pub database_id: DatabaseId, @@ -45,13 +41,13 @@ pub fn propose_index_record( }; let entry = CatalogEntry::PutIndexRecord(Box::new(record.clone())); let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| registry_err(format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { state .credentials .catalog() .put_index_record(&record) - .map_err(|e| registry_err(format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } Ok(()) } @@ -71,13 +67,13 @@ pub fn propose_delete_index_record( collection: collection.to_string(), }; let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| registry_err(format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { state .credentials .catalog() .delete_index_record(database_id.as_u64(), tenant_id.as_u64(), name) - .map_err(|e| registry_err(format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } Ok(()) } diff --git a/nodedb/src/control/server/shared/ddl/mod.rs b/nodedb/src/control/server/shared/ddl/mod.rs index 6b1d4b07e..290949506 100644 --- a/nodedb/src/control/server/shared/ddl/mod.rs +++ b/nodedb/src/control/server/shared/ddl/mod.rs @@ -11,6 +11,7 @@ pub mod result; pub mod schema_validation; pub mod sql_parse; pub mod sqlstate; +pub mod static_sqlstate; pub mod sync_dispatch; pub mod user_dispatch; diff --git a/nodedb/src/control/server/shared/ddl/neutral/alert/create.rs b/nodedb/src/control/server/shared/ddl/neutral/alert/create.rs index 04df9df7f..8cb1fe73b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/alert/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/alert/create.rs @@ -104,7 +104,7 @@ pub fn create_alert( let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| err("XX000", "system clock error".to_string()))? + .map_err(|_| DdlError::internal("system clock error"))? .as_secs(); let def = AlertDef { diff --git a/nodedb/src/control/server/shared/ddl/neutral/alert/replicate.rs b/nodedb/src/control/server/shared/ddl/neutral/alert/replicate.rs index a89a46b90..ddac16443 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/alert/replicate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/alert/replicate.rs @@ -14,10 +14,6 @@ use crate::event::alert::types::AlertDef; use super::super::super::result::DdlError; use super::super::replicate::propose_and_apply; -fn err(sqlstate: &str, message: String) -> DdlError { - DdlError::new(sqlstate, message) -} - /// Propose the full alert record. CREATE and ALTER both re-put the row. /// /// The leader validates before proposing, so apply never rejects. @@ -28,7 +24,7 @@ pub(super) fn propose_put(state: &SharedState, def: &AlertDef) -> Result<(), Ddl .credentials .catalog() .put_alert_rule(def) - .map_err(|e| err("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; post_apply::put(def, state); Ok(()) }) @@ -52,7 +48,7 @@ pub(super) fn propose_delete( .credentials .catalog() .delete_alert_rule(database_id, tenant_id, name) - .map_err(|e| err("XX000", format!("catalog delete: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; post_apply::delete(database_id, tenant_id, name, state); Ok(()) }) diff --git a/nodedb/src/control/server/shared/ddl/neutral/apikey/create.rs b/nodedb/src/control/server/shared/ddl/neutral/apikey/create.rs index ba471fc00..d0dbaa05f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/apikey/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/apikey/create.rs @@ -54,10 +54,9 @@ pub fn create_api_key( require_tenant_admin(identity, "create API keys for other users")?; } - // Look up the target user. - let target_user = state - .credentials - .get_user(target_username) + // Look up the target user as this statement sees it: a user created + // earlier in the transaction counts, one dropped in it does not. + let target_user = super::super::role_checks::visible_user(state, target_username) .ok_or_else(|| err("42704", format!("user '{target_username}' not found")))?; // Parse optional EXPIRES. @@ -119,19 +118,19 @@ pub fn create_api_key( .prepare_key(crate::control::security::apikey::CreateKeyParams { username: target_username, user_id: target_user.user_id, - tenant_id: target_user.tenant_id, + tenant_id: crate::types::TenantId::new(target_user.tenant_id), expires_secs, scope: key_scopes, accessible_databases, }); let entry = crate::control::catalog_entry::CatalogEntry::PutApiKey(Box::new(stored.clone())); let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| err("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { let catalog = state.credentials.catalog(); catalog .put_api_key(&stored) - .map_err(|e| err("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; state.api_keys.install_replicated_key(&stored); } diff --git a/nodedb/src/control/server/shared/ddl/neutral/apikey/manage.rs b/nodedb/src/control/server/shared/ddl/neutral/apikey/manage.rs index 3069870d2..886e55c1d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/apikey/manage.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/apikey/manage.rs @@ -54,13 +54,13 @@ pub fn revoke_api_key( key_id: key_id.to_string(), }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| err("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; let revoked = if outcome.needs_local_apply() { let catalog = state.credentials.catalog(); state .api_keys .revoke_key(key_id, Some(catalog)) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? } else { // Cluster mode: trust the committed log index — the // in-memory cache update runs in a spawned tokio task and diff --git a/nodedb/src/control/server/shared/ddl/neutral/apikey/parse.rs b/nodedb/src/control/server/shared/ddl/neutral/apikey/parse.rs index cc1669afd..0d8174447 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/apikey/parse.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/apikey/parse.rs @@ -108,7 +108,7 @@ pub(super) fn parse_with_databases( for name in raw_names { let resolved: Option = catalog .get_database_id_by_name(name) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; match resolved { Some(id) => ids.push(id), None => { @@ -120,17 +120,20 @@ pub(super) fn parse_with_databases( Ok(Some(ids)) } -/// Build the owner's `DatabaseSet` from a `UserRecord` for CREATE-time subset validation. +/// Build the owner's `DatabaseSet` for CREATE-time subset validation, from +/// the user as the statement sees it. pub(super) fn build_owner_database_set_for_user( state: &SharedState, - user: &crate::control::security::credential::record::UserRecord, + user: &crate::control::security::catalog::auth_types::user::StoredUser, ) -> Result { if user.is_superuser { return Ok(DatabaseSet::All); } if user.is_service_account && !user.accessible_databases.is_empty() { return Ok(DatabaseSet::Some(SmallVec::from_iter( - user.accessible_databases.iter().copied(), + user.accessible_databases + .iter() + .map(|&id| crate::types::DatabaseId::new(id)), ))); } // Regular user or legacy service account: read from database_grants. @@ -138,6 +141,6 @@ pub(super) fn build_owner_database_set_for_user( .credentials .catalog() .list_user_grant_databases(user.user_id) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; Ok(DatabaseSet::Some(SmallVec::from_iter(db_ids))) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/auth_user.rs b/nodedb/src/control/server/shared/ddl/neutral/auth_user.rs index 66e84db8f..525f2f81b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/auth_user.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/auth_user.rs @@ -75,7 +75,7 @@ fn deactivate_auth_user( let found = state .auth_users .deactivate(user_id) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if !found { return Err(err("42704", format!("auth user '{user_id}' not found"))); @@ -112,7 +112,7 @@ fn alter_auth_user_status( let found = state .auth_users .set_status(user_id, status_val) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if !found { return Err(err("42704", format!("auth user '{user_id}' not found"))); @@ -165,7 +165,7 @@ pub fn purge_auth_users( let purged = state .auth_users .purge_inactive(cutoff) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/blacklist.rs b/nodedb/src/control/server/shared/ddl/neutral/blacklist.rs index 5ce7301da..770ea9f62 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/blacklist.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/blacklist.rs @@ -93,7 +93,7 @@ fn handle_blacklist_user( state .blacklist .blacklist_user(user_id, &reason, &identity.username, expires_at) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; // WITH KILL SESSIONS — terminate active sessions immediately. let kill_sessions = parts.iter().any(|p| p.to_uppercase() == "KILL"); @@ -140,7 +140,7 @@ fn handle_blacklist_ip( state .blacklist .blacklist_ip(addr, &reason, &identity.username, expires_at) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, @@ -221,7 +221,7 @@ fn lift( &crate::control::security::blacklist::store::BlacklistStore, ) -> crate::Result, ) -> Result, DdlError> { - let removed = remove(&state.blacklist).map_err(|e| err("XX000", e.to_string()))?; + let removed = remove(&state.blacklist).map_err(|e| DdlError::from_error(&e))?; if !removed { return Err(err( "42704", diff --git a/nodedb/src/control/server/shared/ddl/neutral/change_stream/create.rs b/nodedb/src/control/server/shared/ddl/neutral/change_stream/create.rs index 8c9eb134c..a72684551 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/change_stream/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/change_stream/create.rs @@ -130,7 +130,7 @@ pub fn create_change_stream( let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| DdlError::new("XX000", "system clock before UNIX epoch"))? + .map_err(|_| DdlError::internal("system clock before UNIX epoch"))? .as_secs(); // Capture the creating principal's roles onto the subscription record. diff --git a/nodedb/src/control/server/shared/ddl/neutral/change_stream/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/change_stream/drop.rs index 80795fc90..db30f3e3e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/change_stream/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/change_stream/drop.rs @@ -81,11 +81,11 @@ pub fn drop_change_stream( name: name.clone(), }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { let _ = catalog .delete_change_stream(database_id, tenant_id, &name) - .map_err(|e| DdlError::new("XX000", format!("catalog delete: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; state .stream_registry .unregister(database_id, tenant_id, &name); diff --git a/nodedb/src/control/server/shared/ddl/neutral/cluster/raft.rs b/nodedb/src/control/server/shared/ddl/neutral/cluster/raft.rs index 19c2977da..08362b1d4 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/cluster/raft.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/cluster/raft.rs @@ -260,7 +260,7 @@ pub fn alter_raft_group( }; let data = change .to_entry_data() - .map_err(|e| ddl_err("XX000", format!("conf_change encode: {e}")))?; + .map_err(|e| DdlError::internal(format!("conf_change encode: {e}")))?; // Find a vShard that maps to this group to propose through Raft. let routing = match &state.cluster_routing { @@ -286,6 +286,6 @@ pub fn alter_raft_group( command: "ALTER RAFT GROUP".to_string(), rows_affected: None, }]), - Err(e) => Err(ddl_err("XX000", format!("propose failed: {e}"))), + Err(e) => Err(DdlError::from_error_in_context("propose failed", &e)), } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/cluster/rebalance_cmd.rs b/nodedb/src/control/server/shared/ddl/neutral/cluster/rebalance_cmd.rs index 593379f46..a9086ecd4 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/cluster/rebalance_cmd.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/cluster/rebalance_cmd.rs @@ -51,7 +51,7 @@ pub fn rebalance( let topo = topo.read().unwrap_or_else(|p| p.into_inner()); let plan = nodedb_cluster::compute_plan(&routing, &topo) - .map_err(|e| ddl_err("XX000", format!("rebalance planning failed: {e}")))?; + .map_err(|e| DdlError::internal(format!("rebalance planning failed: {e}")))?; if plan.is_empty() { let mut row = Map::new(); diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/add_column.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/add_column.rs index 6976f6749..d2facea4d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/add_column.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/add_column.rs @@ -126,7 +126,7 @@ pub(super) async fn alter_table_add_column( if let Some(ref coll) = updated { super::super::register::dispatch_register_from_stored(state, coll) .await - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; super::strict_schema::recompile_rls_policies(state, coll)?; } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/materialized_sum.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/materialized_sum.rs index 89fdbc1aa..e9aa64422 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/materialized_sum.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/materialized_sum.rs @@ -90,7 +90,7 @@ pub(super) async fn add_materialized_sum( let existing_bindings: Vec = catalog .load_collections_for_tenant(database_id, tenant_id) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .into_iter() .flat_map(|c| c.materialized_sums) .collect(); @@ -107,7 +107,7 @@ pub(super) async fn add_materialized_sum( // maintenance write is still rejected on an unknown field. super::super::register::dispatch_register_from_stored(state, &coll) .await - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; // The SOURCE has to be re-registered too, and it is the half that decides // whether anything is folded at all: the binding is stored here on the @@ -117,7 +117,7 @@ pub(super) async fn add_materialized_sum( // nothing and the total silently stays where it was. super::super::register::dispatch_register_for_sum_sources(state, &coll) .await - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); @@ -153,13 +153,13 @@ fn declare_target_column( } let config_json = coll.timeseries_config.as_deref().ok_or_else(|| { - err( - "XX000", - format!("strict collection '{}' has no stored schema", coll.name), - ) + DdlError::internal(format!( + "strict collection '{}' has no stored schema", + coll.name + )) })?; let mut schema: StrictSchema = sonic_rs::from_str(config_json) - .map_err(|e| err("XX000", format!("strict schema decode: {e}")))?; + .map_err(|e| DdlError::internal(format!("strict schema decode: {e}")))?; if schema.columns.iter().any(|c| c.name == column) { return Err(err( diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/ownership.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/ownership.rs index 0592adb88..81a61fe70 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/ownership.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/ownership.rs @@ -76,11 +76,11 @@ pub(super) fn alter_collection_owner( stored.owner = new_owner.to_string(); let entry = CatalogEntry::PutCollection(Box::new(stored.clone())); let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| err("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { catalog .put_collection(database_id, &stored) - .map_err(|e| err("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; state.permissions.install_replicated_owner( &crate::control::security::catalog::StoredOwner { database_id: stored.database_id.as_u64(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/strict_schema.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/strict_schema.rs index 2e48e4b01..c4699df0f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/strict_schema.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/strict_schema.rs @@ -49,7 +49,7 @@ pub(super) fn load_strict_collection( .timeseries_config .as_deref() .and_then(|s| sonic_rs::from_str(s).ok()) - .ok_or_else(|| err("XX000", "strict schema missing or malformed"))?; + .ok_or_else(|| DdlError::internal("strict schema missing or malformed"))?; Ok((coll, schema)) } @@ -120,7 +120,7 @@ pub(super) async fn persist_schema_change( super::super::register::dispatch_register_from_stored(state, updated) .await - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; recompile_rls_policies(state, updated)?; state.schema_version.bump(); Ok(()) @@ -144,5 +144,5 @@ pub(super) fn recompile_rls_policies( updated.tenant_id, &updated.name, ) - .map_err(|e| err("XX000", format!("rls recompile: {e}"))) + .map_err(|e| DdlError::from_error_in_context("rls recompile", &e)) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/support.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/support.rs index 7e3254363..81f13cc47 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/support.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/support.rs @@ -6,7 +6,8 @@ //! and messages the pgwire handlers produced), the single-row `ALTER`-status //! result builder, and the neutral `propose_and_apply` mirror of the pgwire //! `ddl::catalog_propose::propose_and_apply` (same propose + local-apply -//! ordering, same `XX000` / `"metadata propose: {e}"` error). +//! ordering). A propose error keeps its own SQLSTATE under a +//! `"metadata propose"` prefix. use nodedb_types::DatabaseId; @@ -52,7 +53,7 @@ pub(super) fn load_active_collection( .credentials .catalog() .get_collection(database_id, tenant_id, name) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .filter(|c| c.is_active) .ok_or_else(|| err("42P01", format!("collection '{name}' does not exist"))) } @@ -67,7 +68,7 @@ pub(super) fn propose_and_apply( entry: &CatalogEntry, ) -> Result { let outcome = propose_catalog_entry(state, entry) - .map_err(|e| err("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; apply_locally_if_needed(state, entry, outcome); Ok(outcome) } @@ -91,7 +92,7 @@ pub(super) async fn propose_and_apply_async( entry: CatalogEntry, ) -> Result { let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| err("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { // Clone only the cheap `Arc` handle (not `SharedState`) so // the blocking closure owns exactly what the apply needs. @@ -100,8 +101,8 @@ pub(super) async fn propose_and_apply_async( crate::control::catalog_entry::apply::apply_to(&entry, &catalog) }) .await - .map_err(|e| err("XX000", format!("catalog apply join: {e}")))? - .map_err(|e| err("XX000", format!("catalog apply: {e}")))?; + .map_err(|e| DdlError::internal(format!("catalog apply join: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog apply", &e))?; } Ok(outcome) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/vector_model.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/vector_model.rs index 9bb4b65a3..b3885a5d8 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/vector_model.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/vector_model.rs @@ -18,7 +18,6 @@ use crate::control::server::shared::ddl::result::DdlError; use crate::control::state::SharedState; use super::super::super::vector_replicate::{propose_delete_model, propose_put_model}; -use super::support::err; /// Drop `column`'s embedding-model row on every node. pub(super) fn drop_vector_model_row( @@ -52,7 +51,7 @@ pub(super) fn move_vector_model_row( .credentials .catalog() .get_vector_model(db, tenant_id, collection, old_column) - .map_err(|e| err("XX000", format!("read vector model: {e}")))? + .map_err(|e| DdlError::from_error_in_context("read vector model", &e))? else { return Ok(()); }; @@ -74,7 +73,7 @@ fn model_row_exists( .catalog() .get_vector_model(database_id, tenant_id, collection, column) .map(|row| row.is_some()) - .map_err(|e| err("XX000", format!("read vector model: {e}"))) + .map_err(|e| DdlError::from_error_in_context("read vector model", &e)) } #[cfg(test)] diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/entry.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/entry.rs index a351d2ff8..fe76a4f37 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/entry.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/entry.rs @@ -162,9 +162,9 @@ fn check_engine_support( Ok(Some(c)) => c, Ok(None) => return Ok(()), // Collection doesn't exist yet — will fail at INSERT. Err(e) => { - return Err(ddl_err( - "XX000", - format!("COPY: catalog lookup failed: {e}"), + return Err(DdlError::from_error_in_context( + "COPY: catalog lookup failed", + &e, )); } }; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/entry.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/entry.rs index 6692abece..07b65fdee 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/entry.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/entry.rs @@ -127,9 +127,9 @@ fn check_collection_exists( "42P01", format!("COPY TO: collection \"{collection}\" does not exist"), )), - Err(e) => Err(ddl_err( - "XX000", - format!("COPY TO: catalog lookup failed: {e}"), + Err(e) => Err(DdlError::from_error_in_context( + "COPY TO: catalog lookup failed", + &e, )), } } @@ -194,7 +194,7 @@ async fn execute_and_collect( TraceId::ZERO, ) .await - .map_err(|e| ddl_err("XX000", format!("COPY TO: dispatch failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("COPY TO: dispatch failed", &e))?; if resp.payload.is_empty() { continue; @@ -224,12 +224,8 @@ fn extract_json_rows( if json.is_empty() { return Ok(()); } - let mut parsed: serde_json::Value = sonic_rs::from_str(json).map_err(|e| { - ddl_err( - "XX000", - format!("COPY TO: failed to decode result rows: {e}"), - ) - })?; + let mut parsed: serde_json::Value = sonic_rs::from_str(json) + .map_err(|e| DdlError::internal(format!("COPY TO: failed to decode result rows: {e}")))?; redact_decoded_value(Some(redaction), store, &mut parsed); match parsed { serde_json::Value::Array(items) => { diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/format.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/format.rs index 7d2003dc3..b552612ac 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/format.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/format.rs @@ -36,7 +36,7 @@ fn serialize_ndjson(rows: &[serde_json::Value]) -> Result, DdlError> { let mut out = Vec::with_capacity(rows.len() * 64); for row in rows { let line = sonic_rs::to_vec(row) - .map_err(|e| ddl_err("XX000", format!("COPY TO: JSON serialization error: {e}")))?; + .map_err(|e| DdlError::internal(format!("COPY TO: JSON serialization error: {e}")))?; out.extend_from_slice(&line); out.push(b'\n'); } @@ -46,12 +46,8 @@ fn serialize_ndjson(rows: &[serde_json::Value]) -> Result, DdlError> { fn serialize_json_array(rows: &[serde_json::Value]) -> Result, DdlError> { // Build a serde_json::Value::Array and serialize once. let arr = serde_json::Value::Array(rows.to_vec()); - let bytes = sonic_rs::to_vec(&arr).map_err(|e| { - ddl_err( - "XX000", - format!("COPY TO: JSON array serialization error: {e}"), - ) - })?; + let bytes = sonic_rs::to_vec(&arr) + .map_err(|e| DdlError::internal(format!("COPY TO: JSON array serialization error: {e}")))?; Ok(bytes) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/create/build.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/create/build.rs index 324cd03e6..836260ad6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/create/build.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/create/build.rs @@ -116,7 +116,7 @@ pub async fn build_and_persist( let catalog = state.credentials.catalog(); if catalog .get_materialized_view(database_id.as_u64(), tenant_id.as_u64(), name) - .map_err(|error| err("XX000", error.to_string()))? + .map_err(|error| DdlError::from_error(&error))? .is_some() { return Err(err( @@ -130,7 +130,7 @@ pub async fn build_and_persist( // over a soft-deleted incarnation's still-present storage. let existing = catalog .get_collection(database_id, tenant_id.as_u64(), name) - .map_err(|error| err("XX000", error.to_string()))?; + .map_err(|error| DdlError::from_error(&error))?; if let Some(existing) = existing { if existing.is_active { return Err(err( @@ -176,7 +176,7 @@ pub async fn build_and_persist( { guard.disarm(); } - return Err(err("XX000", failure.error.to_string())); + return Err(DdlError::from_error(&failure.error)); } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/indexed_vector_fields.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/indexed_vector_fields.rs new file mode 100644 index 000000000..f9a4d928c --- /dev/null +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/indexed_vector_fields.rs @@ -0,0 +1,75 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The fields of a `{ ... }` INSERT that the document write already indexes +//! into a vector index. +//! +//! The Data Plane indexes a document's vectors when it stores the document: +//! +//! - A strict collection indexes every `VECTOR(n)` column. +//! - Any other collection indexes each field that has its own vector index. +//! When no field has one, a default-field vector index covers `embedding`. +//! +//! The `{ ... }` handler also sends a vector insert for numeric-array +//! fields, so a field no index covers is still searchable. It skips the +//! fields listed here. A second insert of such a field appends a second +//! HNSW node for the same row. + +use std::collections::HashSet; + +use nodedb_types::{CollectionType, ColumnType, DatabaseId, DocumentMode}; + +use crate::control::server::shared::ddl::result::DdlError; +use crate::control::state::SharedState; + +/// The default vector field a vector index without a field name covers. +const DEFAULT_VECTOR_FIELD: &str = "embedding"; + +/// The fields of `collection` the document write indexes into a vector +/// index. Fails when the catalog cannot be read: guessing would either +/// index a field twice or not at all. +pub(super) fn indexed_vector_fields( + state: &SharedState, + database_id: DatabaseId, + tenant_id: u64, + collection: &str, + collection_type: Option<&CollectionType>, +) -> Result, DdlError> { + if let Some(CollectionType::Document(DocumentMode::Strict(schema))) = collection_type { + let strict: HashSet = schema + .columns + .iter() + .filter(|c| matches!(c.column_type, ColumnType::Vector(_))) + .map(|c| c.name.clone()) + .collect(); + if !strict.is_empty() { + return Ok(strict); + } + } + + let params = state + .credentials + .catalog() + .list_vector_index_params_in_database(database_id.as_u64()) + .map_err(|e| { + DdlError::from_error_in_context( + &format!("read vector indexes of \"{collection}\" for INSERT"), + &e, + ) + })?; + let mut named = HashSet::new(); + let mut has_default = false; + for p in params + .iter() + .filter(|p| p.tenant_id == tenant_id && p.collection == collection) + { + if p.field_name.is_empty() { + has_default = true; + } else { + named.insert(p.field_name.clone()); + } + } + if named.is_empty() && has_default { + named.insert(DEFAULT_VECTOR_FIELD.to_string()); + } + Ok(named) +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs index 031b042f4..0d6573484 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs @@ -15,6 +15,7 @@ use crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate; use crate::control::server::shared::session::{DmlTxnCtx, PendingFieldInference}; use crate::control::state::SharedState; +use super::indexed_vector_fields::indexed_vector_fields; use super::parse::{ authorize_write_target, dispatch_plan, extract_vector_fields, fields_to_insert_sql, parse_write_statement, plan_and_dispatch, @@ -100,9 +101,10 @@ pub async fn insert_document( fields.insert(field_def.name.clone(), typed_val); } Err(e) => { - return Some(Err(ddl_err( - "XX000", - format!("sequence '{seq_name}' error: {e}"), + return Some(Err(DdlError::from_error( + &crate::control::sequence::error_map::sequence_error_to_error( + seq_name, e, + ), ))); } } @@ -224,9 +226,9 @@ pub async fn insert_document( &pending.fields, ) { - return Some(Err(ddl_err( - "XX000", - format!("record inferred schema fields: {e}"), + return Some(Err(DdlError::from_error_in_context( + "record inferred schema fields", + &e, ))); } } @@ -245,10 +247,25 @@ pub async fn insert_document( return Some(err); } - // Dispatch VectorInsert for vector fields. + // Dispatch VectorInsert for the numeric-array fields no vector index + // covers. The document write above already indexed the covered ones, and + // a second insert would append a second HNSW node for the same row. + let indexed = match indexed_vector_fields( + state, + database_id, + tenant_id.as_u64(), + &parsed.coll_name, + parsed.collection_type.as_ref(), + ) { + Ok(indexed) => indexed, + Err(e) => return Some(Err(e)), + }; let vec_vshard = - crate::types::VShardId::from_collection_in_database(database_id, &parsed.coll_name); + nodedb_types::CollectionKey::from_bare(database_id, &parsed.coll_name).vshard(); for (field_name, vector) in extract_vector_fields(&fields) { + if indexed.contains(&field_name) { + continue; + } let dim = vector.len(); { @@ -276,14 +293,13 @@ pub async fn insert_document( } } let surrogate = match state.surrogate_assigner.assign( - database_id, + nodedb_types::CollectionKey::from_bare(database_id, &parsed.coll_name), tenant_id, - &parsed.coll_name, parsed.doc_id.as_bytes(), ) { Ok(s) => s, Err(e) => { - return Some(Err(ddl_err("XX000", format!("surrogate assign: {e}")))); + return Some(Err(DdlError::from_error_in_context("surrogate assign", &e))); } }; let vec_plan = crate::bridge::envelope::PhysicalPlan::Vector(VectorOp::Insert { diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/mod.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/mod.rs index 544c1b542..d2f1a15a9 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/mod.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/mod.rs @@ -2,6 +2,7 @@ //! Protocol-neutral collection DML: INSERT INTO / UPSERT INTO. +mod indexed_vector_fields; mod insert; mod parse; mod triggers; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs index 0e87463ab..92f0373ff 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs @@ -27,7 +27,8 @@ use crate::types::TraceId; use super::types::ddl_err; -/// Dispatch a plan to WAL + Data Plane, returning an error response on failure. +/// Dispatch a write plan on the durable route, returning an error response on +/// failure. `None` means the write applied. pub(in crate::control::server::shared::ddl::neutral::collection) async fn dispatch_plan( state: &SharedState, identity: &AuthenticatedIdentity, @@ -69,18 +70,30 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn dispat } }; - if let Err(error) = - crate::control::server::dispatch_utils::dispatch_authorized_autocommit_write( - state, - checked, - TraceId::ZERO, - ) - .await + // The durable route: Raft in cluster mode, else the funnel's `AppendHere`. + match crate::control::server::dispatch_utils::dispatch_authorized_durable_write( + state, + checked, + TraceId::ZERO, + ) + .await { - let (_, sqlstate, message) = error_to_sqlstate(&error); - return Some(Err(ddl_err(sqlstate, message))); + Err(error) => { + let (_, sqlstate, message) = error_to_sqlstate(&error); + Some(Err(ddl_err(sqlstate, message))) + } + // A refusal arrives as an error status inside an `Ok` response. + Ok(response) if response.status == crate::bridge::envelope::Status::Error => { + Some(Err(match response.error_code.as_deref() { + Some(code) => { + let (_, sqlstate, message) = error_code_to_sqlstate(code); + ddl_err(sqlstate, message) + } + None => DdlError::internal("unknown data plane error"), + })) + } + Ok(_) => None, } - None } /// Authorize a write target before triggers, sequences, or catalog reads run. @@ -139,7 +152,10 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a // un-injected copies. let (mut tasks, output_schema, versions) = { let scope = RequestAuthScope::for_database(identity, state.auth_stores(), database_id); - let permission_cache = state.permission_cache.read().await; + let permission_cache = + crate::control::security::auth_fence::permission_view(state, tenant_id) + .await + .map_err(|error| DdlError::from_error(&error))?; let sec = PlanSecurityContext { identity, auth: scope.auth(), @@ -252,13 +268,12 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a return Err(ddl_err(sqlstate, message)); } + let sum_read_vshards = crate::control::planner::calvin::read_vshards_of(&sum_target_reads) + .map_err(|error| DdlError::from_error(&error))?; if !in_txn_block && state.sequencer_inbox.get().is_some() && matches!( - crate::control::planner::calvin::classify_dispatch( - &tasks, - &crate::control::planner::calvin::read_vshards_of(&sum_target_reads), - ), + crate::control::planner::calvin::classify_dispatch(&tasks, &sum_read_vshards), crate::control::planner::calvin::DispatchClass::MultiShard { .. } ) { @@ -336,14 +351,13 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a Arc::clone(&plan_lease_scope), ) { - return Err(ddl_err( - "XX000", + return Err(DdlError::internal( "internal error: failed to retain descriptor leases for buffered transaction tasks", )); } let task = match routed { - Ok(InTxnRoute::Read(task)) => *task, + Ok(InTxnRoute::Read(task) | InTxnRoute::Autocommit(task)) => *task, Ok(InTxnRoute::Buffered) | Ok(InTxnRoute::Staged(_)) => { drop(initial_authorized); // A buffered/staged write produces its rows at COMMIT, not @@ -362,11 +376,13 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a return Err(ddl_err(sqlstate, message)); } Err(StagingGateError::Rejected { code }) => { - let (_, sqlstate, message) = match code { - Some(code) => error_code_to_sqlstate(&code), - None => ("ERROR", "XX000", "unknown data plane error".to_owned()), - }; - return Err(ddl_err(sqlstate, message)); + return Err(match code { + Some(code) => { + let (_, sqlstate, message) = error_code_to_sqlstate(&code); + ddl_err(sqlstate, message) + } + None => DdlError::internal("unknown data plane error"), + }); } }; @@ -389,8 +405,10 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a ddl_err(sqlstate, message) })? { crate::control::server::shared::clone_write::CloneCheckedOutcome::Handled(resp) => resp, + // A write takes the durable route: Raft in cluster mode, else the + // funnel's `AppendHere`. A read takes the read route. crate::control::server::shared::clone_write::CloneCheckedOutcome::Proceed(checked) => { - crate::control::server::dispatch_utils::dispatch_authorized_autocommit_write( + crate::control::server::dispatch_utils::dispatch_authorized_task_by_class( state, checked, TraceId::ZERO, @@ -404,15 +422,13 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a }; if response.status == crate::bridge::envelope::Status::Error { - let (_, sqlstate, message) = match response.error_code.as_deref() { - Some(code) => error_code_to_sqlstate(code), - None => ( - "ERROR", - "XX000", - String::from_utf8_lossy(&response.payload).into_owned(), - ), - }; - return Err(ddl_err(sqlstate, message)); + return Err(match response.error_code.as_deref() { + Some(code) => { + let (_, sqlstate, message) = error_code_to_sqlstate(code); + ddl_err(sqlstate, message) + } + None => DdlError::internal(String::from_utf8_lossy(&response.payload)), + }); } // Shape the STORED rows the write returned, redacted for the caller — @@ -443,7 +459,7 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a redaction: Some(redaction.ctx(&state.redaction)), sequences: Some(&sequences), }) - .map_err(|error| ddl_err("XX000", error.message().to_string()))?; + .map_err(|error| DdlError::from_error(&crate::Error::from(error)))?; // Folded rather than pushed: a statement is ONE result set, however // many tasks it planned to. if let ShapeOutcome::Rows(shaped) = outcome { diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/triggers.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/triggers.rs index 16c4bac48..b9292a1e3 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/triggers.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/triggers.rs @@ -42,7 +42,7 @@ pub(super) async fn fire_sync_after_triggers( .await .into_result() { - return Some(Err(ddl_err("XX000", &format!("trigger error: {e}")))); + return Some(Err(DdlError::from_error_in_context("trigger error", &e))); } None } @@ -84,7 +84,7 @@ pub(super) async fn fire_sync_after_update_triggers( .await .into_result() { - return Some(Err(ddl_err("XX000", &format!("trigger error: {e}")))); + return Some(Err(DdlError::from_error_in_context("trigger error", &e))); } None } @@ -120,7 +120,7 @@ pub(super) async fn fire_instead_triggers( }])) } Ok(crate::control::trigger::fire_instead::InsteadOfResult::NoTrigger) => None, - Err(e) => Some(Err(ddl_err("XX000", &format!("trigger error: {e}")))), + Err(e) => Some(Err(DdlError::from_error_in_context("trigger error", &e))), } } @@ -146,10 +146,9 @@ pub(super) async fn fire_before_triggers( .await { Ok(f) => Ok(f), - Err(e) => Err(Err(ddl_err("XX000", &format!("BEFORE trigger error: {e}")))), + Err(e) => Err(Err(DdlError::from_error_in_context( + "BEFORE trigger error", + &e, + ))), } } - -fn ddl_err(sqlstate: &str, msg: &str) -> DdlError { - DdlError::new(sqlstate, msg) -} diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/drop.rs index 7fb2afb8b..6a479f455 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/drop.rs @@ -98,7 +98,7 @@ pub fn drop_collection( name, &mut visited, ) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? }; // Implicit SERIAL/BIGSERIAL sequences (`{collection}_{field}_seq`) @@ -187,7 +187,7 @@ pub fn drop_collection( let catalog = state.credentials.catalog(); if catalog .get_materialized_view(database_id.as_u64(), tenant_id.as_u64(), name) - .map_err(|error| err("XX000", error.to_string()))? + .map_err(|error| DdlError::from_error(&error))? .is_some() { return Err(err( @@ -271,7 +271,7 @@ pub fn drop_collection( None }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|error| err("XX000", error.to_string()))?; + .map_err(|error| DdlError::from_error(&error))?; if outcome.needs_local_apply() { let catalog = state.credentials.catalog(); if purge { @@ -323,7 +323,9 @@ pub fn drop_collection( }, catalog, ) - .map_err(|error| err("XX000", format!("catalog deactivate failed: {error}")))?; + .map_err(|error| { + DdlError::from_error_in_context("catalog deactivate failed", &error) + })?; } } @@ -337,9 +339,9 @@ pub fn drop_collection( catalog .delete_sequence(database_id.as_u64(), tenant_id.as_u64(), &seq.name) .map_err(|e| { - err( - "XX000", - format!("failed to drop sequence '{}': {e}", seq.name), + DdlError::from_error_in_context( + &format!("failed to drop sequence '{}'", seq.name), + &e, ) })?; // Best-effort: registry removal is non-critical since catalog diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs new file mode 100644 index 000000000..d350dc8f0 --- /dev/null +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs @@ -0,0 +1,160 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Build a secondary index committed in the `Building` state: backfill it on +//! every node, then mark it `Ready`. +//! +//! An autocommit `CREATE INDEX` runs this at once. Inside an explicit +//! transaction the statement buffers the `Building` entry and defers this to +//! COMMIT, so a ROLLBACK leaves no index entry behind. + +use crate::control::security::catalog::IndexBuildState; +use crate::control::server::shared::session::ddl_effect::SecondaryIndexBuild; +use crate::control::state::SharedState; +use crate::types::TraceId; + +use super::super::super::super::result::DdlError; +use super::commit::commit_collection_mutation; + +/// Backfill `build` on every node and flip it to `Ready`. +/// +/// A collection that no longer carries the index by name is skipped: the +/// same transaction dropped it before COMMIT. UNIQUE violations surface as +/// SQLSTATE 23505 and leave the index `Building`, so a later `DROP INDEX` +/// removes it after the data is fixed. +pub(crate) async fn build_secondary_index( + state: &SharedState, + build: &SecondaryIndexBuild, +) -> Result<(), DdlError> { + let SecondaryIndexBuild { + tenant_id, + database_id, + collection, + index_name, + extraction_path, + is_array, + unique, + case_insensitive, + predicate, + } = build; + let (tenant_id, database_id) = (*tenant_id, *database_id); + let catalog = state.credentials.catalog(); + let Some(coll) = catalog + .get_collection(database_id, tenant_id.as_u64(), collection) + .map_err(|e| DdlError::from_error(&e))? + .filter(|coll| coll.indexes.iter().any(|i| &i.name == index_name)) + else { + return Ok(()); + }; + + // A key-value collection's rows live in the KV engine, which keeps its + // own secondary indexes: the index is built there, with a backfill. + if coll.collection_type.is_kv() { + let field = super::kv_index::kv_field( + collection, + &build_path(extraction_path, *is_array), + &super::kv_index::KvIndexOptions { + unique: *unique, + case_insensitive: *case_insensitive, + predicate: predicate.as_deref(), + }, + )?; + super::kv_index::register_kv_index(state, tenant_id, database_id, &coll, &field).await?; + return mark_ready(state, build).await; + } + + // Register the Building index on this node before the backfill scans, + // so a write that lands after the scan maintains it. The post-apply + // register of a transaction's buffered entry runs asynchronously. + super::super::dispatch_register_from_stored(state, &coll) + .await + .map_err(|e| DdlError::from_error(&e))?; + + // The backfill runs on the local Data Plane (single node) or the leader + // (cluster), vShard-local per core. + let vshard = nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(); + let backfill_plan = crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::BackfillIndex { + collection: nodedb_types::QualifiedCollection::new(database_id, collection), + path: extraction_path.clone(), + is_array: *is_array, + unique: *unique, + case_insensitive: *case_insensitive, + predicate: predicate.clone(), + }, + ); + let backfill_resp = crate::control::server::dispatch_utils::dispatch_to_data_plane( + state, + tenant_id, + database_id, + vshard, + backfill_plan, + TraceId::ZERO, + ) + .await + .map_err(|e| DdlError::from_error(&e))?; + + if backfill_resp.status == crate::bridge::envelope::Status::Error { + // A coded refusal keeps its SQLSTATE: a duplicate key is `23505`. + return Err(match backfill_resp.error_code.as_deref() { + Some(code) => DdlError::from_error_in_context( + "index backfill", + &crate::Error::DataPlane(code.clone()), + ), + None => DdlError::internal(format!( + "index backfill: {}", + String::from_utf8_lossy(&backfill_resp.payload) + )), + }); + } + + // Every other node backfills the rows it hosts. Single-node and peerless + // clusters return at once. + super::super::index_fanout::backfill_on_peers( + state, + super::super::index_fanout::PeerBackfill { + tenant_id, + database_id, + collection, + path: extraction_path, + is_array: *is_array, + unique: *unique, + case_insensitive: *case_insensitive, + predicate: predicate.as_deref(), + }, + ) + .await?; + + mark_ready(state, build).await +} + +/// The path as the statement named it: an array path keeps its `[]` suffix. +fn build_path(extraction_path: &str, is_array: bool) -> String { + if is_array { + format!("{extraction_path}[]") + } else { + extraction_path.to_string() + } +} + +/// Flip the built index to `Ready` on a fresh read, so a concurrent mutation +/// of the collection is folded in before the index vector is rewritten. A +/// collection dropped meanwhile has no index left to flip. +async fn mark_ready(state: &SharedState, build: &SecondaryIndexBuild) -> Result<(), DdlError> { + let catalog = state.credentials.catalog(); + if let Some(mut ready_coll) = catalog + .get_collection( + build.database_id, + build.tenant_id.as_u64(), + &build.collection, + ) + .map_err(|e| DdlError::from_error(&e))? + { + for idx in ready_coll.indexes.iter_mut() { + if idx.name == build.index_name { + idx.state = IndexBuildState::Ready; + } + } + commit_collection_mutation(state, &ready_coll, build.database_id).await?; + } + Ok(()) +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/commit.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/commit.rs index bc21d03ec..ff7ebfa1f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/commit.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/commit.rs @@ -24,20 +24,20 @@ pub(super) async fn commit_collection_mutation( ) -> Result<(), DdlError> { let entry = crate::control::catalog_entry::CatalogEntry::PutCollection(Box::new(coll.clone())); let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if outcome.needs_local_apply() { { let catalog = state.credentials.catalog(); catalog .put_collection(database_id, coll) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; } // Single-node path bypasses the applier post-apply hook, so the // Register refresh has to be fired here. In cluster mode the // applier's `put_async` does it on every node. super::super::dispatch_register_from_stored(state, coll) .await - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; } Ok(()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/create.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/create.rs index d327bb818..9c4f5c330 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/create.rs @@ -21,11 +21,13 @@ use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::ddl::index_registry::{ IndexRegistration, propose_index_record, }; +use crate::control::server::shared::session::ddl_buffer; +use crate::control::server::shared::session::ddl_effect::{DeferredDdlEffect, SecondaryIndexBuild}; use crate::control::state::SharedState; use crate::types::DatabaseId; -use crate::types::TraceId; use super::super::super::super::result::{DdlError, DdlResult}; +use super::build::build_secondary_index; use super::commit::{commit_collection_mutation, err}; /// Normalize a user-supplied field reference into the canonical JSON path @@ -147,7 +149,7 @@ pub async fn create_index( // loudly — only a genuine name collision is absorbed by `IF NOT EXISTS`. if let Some(existing) = catalog .get_index_record(database_id.as_u64(), tenant_id.as_u64(), &index_name) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? { if if_not_exists { return Ok(create_index_ok()); @@ -170,6 +172,21 @@ pub async fn create_index( .unwrap_or(&canonical_field) .to_string(); + // A key-value collection's index lives in the KV engine, which indexes + // one top-level field by equality. Refuse what it cannot build before + // any catalog entry is written. + if coll.collection_type.is_kv() { + super::kv_index::kv_field( + collection, + &canonical_field, + &super::kv_index::KvIndexOptions { + unique: is_unique, + case_insensitive, + predicate: where_condition.as_deref(), + }, + )?; + } + // Two-phase Building→Ready pipeline. Phase 1: stamp `Building` and // commit — readers skip the index (planner filters to Ready), writers // dual-write (extraction iterates every registered path regardless of @@ -189,85 +206,22 @@ pub async fn create_index( commit_collection_mutation(state, &coll, database_id).await?; - // Phase 2: dispatch the backfill op. This runs on the local Data - // Plane (single-node) or the leader (cluster — distributed backfill - // across vShards is handled inside the handler by the existing scan - // primitive, which is vShard-local per core). UNIQUE violations here - // surface as a Data Plane error; we propagate as SQLSTATE 23505 and - // leave the index in `Building` so a subsequent retry can DROP + try - // with a wider data fix. - let vshard = crate::types::VShardId::from_collection_in_database(database_id, collection); - let backfill_plan = crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::BackfillIndex { - collection: nodedb_types::QualifiedCollection::new(database_id, collection), - path: extraction_path.clone(), - is_array, - unique: is_unique, - case_insensitive, - predicate: where_condition.clone(), - }, - ); - let backfill_resp = crate::control::server::dispatch_utils::dispatch_to_data_plane( - state, + // Phase 2 and 3: backfill on every node, then flip to Ready. Inside an + // explicit transaction the build waits for COMMIT, after the Building + // entry lands; a ROLLBACK discards it with the entry. + let build = SecondaryIndexBuild { tenant_id, database_id, - vshard, - backfill_plan, - TraceId::ZERO, - ) - .await - .map_err(|e| err("XX000", e.to_string()))?; - - if backfill_resp.status == crate::bridge::envelope::Status::Error { - let detail = match backfill_resp.error_code.as_deref() { - Some(crate::bridge::envelope::ErrorCode::Internal { detail, .. }) => detail.clone(), - Some(other) => format!("{other:?}"), - None => String::from_utf8_lossy(&backfill_resp.payload).into_owned(), - }; - let code = if detail.to_lowercase().contains("unique") { - "23505" - } else { - "XX000" - }; - return Err(err(code, detail)); - } - - // Phase 2b: fan the same backfill op to every other cluster node. - // `execute_backfill_index` is vShard-local per core, so without - // this step non-coordinator nodes never populate the index for - // the rows they host — the silent-miss bug. Single-node and - // peerless clusters short-circuit inside the helper. - super::super::index_fanout::backfill_on_peers( - state, - super::super::index_fanout::PeerBackfill { - tenant_id, - database_id, - collection, - path: &extraction_path, - is_array, - unique: is_unique, - case_insensitive, - predicate: where_condition.as_deref(), - }, - ) - .await?; - - // Phase 3: flip to Ready. Re-read the collection so any concurrent - // mutation (e.g. another DDL on the same collection — blocked by - // descriptor drain in cluster mode, serialized by pgwire session in - // single-node) is folded in before we rewrite the index vector. - if let Some(latest) = catalog - .get_collection(database_id, tenant_id.as_u64(), collection) - .ok() - .flatten() - { - let mut ready_coll = latest; - for idx in ready_coll.indexes.iter_mut() { - if idx.name == index_name { - idx.state = IndexBuildState::Ready; - } - } - commit_collection_mutation(state, &ready_coll, database_id).await?; + collection: collection.to_string(), + index_name: index_name.clone(), + extraction_path: extraction_path.clone(), + is_array, + unique: is_unique, + case_insensitive, + predicate: where_condition.clone(), + }; + if !ddl_buffer::defer_effect(DeferredDdlEffect::SecondaryIndexBuild(build.clone())) { + build_secondary_index(state, &build).await?; } // Identity record: what SHOW INDEXES lists and DROP INDEX resolves. diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/drop.rs index 4e2639745..c348f59a1 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/drop.rs @@ -20,6 +20,7 @@ use crate::control::security::audit::AuditEvent; use crate::control::security::catalog::{IndexKind, StoredIndexRecord}; use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::shared::session::ddl_buffer; use crate::control::state::SharedState; use crate::types::DatabaseId; @@ -67,7 +68,7 @@ pub async fn drop_index( .credentials .catalog() .get_index_record(database_id.as_u64(), tenant_id.as_u64(), index_name) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? // An index whose collection is soft-deleted is not listed and cannot // be dropped on its own: the collection owns its lifecycle, and // UNDROP must bring it back intact. @@ -118,9 +119,16 @@ pub async fn drop_index( )); } - // Engine + kind-specific catalog state first: if any of it survives, the - // identity record must survive with it so the drop can be retried. - super::teardown::teardown(state, &record, database_id, tenant_id).await?; + // Autocommit: engine and kind-specific catalog state first. If any of it + // survives, the identity record survives with it so the drop can be + // retried. Inside an explicit transaction every entry is buffered and + // the engine teardown waits for COMMIT, so the records are buffered + // first and the teardown's deferred effects ride on this statement's + // own entries: a ROLLBACK TO SAVEPOINT then discards them together. + let in_transaction = ddl_buffer::is_active(); + if !in_transaction { + super::teardown::teardown(state, &record, database_id, tenant_id).await?; + } super::super::super::super::index_registry::propose_delete_index_record( state, @@ -138,6 +146,10 @@ pub async fn drop_index( index_name, )?; + if in_transaction { + super::teardown::teardown(state, &record, database_id, tenant_id).await?; + } + state.audit_record( AuditEvent::AdminAction, Some(tenant_id), diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/kv_index.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/kv_index.rs new file mode 100644 index 000000000..5222ebec0 --- /dev/null +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/kv_index.rs @@ -0,0 +1,183 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The key-value engine side of a secondary index. +//! +//! A KV collection keeps its rows in the `KvEngine`, which maintains its own +//! per-field secondary indexes on every write. `CREATE INDEX` on a KV +//! collection therefore builds the index there, and `DROP INDEX` removes it +//! there. Both run through the autocommit write funnel, which appends the +//! `kv_register_index` / `kv_drop_index` WAL record. That record and the KV +//! checkpoint carry the index across a restart. +//! +//! The KV engine indexes one top-level field by equality. It has no UNIQUE +//! check, no COLLATE NOCASE fold, no partial predicate and no array or nested +//! path, so `CREATE INDEX` refuses those forms on a KV collection before any +//! catalog entry is written. + +use crate::control::security::catalog::StoredCollection; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId, TraceId}; +use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; +use nodedb_types::QualifiedCollection; + +use super::super::super::super::result::DdlError; +use super::commit::err; + +/// The index options a `CREATE INDEX` statement asks for. +pub(super) struct KvIndexOptions<'a> { + pub unique: bool, + pub case_insensitive: bool, + pub predicate: Option<&'a str>, +} + +/// The KV field a canonical index path names. +/// +/// Refuses the options and paths the KV engine cannot index, with SQLSTATE +/// 0A000. +pub(super) fn kv_field( + collection: &str, + canonical_field: &str, + options: &KvIndexOptions<'_>, +) -> Result { + let refuse = |what: &str| { + err( + "0A000", + format!( + "{what} is not supported on key-value collection '{collection}': \ + its index matches one top-level field by equality" + ), + ) + }; + if options.unique { + return Err(refuse("a UNIQUE index")); + } + if options.case_insensitive { + return Err(refuse("COLLATE NOCASE")); + } + if options.predicate.is_some() { + return Err(refuse("a partial index (WHERE)")); + } + let field = canonical_field + .strip_prefix("$.") + .unwrap_or(canonical_field); + if field.is_empty() || field.contains('.') || field.contains('[') || field.starts_with('$') { + return Err(refuse(&format!("the index path '{canonical_field}'"))); + } + Ok(field.to_string()) +} + +/// Build the index on the core that holds the collection's rows, backfilled +/// from every row it already holds. +pub(super) async fn register_kv_index( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + coll: &StoredCollection, + field: &str, +) -> Result<(), DdlError> { + // The engine keeps a field's schema position with the index for the + // checkpoint. An undeclared field has none, and extraction is by name. + let field_position = coll + .fields + .iter() + .position(|(name, _)| name == field) + .unwrap_or(0); + let plan = PhysicalPlan::Kv(KvOp::RegisterIndex { + collection: QualifiedCollection::new(database_id, &coll.name), + field: field.to_string(), + field_position, + backfill: true, + }); + dispatch_durable(state, tenant_id, database_id, &coll.name, plan, "build").await +} + +/// Remove the index from the core that holds the collection's rows. A field +/// with no index is already in the state the drop asks for. +pub(crate) async fn drop_kv_index( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection: &str, + field: &str, +) -> Result<(), DdlError> { + let plan = PhysicalPlan::Kv(KvOp::DropIndex { + collection: QualifiedCollection::new(database_id, collection), + field: field.to_string(), + }); + dispatch_durable(state, tenant_id, database_id, collection, plan, "drop").await +} + +/// Dispatch a KV index plan through the autocommit write funnel, which +/// appends its WAL record, and fail on a refused reply. A refusal keeps its +/// SQLSTATE and code. +async fn dispatch_durable( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection: &str, + plan: PhysicalPlan, + step: &str, +) -> Result<(), DdlError> { + crate::control::server::dispatch_utils::dispatch_autocommit_write( + state, + crate::control::server::dispatch_utils::AutocommitWrite { + tenant_id, + database_id, + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(), + plan, + trace_id: TraceId::ZERO, + event_source: crate::event::EventSource::User, + txn_id: None, + }, + ) + .await + .and_then(crate::control::server::shared::response_payload::payload_or_typed_error) + .map_err(|e| { + DdlError::from_error_in_context(&format!("key-value index {step} on '{collection}'"), &e) + })?; + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn plain() -> KvIndexOptions<'static> { + KvIndexOptions { + unique: false, + case_insensitive: false, + predicate: None, + } + } + + #[test] + fn a_top_level_field_is_indexed_by_its_name() { + assert_eq!( + kv_field("sessions", "$.region", &plain()).expect("plain field"), + "region" + ); + } + + #[test] + fn options_and_paths_the_engine_cannot_index_are_refused() { + for path in ["$.a.b", "$.tags[]", "$"] { + let error = kv_field("sessions", path, &plain()).expect_err(path); + assert_eq!(error.sqlstate, "0A000", "{path}"); + } + let unique = KvIndexOptions { + unique: true, + ..plain() + }; + assert_eq!( + kv_field("sessions", "$.region", &unique) + .expect_err("unique") + .sqlstate, + "0A000" + ); + let partial = KvIndexOptions { + predicate: Some("region = 'eu'"), + ..plain() + }; + assert!(kv_field("sessions", "$.region", &partial).is_err()); + } +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/mod.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/mod.rs index 52b48651f..9ffb03097 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/mod.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/mod.rs @@ -2,9 +2,11 @@ //! Protocol-neutral index DDL: CREATE INDEX, DROP INDEX. +pub mod build; pub mod commit; pub mod create; pub mod drop; +pub mod kv_index; pub mod teardown; pub use create::{CreateIndexRequest, create_index}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs index 1058ba637..ce0ba9a2a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs @@ -7,7 +7,7 @@ //! //! | kind | durable state created | //! |-----------|----------------------------------------------------| -//! | secondary | `StoredCollection.indexes` entry + sparse-engine index entries | +//! | secondary | `StoredCollection.indexes` entry + sparse-engine index entries, or the KV engine's field index on a key-value collection | //! | vector | `_system.vector_index_params` row + Data Plane index + checkpoint | //! | fulltext | the collection's analyzer / fuzzy binding (per collection) | //! | spatial | none beyond the registry + ownership rows | @@ -23,11 +23,15 @@ //! cannot propagate and files a `Capture` instead. use crate::control::security::catalog::{IndexKind, StoredIndexRecord}; +use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId, TraceId}; use super::super::super::super::result::DdlError; -use super::commit::{commit_collection_mutation, err}; +use crate::control::server::shared::session::ddl_buffer; +use crate::control::server::shared::session::ddl_effect::DeferredDdlEffect; + +use super::commit::commit_collection_mutation; /// Remove every piece of engine and catalog state belonging to `record`, /// except the registry and ownership rows the caller removes afterwards. @@ -53,6 +57,15 @@ pub(super) async fn teardown( // write — so its removal goes through the same route // `DROP SORTED INDEX` uses. IndexKind::Sorted => { + let deferred = ddl_buffer::defer_effect(DeferredDdlEffect::SortedIndexDrop { + tenant_id, + database_id, + collection: record.collection.clone(), + index_name: record.name.clone(), + }); + if deferred { + return Ok(()); + } super::super::super::kv_sorted_index::drop_in_engine( state, &super::super::super::kv_sorted_index::SortedIndexTarget { @@ -68,7 +81,8 @@ pub(super) async fn teardown( } /// Drop the `StoredIndex` entry from the owning collection and purge the -/// sparse engine's entries for the indexed path. +/// indexed path's entries: from the sparse engine, or from the KV engine on +/// a key-value collection. async fn secondary( state: &SharedState, record: &StoredIndexRecord, @@ -78,7 +92,7 @@ async fn secondary( let catalog = state.credentials.catalog(); let Some(mut coll) = catalog .get_collection(database_id, tenant_id.as_u64(), &record.collection) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? else { // The registry outlived its collection — the collection teardown // path already reclaimed every engine surface, so there is nothing @@ -99,13 +113,34 @@ async fn secondary( let Some(field) = dropped_field.or_else(|| record.fields.first().cloned()) else { return Ok(()); }; + // A key-value collection's index lives in the KV engine. + if coll.collection_type.is_kv() { + let field = field.strip_prefix("$.").unwrap_or(&field).to_string(); + let deferred = ddl_buffer::defer_effect(DeferredDdlEffect::KvIndexDrop { + tenant_id, + database_id, + collection: record.collection.clone(), + field: field.clone(), + }); + if deferred { + return Ok(()); + } + return super::kv_index::drop_kv_index( + state, + tenant_id, + database_id, + &record.collection, + &field, + ) + .await; + } let plan = crate::bridge::envelope::PhysicalPlan::Document( nodedb_physical::physical_plan::DocumentOp::DropIndex { collection: nodedb_types::QualifiedCollection::new(database_id, &record.collection), field, }, ); - dispatch(state, tenant_id, database_id, &record.collection, plan).await + teardown_now_or_at_commit(state, tenant_id, database_id, &record.collection, plan).await } /// Remove the vector index's durable build parameters and its Data Plane @@ -144,30 +179,61 @@ async fn vector( // WAL first: the `VectorParams` record that created this index is still // in the log, so without a durable drop record a restart rebuilds the // index the user just dropped. - let vshard = - crate::types::VShardId::from_collection_in_database(database_id, &record.collection); - let appended = crate::control::server::wal_dispatch::wal_append_if_write( - &state.wal, + // + // The record's outcome-floor window opens before the append and closes + // from the drop's outcome. + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &record.collection).vshard(); + let owner = RecordOwner { tenant_id, - vshard, database_id, + vshard_id: vshard, + }; + let minted = MintedRecords::open(&state.outcome_floor); + let appended = match minted.append_plan( + &state.wal, + owner, &plan, - ) - .map_err(|e| err("XX000", format!("persist vector index drop to WAL: {e}")))?; + // The drop is dispatched as a client statement. + crate::event::EventSource::User, + ) { + Ok(appended) => appended, + Err(e) => { + // Any record appended before the error never reaches a core. + minted.cancel(&state.wal, owner, 0).await.map_err(|c| { + DdlError::from_error_in_context("cancel vector index drop record", &c) + })?; + return Err(DdlError::from_error_in_context( + "persist vector index drop to WAL", + &e, + )); + } + }; // An append only buffers. The records this drop cancels were already // fsynced by the writes that acked them, so a buffered-only drop is lost on // restart while replay still rebuilds the index from those records. - let lsn = appended - .lsn - .ok_or_else(|| err("XX000", "vector index drop minted no WAL record"))?; - state - .wal - .wait_durable(lsn) - .await - .map_err(|e| err("XX000", format!("fsync vector index drop: {e}")))?; + let Some(lsn) = appended.lsn else { + minted.settle(); + return Err(DdlError::internal("vector index drop minted no WAL record")); + }; + if let Err(e) = state.wal.wait_durable(lsn).await { + // The record can still be on disk, so restart replay can reach it. + minted.hold(); + return Err(DdlError::from_error_in_context( + "fsync vector index drop", + &e, + )); + } - dispatch(state, tenant_id, database_id, &record.collection, plan).await + dispatch( + state, + tenant_id, + database_id, + &record.collection, + plan, + Some(minted), + ) + .await } /// Reset the collection's FTS binding once its last full-text index is gone. @@ -191,7 +257,7 @@ async fn fulltext( tenant_id.as_u64(), &record.collection, ) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .into_iter() .filter(|r| r.kind == IndexKind::FullText && r.name != record.name) .count(); @@ -206,37 +272,73 @@ async fn fulltext( fuzzy_default: Some(false), }, ); - dispatch(state, tenant_id, database_id, &record.collection, plan).await + teardown_now_or_at_commit(state, tenant_id, database_id, &record.collection, plan).await } -/// Dispatch one teardown plan to the Data Plane, surfacing both transport and -/// handler-side failures. -async fn dispatch( +/// Run one teardown plan now, or at COMMIT inside an explicit transaction: +/// the catalog entry that drops the index is buffered, so its engine state +/// must survive a ROLLBACK. +async fn teardown_now_or_at_commit( state: &SharedState, tenant_id: TenantId, database_id: DatabaseId, collection: &str, plan: crate::bridge::envelope::PhysicalPlan, ) -> Result<(), DdlError> { - let vshard = crate::types::VShardId::from_collection_in_database(database_id, collection); - let response = crate::control::server::dispatch_utils::dispatch_to_data_plane( - state, + let deferred = ddl_buffer::defer_effect(DeferredDdlEffect::IndexTeardown { tenant_id, database_id, - vshard, - plan, - TraceId::ZERO, - ) - .await - .map_err(|e| err("XX000", format!("index teardown dispatch failed: {e}")))?; + collection: collection.to_string(), + plan: plan.clone(), + }); + if deferred { + return Ok(()); + } + dispatch(state, tenant_id, database_id, collection, plan, None).await +} + +/// Dispatch one teardown plan to the Data Plane, surfacing both transport and +/// handler-side failures. `minted` holds the record appended for the plan; +/// the funnel closes its outcome-floor window from the plan's outcome. +pub(crate) async fn dispatch( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection: &str, + plan: crate::bridge::envelope::PhysicalPlan, + minted: Option, +) -> Result<(), DdlError> { + let vshard = nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(); + let response = + crate::control::server::dispatch_utils::dispatch_trusted_internal_write_to_data_plane( + state, + crate::control::server::dispatch_utils::WriteDispatch { + tenant_id, + database_id, + vshard_id: vshard, + plan, + trace_id: TraceId::ZERO, + event_source: crate::event::EventSource::User, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + minted, + }, + ) + .await + .map_err(|e| DdlError::from_error_in_context("index teardown dispatch failed", &e))?; if response.status == crate::bridge::envelope::Status::Error { - let detail = match response.error_code.as_deref() { - Some(crate::bridge::envelope::ErrorCode::Internal { detail, .. }) => detail.clone(), - Some(other) => format!("{other:?}"), - None => String::from_utf8_lossy(&response.payload).into_owned(), - }; - return Err(err("XX000", format!("index teardown failed: {detail}"))); + return Err(match response.error_code.as_deref() { + Some(code) => DdlError::from_error_in_context( + "index teardown failed", + &crate::Error::DataPlane(code.clone()), + ), + None => DdlError::internal(format!( + "index teardown failed: {}", + String::from_utf8_lossy(&response.payload) + )), + }); } Ok(()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index_fanout.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index_fanout.rs index 98df897c4..6ae018608 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index_fanout.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index_fanout.rs @@ -30,10 +30,6 @@ use nodedb_physical::physical_plan::wire as plan_wire; use super::super::super::result::DdlError; -fn err(sqlstate: &str, message: impl Into) -> DdlError { - DdlError::new(sqlstate, message) -} - /// Remaining budget for per-peer RPCs. Chosen to cover backfill on /// collections with up to ~1M rows at the Data Plane's current /// throughput; large production collections will need a streaming @@ -42,9 +38,9 @@ const PEER_BACKFILL_DEADLINE: Duration = Duration::from_secs(120); /// Run `DocumentOp::BackfillIndex` on every cluster node other than /// this coordinator. Returns `Ok(())` only when every peer reports -/// success; any peer failure is returned as a DDL error with -/// SQLSTATE 23505 for duplicates and XX000 otherwise, matching the -/// single-node path. +/// success. A peer's typed refusal keeps its SQLSTATE, so a duplicate +/// key is `23505` as on the single-node path. A transport fault is +/// `XX000`. /// /// Single-node clusters (no peers) return `Ok(())` immediately — the /// coordinator's local dispatch already covered everything. @@ -106,8 +102,8 @@ pub(super) async fn backfill_on_peers( case_insensitive: args.case_insensitive, predicate: args.predicate.map(str::to_string), }); - let plan_bytes = - plan_wire::encode(&plan).map_err(|e| err("XX000", format!("backfill plan encode: {e}")))?; + let plan_bytes = plan_wire::encode(&plan) + .map_err(|e| DdlError::internal(format!("backfill plan encode: {e}")))?; // Fan out in parallel; collect per-peer outcomes. Any failure // aborts the commit — we do NOT compensate by dropping the index @@ -139,27 +135,20 @@ pub(super) async fn backfill_on_peers( for join in joins { let (node_id, outcome) = join .await - .map_err(|e| err("XX000", format!("peer backfill join: {e}")))?; + .map_err(|e| DdlError::internal(format!("peer backfill join: {e}")))?; let resp = outcome.map_err(|e| { - err( - "XX000", - format!("peer backfill transport to node {node_id}: {e}"), - ) + DdlError::internal(format!("peer backfill transport to node {node_id}: {e}")) })?; let RaftRpc::ExecuteResponse(resp) = resp else { - return Err(err( - "XX000", - format!("peer backfill on node {node_id}: unexpected RPC variant {resp:?}"), - )); + return Err(DdlError::internal(format!( + "peer backfill on node {node_id}: unexpected RPC variant {resp:?}" + ))); }; if let Some(e) = resp.error { - let detail = format!("peer backfill on node {node_id}: {e:?}"); - let code = if detail.to_lowercase().contains("unique") { - "23505" - } else { - "XX000" - }; - return Err(err(code, detail)); + return Err(DdlError::from_error_in_context( + &format!("peer backfill on node {node_id}"), + &crate::Error::from(e), + )); } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/purge/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/purge/dispatch.rs index b8b550691..acf1db706 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/purge/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/purge/dispatch.rs @@ -52,7 +52,7 @@ pub async fn dispatch_unregister_collection( // cannot resolve it, so the files are never orphaned. let homing_core = dispatcher .router() - .resolve(VShardId::from_collection_in_database(database, name)) + .resolve(nodedb_types::CollectionKey::from_bare(database, name).vshard()) .unwrap_or(0); for core_id in 0..num_cores { let request_id = state.next_request_id(); @@ -63,7 +63,11 @@ pub async fn dispatch_unregister_collection( vshard_id: VShardId::new(core_id as u32), plan: PhysicalPlan::Meta(MetaOp::UnregisterCollection { tenant_id, - name: name.to_string(), + // The Data Plane keys the collection's state by its + // database-qualified name outside the default database. + name: nodedb_types::QualifiedCollection::new(database, name) + .as_str() + .to_string(), purge_lsn, reclaim_l1_files: core_id == homing_core, }), @@ -105,14 +109,17 @@ pub async fn dispatch_unregister_collection( .ok_or_else(|| crate::Error::Dispatch { detail: format!("collection reclaim channel closed on core {core_id}"), })?; + // A coded refusal keeps its Data-Plane code. if response.status != Status::Ok { - return Err(crate::Error::Storage { - engine: "collection-purge".into(), - detail: format!( - "UnregisterCollection for tenant {tenant_id} collection '{name}' \ - failed on core {core_id}: {:?}", - response.error_code - ), + return Err(match response.error_code { + Some(code) => crate::Error::DataPlane(*code), + None => crate::Error::Storage { + engine: "collection-purge".into(), + detail: format!( + "UnregisterCollection for tenant {tenant_id} collection '{name}' \ + failed on core {core_id} with no error code" + ), + }, }); } Ok(()) diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/show_indexes.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/show_indexes.rs index 199757094..18f7c0904 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/show_indexes.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/show_indexes.rs @@ -55,7 +55,7 @@ pub fn show_indexes( .credentials .catalog() .list_index_records(database_id.as_u64(), tenant_id.as_u64()) - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; records.retain(StoredIndexRecord::is_visible); if let Some(collection) = filter_collection.as_deref() { records.retain(|r| r.collection == collection); diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/undrop.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/undrop.rs index 5122fd97a..feeac7c9b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/undrop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/undrop.rs @@ -76,7 +76,7 @@ pub fn undrop_collection( )); } Err(e) => { - return Err(DdlError::new("XX000", e.to_string())); + return Err(DdlError::from_error(&e)); } }; if stored.is_active { @@ -133,7 +133,7 @@ pub fn undrop_collection( let entry = crate::control::catalog_entry::CatalogEntry::PutCollection(Box::new(stored.clone())); let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if outcome.needs_local_apply() { // Single-node fallback: run the same applier the replicated path runs // on every node, so the restore carries every invariant of a @@ -141,7 +141,7 @@ pub fn undrop_collection( // visibility of the indexes the soft-delete hid. Writing the row // directly here restored a collection whose indexes stayed hidden. crate::control::catalog_entry::apply::collection::put(&stored, catalog) - .map_err(|e| DdlError::new("XX000", format!("catalog restore failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog restore failed", &e))?; } let completion = UndropAuditDetail::new(name, UndropStage::Completed, owner_user_missing) diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/vector_metadata.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/vector_metadata.rs index e77c3a0ba..a2779e14e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/vector_metadata.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/vector_metadata.rs @@ -178,7 +178,7 @@ pub fn handle_show_vector_models( let entries = catalog .list_vector_models(database_id.as_u64(), tenant_id) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let columns = vec![ "collection".to_string(), @@ -235,7 +235,7 @@ pub fn handle_vector_metadata_query( let entry = catalog .get_vector_model(database_id.as_u64(), tenant_id, collection, column) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let json = match entry { Some(e) => { diff --git a/nodedb/src/control/server/shared/ddl/neutral/column_default.rs b/nodedb/src/control/server/shared/ddl/neutral/column_default.rs index d7c07eb6f..179145088 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/column_default.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/column_default.rs @@ -105,8 +105,8 @@ pub(super) fn validate_column_default( /// The expression is classified and parsed, never evaluated, so a /// `DEFAULT nextval('s')` column never advances its sequence at DDL time. /// -/// An unregistered function name raises SQLSTATE `42883`; every other -/// rejection raises SQLSTATE `42601`. +/// An unregistered function name raises SQLSTATE `42883`, a type error +/// `42804`, a value out of range `22003`, and a statement error `42601`. pub(super) fn validate_clause_expr(clause: &str, owner: &str, expr: &str) -> Result<(), DdlError> { nodedb_sql::planner::defaults::validate_default_expr(expr, owner) .map_err(|error| clause_error(clause, owner, &error)) @@ -117,10 +117,41 @@ fn clause_error(clause: &str, owner: &str, error: &SqlError) -> DdlError { let sqlstate = match error { SqlError::UndefinedFunction { .. } => sqlstate::UNDEFINED_FUNCTION, SqlError::TypeMismatch { .. } => sqlstate::DATATYPE_MISMATCH, - SqlError::IntegerOutOfRange { .. } | SqlError::FloatOutOfRange { .. } => { - sqlstate::NUMERIC_VALUE_OUT_OF_RANGE + SqlError::IntegerOutOfRange { .. } + | SqlError::FloatOutOfRange { .. } + | SqlError::ConstantOverflow { .. } => sqlstate::NUMERIC_VALUE_OUT_OF_RANGE, + SqlError::DivisionByZero => sqlstate::DIVISION_BY_ZERO, + SqlError::DataException { .. } => sqlstate::DATA_EXCEPTION, + SqlError::InvalidLimitValue { .. } => sqlstate::INVALID_LIMIT_VALUE, + SqlError::UnknownTable { .. } | SqlError::CollectionDeactivated { .. } => { + sqlstate::UNDEFINED_TABLE } - _ => sqlstate::SYNTAX_ERROR, + SqlError::UnknownColumn { .. } => sqlstate::UNDEFINED_COLUMN, + SqlError::AmbiguousColumn { .. } => sqlstate::AMBIGUOUS_COLUMN, + SqlError::UndefinedObject { .. } => sqlstate::UNDEFINED_OBJECT, + SqlError::ObjectNotInPrerequisiteState { .. } => sqlstate::OBJECT_NOT_IN_PREREQUISITE_STATE, + SqlError::SequencePerRowUnsupported { .. } + | SqlError::SearchFunctionOutsideSearch { .. } + | SqlError::UnsupportedConstraint { .. } + | SqlError::ConflictingEngineClause { .. } => sqlstate::FEATURE_NOT_SUPPORTED, + SqlError::RetryableSchemaChanged { .. } => sqlstate::SERIALIZATION_FAILURE, + SqlError::RecursionDepthExceeded { .. } => sqlstate::PROGRAM_LIMIT_EXCEEDED, + SqlError::Parse { .. } + | SqlError::Arity { .. } + | SqlError::Unsupported { .. } + | SqlError::UnevaluableDefault { .. } + | SqlError::SetvalInColumnDefault { .. } + | SqlError::InvalidFunction { .. } + | SqlError::InvalidWindowFrame { .. } + | SqlError::MissingField { .. } + | SqlError::InsertColumnArityMismatch { .. } + | SqlError::PositionalKvInsertUnsupported { .. } + | SqlError::InvalidIdentifier { .. } + | SqlError::ReservedIdentifier { .. } + | SqlError::InvalidRecursiveSetOp { .. } + | SqlError::InvalidRecursiveSelfRef { .. } + | SqlError::RecursiveColumnMismatch { .. } + | SqlError::DuplicateRecursiveColumn { .. } => sqlstate::SYNTAX_ERROR, }; DdlError::new( sqlstate, diff --git a/nodedb/src/control/server/shared/ddl/neutral/conflict_policy.rs b/nodedb/src/control/server/shared/ddl/neutral/conflict_policy.rs index daaa8571f..20ce9e541 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/conflict_policy.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/conflict_policy.rs @@ -60,14 +60,14 @@ pub async fn alter_set_on_conflict( let catalog = state.credentials.catalog(); let mut coll = catalog .get_collection(database_id, tenant_id, collection) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", format!("collection '{collection}' not found")))?; // Step 1: read the durable policy, falling back to the same ephemeral // default the in-memory `PolicyRegistry` uses for an unregistered // collection. let mut policy: CollectionPolicy = match &coll.conflict_policy { - Some(json) => sonic_rs::from_str(json).map_err(|e| err("XX000", e.to_string()))?, + Some(json) => sonic_rs::from_str(json).map_err(|e| DdlError::internal(e.to_string()))?, None => CollectionPolicy::ephemeral(), }; @@ -76,7 +76,8 @@ pub async fn alter_set_on_conflict( apply_conflict_policy(&mut policy, constraint_kind, new_conflict_policy); // Step 3: persist on the catalog record and re-broadcast. - let policy_json = sonic_rs::to_string(&policy).map_err(|e| err("XX000", e.to_string()))?; + let policy_json = + sonic_rs::to_string(&policy).map_err(|e| DdlError::internal(e.to_string()))?; coll.conflict_policy = Some(policy_json); let entry = CatalogEntry::PutCollection(Box::new(coll)); propose_and_apply(state, &entry)?; @@ -105,14 +106,14 @@ pub async fn show_conflict_policy( let catalog = state.credentials.catalog(); let coll = catalog .get_collection(database_id, tenant_id, collection) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", format!("collection '{collection}' not found")))?; let policy: CollectionPolicy = match &coll.conflict_policy { - Some(json) => sonic_rs::from_str(json).map_err(|e| err("XX000", e.to_string()))?, + Some(json) => sonic_rs::from_str(json).map_err(|e| DdlError::internal(e.to_string()))?, None => CollectionPolicy::ephemeral(), }; - let text = sonic_rs::to_string(&policy).map_err(|e| err("XX000", e.to_string()))?; + let text = sonic_rs::to_string(&policy).map_err(|e| DdlError::internal(e.to_string()))?; let mut row = Map::new(); row.insert("policy".to_string(), JsonValue::String(text)); diff --git a/nodedb/src/control/server/shared/ddl/neutral/constraint/handlers.rs b/nodedb/src/control/server/shared/ddl/neutral/constraint/handlers.rs index 8c4eb3ebe..164697490 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/constraint/handlers.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/constraint/handlers.rs @@ -63,7 +63,7 @@ pub fn add_state_constraint( let mut coll = catalog .get_collection(DatabaseId::DEFAULT, tenant_id, &coll_name) - .map_err(|e| err("XX000", &e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", &format!("collection '{coll_name}' not found")))?; if coll @@ -79,7 +79,7 @@ pub fn add_state_constraint( coll.state_constraints.push(def); persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) - .map_err(|e| err("XX000", &e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); @@ -127,7 +127,7 @@ pub fn add_transition_check( let mut coll = catalog .get_collection(DatabaseId::DEFAULT, tenant_id, &coll_name) - .map_err(|e| err("XX000", &e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", &format!("collection '{coll_name}' not found")))?; if coll.transition_checks.iter().any(|c| c.name == check_name) { @@ -139,7 +139,7 @@ pub fn add_transition_check( coll.transition_checks.push(def); persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) - .map_err(|e| err("XX000", &e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); @@ -203,7 +203,7 @@ pub fn add_check_constraint( let mut coll = catalog .get_collection(DatabaseId::DEFAULT, tenant_id, &coll_name) - .map_err(|e| err("XX000", &e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", &format!("collection '{coll_name}' not found")))?; if coll @@ -227,7 +227,7 @@ pub fn add_check_constraint( coll.check_constraints.push(def); persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) - .map_err(|e| err("XX000", &e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); @@ -263,7 +263,7 @@ pub fn drop_constraint( let mut coll = catalog .get_collection(DatabaseId::DEFAULT, tenant_id, &coll_name) - .map_err(|e| err("XX000", &e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", &format!("collection '{coll_name}' not found")))?; let before_state = coll.state_constraints.len(); @@ -285,7 +285,7 @@ pub fn drop_constraint( } persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) - .map_err(|e| err("XX000", &e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); diff --git a/nodedb/src/control/server/shared/ddl/neutral/constraint/show.rs b/nodedb/src/control/server/shared/ddl/neutral/constraint/show.rs index 995fed066..28e2e1cc7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/constraint/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/constraint/show.rs @@ -38,7 +38,7 @@ pub fn show_constraints( let tenant_id = identity.tenant_id.as_u64(); let coll = catalog .get_collection(DatabaseId::DEFAULT, tenant_id, &coll_name) - .map_err(|e| err("XX000", &e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", &format!("collection '{coll_name}' not found")))?; let columns = vec![ diff --git a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/commit.rs b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/commit.rs index d496380c3..33d1d705c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/commit.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/commit.rs @@ -157,7 +157,8 @@ pub async fn commit_offset( ) .map_err(|e| match e { crate::Error::OffsetRegression { .. } => DdlError::new("22023", e.to_string()), - _ => DdlError::new("XX000", format!("offset commit: {e}")), + // Any other error keeps the class the SQLSTATE table gives it. + other => DdlError::from_error_in_context("offset commit", &other), })?; return Ok(status("COMMIT OFFSET")); @@ -245,7 +246,7 @@ pub async fn commit_offset( partition_id, offset, ) - .map_err(|e| DdlError::new("XX000", format!("offset commit: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("offset commit", &e))?; } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/create.rs b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/create.rs index 267634371..d88fd4a86 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/create.rs @@ -99,7 +99,7 @@ pub async fn create_consumer_group( let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| DdlError::new("XX000", "system clock error"))? + .map_err(|_| DdlError::internal("system clock error"))? .as_secs(); let def = ConsumerGroupDef { diff --git a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/identity.rs b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/identity.rs index d98822d3b..921b01586 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/identity.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/identity.rs @@ -71,7 +71,7 @@ pub fn migrate_legacy_topic_group( canonical_stream, group, ) - .map_err(|error| DdlError::new("XX000", format!("consumer-group migration: {error}")))?; + .map_err(|error| DdlError::from_error_in_context("consumer-group migration", &error))?; super::replicate::propose_migrate(state, &def, legacy_stream)?; Ok(true) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/replicate.rs b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/replicate.rs index 2827161ab..2e78a053d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/replicate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/replicate.rs @@ -22,7 +22,7 @@ pub(super) fn propose_create(state: &SharedState, def: &ConsumerGroupDef) -> Res let entry = CatalogEntry::PutConsumerGroupIfAbsent(Box::new(def.clone())); propose_and_apply(state, &entry, || { apply::put_if_absent(def, state.credentials.catalog()) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; post_apply::put_if_absent(def, state); Ok(()) }) @@ -52,7 +52,7 @@ pub(super) fn propose_delete( name, state.credentials.catalog(), ) - .map_err(|e| DdlError::new("XX000", format!("catalog delete: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; post_apply::delete(database_id, tenant_id, stream_name, name, state); Ok(()) }) @@ -73,7 +73,7 @@ pub(super) fn propose_migrate( }; propose_and_apply(state, &entry, || { apply::migrate_stream(def, legacy_stream, state.credentials.catalog()) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; post_apply::migrate_stream(def, legacy_stream, state); Ok(()) }) diff --git a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs index fd436a472..059fd935e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs @@ -144,7 +144,7 @@ pub async fn create_continuous_aggregate( // fields — the def is decoded on register dispatch in // `post_apply::async_dispatch::continuous_aggregate::put_async`. let def_bytes = zerompk::to_msgpack_vec(&def) - .map_err(|e| err("XX000", format!("serialize continuous aggregate def: {e}")))?; + .map_err(|e| DdlError::internal(format!("serialize continuous aggregate def: {e}")))?; let stored = StoredContinuousAggregate { database_id: database_id.as_u64(), @@ -226,7 +226,7 @@ pub async fn create_continuous_aggregate( propose_and_apply(state, &coll_entry)?; collection::dispatch_register_from_stored(state, &target) .await - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; } // Single-node / no-applier path: the async post-apply dispatcher @@ -249,14 +249,13 @@ pub async fn create_continuous_aggregate( sync_dispatch::SystemTask::new( sync_dispatch::SystemReason::CatalogMaintenance, tenant_id, - database_id, - &def.source, + nodedb_types::CollectionKey::from_bare(database_id, &def.source), plan, ), Duration::from_secs(5), ) .await - .map_err(|e| err("XX000", format!("dispatch failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("dispatch failed", &e))?; } tracing::info!( diff --git a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs index 599ecd20a..54a6c77a6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs @@ -75,7 +75,7 @@ pub async fn drop_continuous_aggregate( .credentials .catalog() .get_continuous_aggregate(database_id.as_u64(), tenant_id.as_u64(), &name) - .map_err(|e| err("XX000", format!("catalog read: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog read", &e))? .ok_or_else(|| { err( "42704", @@ -114,14 +114,13 @@ pub async fn drop_continuous_aggregate( sync_dispatch::SystemTask::new( sync_dispatch::SystemReason::CatalogMaintenance, tenant_id, - database_id, - &stored.source, + nodedb_types::CollectionKey::from_bare(database_id, &stored.source), plan, ), Duration::from_secs(5), ) .await - .map_err(|e| err("XX000", format!("dispatch failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("dispatch failed", &e))?; } tracing::info!(name, "continuous aggregate dropped"); diff --git a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/register.rs b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/register.rs index a87d441ee..c93d283b6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/register.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/register.rs @@ -50,8 +50,10 @@ pub async fn register_persisted_continuous_aggregates(state: &SharedState) { sync_dispatch::SystemTask::new( sync_dispatch::SystemReason::CatalogMaintenance, tenant_id, - crate::types::DatabaseId::new(def.database_id), - &def.source, + nodedb_types::CollectionKey::from_bare( + crate::types::DatabaseId::new(def.database_id), + &def.source, + ), plan, ), Duration::from_secs(5), diff --git a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/show.rs b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/show.rs index 2a4a098da..b590371cc 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/show.rs @@ -77,8 +77,7 @@ pub async fn show_continuous_aggregates( sync_dispatch::SystemTask::new( sync_dispatch::SystemReason::CatalogMaintenance, tenant_id, - database_id, - "__system", + nodedb_types::CollectionKey::from_bare(database_id, "__system"), PhysicalPlan::Meta(MetaOp::ListContinuousAggregates), ), Duration::from_secs(5), @@ -87,7 +86,7 @@ pub async fn show_continuous_aggregates( { Ok(payload) => { crate::data::executor::response_codec::decode_payload(&payload).map_err(|e| { - DdlError::new("XX000", format!("continuous aggregate runtime stats: {e}")) + DdlError::from_error_in_context("continuous aggregate runtime stats", &e) })? } Err(_) => Vec::new(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs b/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs index 66fb8f94d..a2fb189cb 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs @@ -11,11 +11,12 @@ use std::time::Duration; use sonic_rs; -use crate::bridge::envelope::PhysicalPlan; +use crate::bridge::envelope::{PhysicalPlan, Status}; use crate::control::catalog_entry::persist_collection_replicated; use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::pgwire::types::error_to_sqlstate; use crate::control::server::shared::ddl::sync_dispatch::{ - SystemReason, SystemTask, dispatch_system, + SystemReason, SystemTask, dispatch_system_response_with_source, }; use crate::control::state::SharedState; use nodedb_physical::physical_plan::MetaOp; @@ -40,7 +41,7 @@ pub async fn convert_collection( let mut coll = catalog .get_collection(database_id, tenant_id.as_u64(), &collection) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", format!("collection '{collection}' does not exist")))?; // Build columns before dispatch — needed for both Data Plane and catalog. @@ -62,7 +63,8 @@ pub async fn convert_collection( }; let schema_json_for_dp = if let Some(ref cols) = columns { - sonic_rs::to_string(cols).map_err(|e| err("XX000", format!("schema serialization: {e}")))? + sonic_rs::to_string(cols) + .map_err(|e| DdlError::internal(format!("schema serialization: {e}")))? } else { String::new() }; @@ -96,19 +98,38 @@ pub async fn convert_collection( source_storage_mode, }); - dispatch_system( + let event_source = SystemReason::DdlApply.event_source(); + let resp = dispatch_system_response_with_source( state, SystemTask::new( SystemReason::DdlApply, tenant_id, - database_id, - &collection, + nodedb_types::CollectionKey::from_bare(database_id, &collection), plan, ), Duration::from_secs(60), + event_source, ) .await - .map_err(|e| err("XX000", format!("conversion failed: {e}")))?; + .map_err(|e| { + let (_, code, message) = error_to_sqlstate(&e); + err(code, format!("conversion failed: {message}")) + })?; + + // A Data-Plane verdict (a row breaking the target schema, say) keeps its + // own typed SQLSTATE instead of collapsing to a generic internal error. + if resp.status != Status::Ok { + let verdict = match resp.error_code { + Some(code) => crate::Error::DataPlane(*code), + None => crate::Error::Internal { + detail: "conversion failed: data plane returned an error status with no error \ + code" + .into(), + }, + }; + let (_, code, message) = error_to_sqlstate(&verdict); + return Err(err(code, message)); + } // Update catalog collection type. let new_type = match target_type.as_str() { @@ -161,7 +182,7 @@ pub async fn convert_collection( } persist_collection_replicated(state, database_id, &coll) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; // Refresh this node's Data Plane `doc_configs` entry to the NEW storage // mode. Without this, every later read of the collection resolves its @@ -170,7 +191,7 @@ pub async fn convert_collection( state, &coll, ) .await - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; tracing::info!( %collection, diff --git a/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs b/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs index 832adb071..73f460d31 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs @@ -96,7 +96,7 @@ pub async fn crdt_state( }, ) .await - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let columns = vec!["crdt_state".to_string()]; @@ -162,8 +162,12 @@ pub async fn crdt_apply( let surrogate = state .surrogate_assigner - .assign(database_id, tenant_id, collection, document_id.as_bytes()) - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .assign( + nodedb_types::CollectionKey::from_bare(database_id, collection), + tenant_id, + document_id.as_bytes(), + ) + .map_err(|e| DdlError::from_error(&e))?; let plan = PhysicalPlan::Crdt(CrdtOp::Apply { collection: nodedb_types::QualifiedCollection::new(database_id, collection), @@ -179,7 +183,7 @@ pub async fn crdt_apply( }); let task = PhysicalTask { tenant_id, - vshard_id: crate::types::VShardId::from_collection_in_database(database_id, collection), + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(), database_id, plan, post_set_op: PostSetOp::None, @@ -196,7 +200,7 @@ pub async fn crdt_apply( .into_tasks() .into_iter() .next() - .ok_or_else(|| DdlError::new("XX000", "authorization returned no capability"))?; + .ok_or_else(|| DdlError::internal("authorization returned no capability"))?; // Route through the Raft proposer gate so the delta is quorum-durable under // replication. A local-only dispatch would land the delta on the receiving @@ -219,14 +223,14 @@ pub async fn crdt_apply( state, crate::control::crdt_admission::AuthorizedCrdtApplyAdmissionRequest { authorized, - collection, + collection: &qualified_collection, timeout: Duration::from_secs(state.tuning.network.default_deadline_secs), event_source: crate::event::EventSource::User, policy: &policy, }, ) .await - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let columns = vec!["result".to_string()]; let mut row = Map::new(); diff --git a/nodedb/src/control/server/shared/ddl/neutral/custom_type.rs b/nodedb/src/control/server/shared/ddl/neutral/custom_type.rs index f23ea6683..837ccf027 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/custom_type.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/custom_type.rs @@ -185,11 +185,11 @@ pub fn drop_type( name: name.to_string(), }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| err("XX000", &format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { catalog .delete_custom_type(database_id_u64, tenant_id, name) - .map_err(|e| err("XX000", &format!("catalog delete: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; } state @@ -289,11 +289,11 @@ fn persist_and_register(state: &SharedState, stored: StoredCustomType) -> Result let entry = crate::control::catalog_entry::CatalogEntry::PutCustomType(Box::new(stored.clone())); let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| err("XX000", &format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { let written = catalog .put_custom_type_assigning_oid(&stored) - .map_err(|e| err("XX000", &format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; state.custom_type_registry.register(written); } @@ -332,7 +332,7 @@ fn current_epoch_secs() -> Result { std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) .map(|d| d.as_secs()) - .map_err(|_| err("XX000", "system clock error")) + .map_err(|_| DdlError::internal("system clock error")) } fn type_summary(def: &CustomTypeDef) -> (String, String) { diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/alter.rs b/nodedb/src/control/server/shared/ddl/neutral/database/alter.rs index 980dec440..78232cc62 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/alter.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/alter.rs @@ -34,13 +34,15 @@ pub fn alter_database( let db_id = catalog .get_database_id_by_name(name) - .map_err(|e| ddl_err("XX000", format!("catalog lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))? .ok_or_else(|| ddl_err("3D000", format!("database '{name}' does not exist")))?; + // A name row whose descriptor is gone is a database a concurrent DROP + // removed between the two reads. let mut descriptor = catalog .get_database(db_id) - .map_err(|e| ddl_err("XX000", format!("catalog read failed: {e}")))? - .ok_or_else(|| ddl_err("XX000", format!("database '{name}' descriptor missing")))?; + .map_err(|e| DdlError::from_error_in_context("catalog read failed", &e))? + .ok_or_else(|| ddl_err("3D000", format!("database '{name}' does not exist")))?; match operation { AlterDatabaseOperation::Rename { new_name } => { @@ -61,7 +63,7 @@ pub fn alter_database( } Ok(_) => {} Err(e) => { - return Err(ddl_err("XX000", format!("catalog lookup failed: {e}"))); + return Err(DdlError::from_error_in_context("catalog lookup failed", &e)); } } descriptor.name = new_name.clone(); @@ -71,7 +73,7 @@ pub fn alter_database( || { catalog .put_database(&descriptor) - .map_err(|e| ddl_err("XX000", format!("catalog write failed: {e}"))) + .map_err(|e| DdlError::from_error_in_context("catalog write failed", &e)) }, )?; @@ -96,7 +98,7 @@ pub fn alter_database( // before/after diff so operators can reconstruct what changed. let before = catalog .get_database_quota(db_id) - .map_err(|e| ddl_err("XX000", format!("quota read failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("quota read failed", &e))? .unwrap_or(QuotaRecord::DEFAULT); let mut record = before.clone(); record.merge(spec); @@ -107,7 +109,7 @@ pub fn alter_database( let ceiling = state.quota_ceiling_snapshot(); catalog .check_database_quota(db_id, &record, &ceiling) - .map_err(|e| ddl_err("53400", format!("{e}")))?; + .map_err(|e| DdlError::from_error(&e))?; // Replicated: every node writes the row and installs the quota in // its live enforcement components via post-apply. @@ -120,7 +122,7 @@ pub fn alter_database( || { catalog .write_database_quota(db_id, &record) - .map_err(|e| ddl_err("53400", format!("{e}")))?; + .map_err(|e| DdlError::from_error(&e))?; crate::control::catalog_entry::post_apply::quota::put_database( db_id, &record, state, ); @@ -175,7 +177,7 @@ pub fn alter_database( || { catalog .put_database(&descriptor) - .map_err(|e| ddl_err("XX000", format!("catalog write failed: {e}"))) + .map_err(|e| DdlError::from_error_in_context("catalog write failed", &e)) }, )?; @@ -208,7 +210,7 @@ pub fn alter_database( || { catalog .put_database(&descriptor) - .map_err(|e| ddl_err("XX000", format!("catalog write failed: {e}"))) + .map_err(|e| DdlError::from_error_in_context("catalog write failed", &e)) }, )?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/backup_restore.rs b/nodedb/src/control/server/shared/ddl/neutral/database/backup_restore.rs index 967d89ff4..3a3950f5f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/backup_restore.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/backup_restore.rs @@ -35,7 +35,7 @@ pub fn backup_database( )); } Err(e) => { - return Err(ddl_err("XX000", format!("catalog lookup failed: {e}"))); + return Err(DdlError::from_error_in_context("catalog lookup failed", &e)); } }; require_database_owner_or_higher(state, identity, db_id, &format!("BACKUP DATABASE {name}"))?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/clone.rs b/nodedb/src/control/server/shared/ddl/neutral/database/clone.rs index ffd340982..8803683b9 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/clone.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/clone.rs @@ -52,7 +52,7 @@ pub async fn clone_database( // ── Resolve source database ─────────────────────────────────────────────── let source_db_id = catalog .get_database_id_by_name(params.source_name) - .map_err(|e| ddl_err("XX000", format!("catalog lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))? .ok_or_else(|| { ddl_err( "42P01", @@ -65,7 +65,7 @@ pub async fn clone_database( let source_descriptor = catalog .get_database(source_db_id) - .map_err(|e| ddl_err("XX000", format!("catalog read failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog read failed", &e))? .ok_or_else(|| { ddl_err( "42P01", @@ -92,7 +92,7 @@ pub async fn clone_database( // ── Enforce MAX_CLONE_DEPTH ──────────────────────────────────────────────── let depth = clone_chain_depth(state, source_db_id) - .map_err(|e| ddl_err("XX000", format!("clone depth check failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("clone depth check failed", &e))?; if depth >= MAX_CLONE_DEPTH { return Err(ddl_err( @@ -115,7 +115,7 @@ pub async fn clone_database( } Ok(None) => {} Err(e) => { - return Err(ddl_err("XX000", format!("catalog lookup failed: {e}"))); + return Err(DdlError::from_error_in_context("catalog lookup failed", &e)); } } @@ -129,7 +129,7 @@ pub async fn clone_database( // empty the WAL frontier is used as the best available approximation, // which is correct for recent timestamps (within the same server session). let now_ms = - current_wall_ms().map_err(|e| ddl_err("XX000", format!("clock read failed: {e}")))?; + current_wall_ms().map_err(|e| DdlError::from_error_in_context("clock read failed", &e))?; let (as_of_lsn, as_of_ms) = match params.as_of { CloneAsOf::Latest => (state.wal.next_lsn(), now_ms), CloneAsOf::SystemTimeMs(ms) => { @@ -175,7 +175,7 @@ pub async fn clone_database( }; let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| ddl_err("XX000", format!("catalog propose failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e))?; // Single-node fast path (`LocalOnly` means "no Raft, apply directly"). // @@ -187,32 +187,34 @@ pub async fn clone_database( if outcome.needs_local_apply() { catalog .add_clone_child(source_db_id, target_db_id) - .map_err(|e| ddl_err("XX000", format!("lineage write failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("lineage write failed", &e))?; if let Err(put_err) = catalog.put_database(&target_descriptor) { // Compensate: remove the lineage edge we just wrote. A failure here // is fatal — surface both errors so on-call can repair the catalog. if let Err(rb_err) = catalog.remove_clone_child(source_db_id, target_db_id) { - return Err(ddl_err( - "XX000", - format!( - "catalog write failed: {put_err}; \ - lineage rollback ALSO failed: {rb_err} — \ + return Err(DdlError::from_error_in_context( + &format!( + "lineage rollback ALSO failed: {rb_err} — \ catalog left with orphan lineage edge \ - (source={source_db_id}, target={target_db_id})", + (source={source_db_id}, target={target_db_id}); catalog write failed", ), + &put_err, )); } - return Err(ddl_err("XX000", format!("catalog write failed: {put_err}"))); + return Err(DdlError::from_error_in_context( + "catalog write failed", + &put_err, + )); } // Stamp every active source collection into the target database with // `cloned_from` set. This lets the SQL planner resolve collection // names against the clone without knowing about clone indirection; // CoW delegation happens at dispatch time. - let source_colls = catalog - .load_all_collections(source_db_id) - .map_err(|e| ddl_err("XX000", format!("clone: enumerate source collections: {e}")))?; + let source_colls = catalog.load_all_collections(source_db_id).map_err(|e| { + DdlError::from_error_in_context("clone: enumerate source collections", &e) + })?; let kv_surrogate_ceiling = Some(state.surrogate_assigner.current_hwm()); for mut coll in source_colls.into_iter().filter(|c| c.is_active) { coll.database_id = target_db_id; @@ -231,12 +233,12 @@ pub async fn clone_database( // the failure is the only way the caller learns the clone is // incomplete. catalog.put_collection(target_db_id, &coll).map_err(|e| { - ddl_err( - "XX000", - format!( - "clone: stamping shadow descriptor for collection '{}' failed: {e}", + DdlError::from_error_in_context( + &format!( + "clone: stamping shadow descriptor for collection '{}' failed", coll.name ), + &e, ) })?; @@ -251,12 +253,12 @@ pub async fn clone_database( owner_username: coll.owner.clone(), }; catalog.put_owner(&owner).map_err(|e| { - ddl_err( - "XX000", - format!( - "clone: stamping owner for collection '{}' failed: {e}", + DdlError::from_error_in_context( + &format!( + "clone: stamping owner for collection '{}' failed", coll.name ), + &e, ) })?; } @@ -272,7 +274,7 @@ pub async fn clone_database( // answers queries the source answers differently, and nothing later // re-copies the row. copy_database_metadata(catalog, source_db_id, target_db_id) - .map_err(|e| ddl_err("XX000", format!("clone: copying catalog metadata: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("clone: copying catalog metadata", &e))?; } // Synonym groups and custom types travel as proposed entries, not as a @@ -329,24 +331,14 @@ async fn copy_synonym_groups( let groups = catalog .load_synonym_groups_in_database(source.as_u64()) .map_err(|e| { - ddl_err( - "XX000", - format!("clone: enumerate source synonym groups: {e}"), - ) + DdlError::from_error_in_context("clone: enumerate source synonym groups", &e) })?; for mut group in groups { group.database_id = target.as_u64(); let entry = CatalogEntry::PutSynonymGroup(Box::new(group.clone())); - let outcome = propose_and_apply(state, &entry).map_err(|e| { - ddl_err( - "XX000", - format!( - "clone: copying synonym group '{}': {}", - group.name, e.message - ), - ) - })?; + let outcome = propose_and_apply(state, &entry) + .map_err(|e| e.in_context(&format!("clone: copying synonym group '{}'", group.name)))?; if outcome.needs_local_apply() { state.synonym_registry.register(group.clone()); crate::control::catalog_entry::post_apply::install_synonym_group(group, state).await; @@ -372,25 +364,17 @@ fn copy_custom_types( let catalog = state.credentials.catalog(); let types = catalog .load_custom_types_in_database(source.as_u64()) - .map_err(|e| { - ddl_err( - "XX000", - format!("clone: enumerate source custom types: {e}"), - ) - })?; + .map_err(|e| DdlError::from_error_in_context("clone: enumerate source custom types", &e))?; for mut custom_type in types { custom_type.database_id = target.as_u64(); custom_type.oid = UNASSIGNED_OID; let entry = CatalogEntry::PutCustomType(Box::new(custom_type.clone())); let outcome = propose_and_apply(state, &entry).map_err(|e| { - ddl_err( - "XX000", - format!( - "clone: copying custom type '{}': {}", - custom_type.name, e.message - ), - ) + e.in_context(&format!( + "clone: copying custom type '{}'", + custom_type.name + )) })?; if outcome.needs_local_apply() { register_written( @@ -430,12 +414,7 @@ fn clone_chain_depth(state: &SharedState, start_db_id: DatabaseId) -> crate::Res if depth > MAX_CLONE_DEPTH { return Ok(depth); } - let desc = catalog - .get_database(current) - .map_err(|e| crate::Error::Storage { - engine: "catalog".into(), - detail: format!("depth walk get_database failed: {e}"), - })?; + let desc = catalog.get_database(current)?; match desc.and_then(|d| d.parent_clone) { None => return Ok(depth), Some(parent) => { diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/create.rs b/nodedb/src/control/server/shared/ddl/neutral/database/create.rs index a4f761fe3..add19995c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/create.rs @@ -79,7 +79,7 @@ pub fn create_database( } Ok(None) => {} Err(e) => { - return Err(ddl_err("XX000", format!("catalog lookup failed: {e}"))); + return Err(DdlError::from_error_in_context("catalog lookup failed", &e)); } } @@ -112,14 +112,14 @@ pub fn create_database( state, &CatalogEntry::PutDatabase(Box::new(descriptor.clone())), ) - .map_err(|e| ddl_err("XX000", format!("catalog propose failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e))?; // Direct write for single-node mode (`LocalOnly`) or as a fallback // when the cluster is in mixed-version compat mode. if outcome.needs_local_apply() { catalog .put_database(&descriptor) - .map_err(|e| ddl_err("XX000", format!("catalog write failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write failed", &e))?; } // Flush the allocator hwm on the periodic threshold so restarts @@ -151,3 +151,57 @@ pub fn create_database( Ok(status("CREATE DATABASE")) } + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use nodedb_types::error::ErrorCode; + + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::control::security::identity::{DatabaseSet, Role}; + use crate::types::TenantId; + use crate::wal::WalManager; + + fn test_state() -> (tempfile::TempDir, Arc) { + let dir = tempfile::tempdir().expect("create test directory"); + let wal = Arc::new( + WalManager::open_for_testing(&dir.path().join("create-database.wal")) + .expect("open test WAL"), + ); + let (dispatcher, _data_sides) = Dispatcher::new(1, 64); + let state = SharedState::new(dispatcher, wal).expect("construct shared state"); + (dir, state) + } + + fn admin() -> AuthenticatedIdentity { + AuthenticatedIdentity::new_internal_service( + 0, + "create_database_test", + TenantId::new(1), + vec![Role::Superuser], + true, + None, + DatabaseSet::All, + ) + } + + /// CREATE DATABASE of a name already taken is `duplicate_database` + /// (`42P04`) with the already-exists code, never an internal error. + #[test] + fn creating_an_existing_database_is_a_duplicate_database() { + let (_dir, state) = test_state(); + let identity = admin(); + create_database(&state, &identity, "orders", false, &[]).expect("first create succeeds"); + + let err = create_database(&state, &identity, "orders", false, &[]) + .expect_err("a second create of the same name is refused"); + assert_eq!(err.sqlstate, "42P04", "{err:?}"); + assert_eq!(err.code, ErrorCode::ALREADY_EXISTS); + + let existing = create_database(&state, &identity, "orders", true, &[]) + .expect("IF NOT EXISTS on an existing name succeeds"); + assert_eq!(existing.len(), 1); + } +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/database/drop.rs index 3a7ee3b98..0931d9ae3 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/drop.rs @@ -47,7 +47,7 @@ pub fn drop_database( let db_id = match catalog .get_database_id_by_name(name) - .map_err(|e| ddl_err("XX000", format!("catalog lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))? { Some(id) => id, None => { @@ -92,7 +92,7 @@ pub fn drop_database( { let descriptor_for_mirror = catalog .get_database(db_id) - .map_err(|e| ddl_err("XX000", format!("catalog read failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog read failed", &e))?; if let Some(descriptor) = descriptor_for_mirror && let Some(origin) = descriptor.mirror_origin.as_ref() // Promoted mirrors are now standalone writable databases — the @@ -131,7 +131,7 @@ pub fn drop_database( // before proceeding. let dependent_ids = catalog .get_clone_children(db_id) - .map_err(|e| ddl_err("XX000", format!("lineage check failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("lineage check failed", &e))?; if !dependent_ids.is_empty() { if !cascade { @@ -164,12 +164,14 @@ pub fn drop_database( // Gated until per-engine row copy lands — surface `0A000` // (`feature_not_supported`) so clients know not to retry. crate::Error::BadRequest { detail } => ddl_err("0A000", detail), - other => ddl_err( - "XX000", - format!( - "force materialization of dependent clone {} failed: {other}", + // Any other error keeps the class the SQLSTATE table + // gives it. + other => DdlError::from_error_in_context( + &format!( + "force materialization of dependent clone {} failed", dep_id.as_u64() ), + &other, ), }, )?; @@ -179,7 +181,7 @@ pub fn drop_database( // ── Cascade: drop all collections ──────────────────────────────────────── let collections = catalog .load_all_collections(db_id) - .map_err(|e| ddl_err("XX000", format!("catalog scan failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog scan failed", &e))?; if !cascade && !collections.is_empty() { return Err(ddl_err( @@ -214,16 +216,16 @@ pub fn drop_database( db_id: db_id.as_u64(), }, ) - .map_err(|e| ddl_err("XX000", format!("catalog propose failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e))?; if outcome.needs_local_apply() { catalog .delete_database(db_id) - .map_err(|e| ddl_err("XX000", format!("catalog delete failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog delete failed", &e))?; // Single-node path: no applier runs, so this branch owns both the // quota row deletion and the live cap release. crate::control::catalog_entry::apply::quota::purge_database_scope(db_id.as_u64(), catalog) - .map_err(|e| ddl_err("XX000", format!("quota purge failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("quota purge failed", &e))?; crate::control::catalog_entry::post_apply::quota::release_database_scope(db_id, state); } @@ -258,13 +260,13 @@ fn drop_all_collections_in_database( catalog .delete_collection(db_id, coll.tenant_id, &coll.name) .map_err(|e| { - ddl_err( - "XX000", - format!( - "CASCADE DROP DATABASE {}: failed to delete collection '{}': {e}", + DdlError::from_error_in_context( + &format!( + "CASCADE DROP DATABASE {}: failed to delete collection '{}'", db_id.as_u64(), coll.name ), + &e, ) })?; } diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/materialize.rs b/nodedb/src/control/server/shared/ddl/neutral/database/materialize.rs index 1396cf711..ba30a415d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/materialize.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/materialize.rs @@ -33,7 +33,7 @@ pub fn alter_database_materialize( let db_id = catalog .get_database_id_by_name(name) - .map_err(|e| ddl_err("XX000", format!("catalog lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))? .ok_or_else(|| ddl_err("3D000", format!("database '{name}' does not exist")))?; require_database_owner_or_higher( @@ -55,9 +55,10 @@ pub fn alter_database_materialize( // per-engine bulk-copy implementation to land). force_materialize_blocking(db_id, state, catalog, Some(&handle)).map_err(|e| match e { crate::Error::BadRequest { detail } => ddl_err("0A000", detail), - other => ddl_err( - "XX000", - format!("clone materialization of '{name}' failed: {other}"), + // Any other error keeps the class the SQLSTATE table gives it. + other => DdlError::from_error_in_context( + &format!("clone materialization of '{name}' failed"), + &other, ), })?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/mirror/create.rs b/nodedb/src/control/server/shared/ddl/neutral/database/mirror/create.rs index d64c066e3..ebcd19e18 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/mirror/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/mirror/create.rs @@ -47,7 +47,7 @@ pub fn mirror_database( } Ok(None) => {} Err(e) => { - return Err(ddl_err("XX000", format!("catalog lookup failed: {e}"))); + return Err(DdlError::from_error_in_context("catalog lookup failed", &e)); } } @@ -110,12 +110,12 @@ pub fn mirror_database( state, &CatalogEntry::PutDatabase(Box::new(descriptor.clone())), ) - .map_err(|e| ddl_err("XX000", format!("catalog propose failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e))?; if outcome.needs_local_apply() { catalog .put_database(&descriptor) - .map_err(|e| ddl_err("XX000", format!("catalog write failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write failed", &e))?; } // Flush allocator hwm on threshold. diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/mirror/promote.rs b/nodedb/src/control/server/shared/ddl/neutral/database/mirror/promote.rs index 85bad9a70..6e120b02f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/mirror/promote.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/mirror/promote.rs @@ -34,7 +34,7 @@ pub fn promote_database( let db_id = catalog .get_database_id_by_name(name) - .map_err(|e| ddl_err("XX000", format!("catalog lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))? .ok_or_else(|| ddl_err("3D000", format!("database '{name}' does not exist")))?; // Gate after db_id resolution so the audit record carries the database id. @@ -45,10 +45,12 @@ pub fn promote_database( &format!("ALTER DATABASE {name} PROMOTE"), )?; + // A name row whose descriptor is gone is a database a concurrent DROP + // removed between the two reads. let mut descriptor = catalog .get_database(db_id) - .map_err(|e| ddl_err("XX000", format!("catalog read failed: {e}")))? - .ok_or_else(|| ddl_err("XX000", format!("database '{name}' descriptor missing")))?; + .map_err(|e| DdlError::from_error_in_context("catalog read failed", &e))? + .ok_or_else(|| ddl_err("3D000", format!("database '{name}' does not exist")))?; // Idempotent: if already promoted (or Active without any mirror_origin), // return success immediately. @@ -95,12 +97,12 @@ pub fn promote_database( state, &CatalogEntry::PutDatabase(Box::new(descriptor.clone())), ) - .map_err(|e| ddl_err("XX000", format!("catalog propose failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e))?; if outcome.needs_local_apply() { catalog .put_database(&descriptor) - .map_err(|e| ddl_err("XX000", format!("catalog write failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write failed", &e))?; } // The database is now writable. Clear the mirror-only catalog state so @@ -113,15 +115,15 @@ pub fn promote_database( // lineage (origin cluster, mode, last applied LSN at promotion). DROP // DATABASE relies on this cleanup having happened — see drop.rs. if let Err(e) = catalog.delete_mirror_collection_map(db_id) { - return Err(ddl_err( - "XX000", - format!("PROMOTE: failed to clear mirror_collection_map: {e}"), + return Err(DdlError::from_error_in_context( + "PROMOTE: failed to clear mirror_collection_map", + &e, )); } if let Err(e) = catalog.delete_mirror_lag(db_id) { - return Err(ddl_err( - "XX000", - format!("PROMOTE: failed to clear mirror_lag: {e}"), + return Err(DdlError::from_error_in_context( + "PROMOTE: failed to clear mirror_lag", + &e, )); } diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/mirror/show.rs b/nodedb/src/control/server/shared/ddl/neutral/database/mirror/show.rs index 25599022d..8ccc5f73f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/mirror/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/mirror/show.rs @@ -34,7 +34,7 @@ pub fn show_database_mirror_status( let all_databases = catalog .list_databases() - .map_err(|e| ddl_err("XX000", format!("catalog list failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog list failed", &e))?; let columns = vec![ "name".to_string(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/show.rs b/nodedb/src/control/server/shared/ddl/neutral/database/show.rs index 7c47eb155..a8d9acb9c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/show.rs @@ -17,7 +17,7 @@ use crate::control::state::SharedState; use super::super::super::result::{DdlError, DdlResult}; use super::gate::require_tenant_admin; -use super::support::{ddl_err, text_rows}; +use super::support::text_rows; /// Handle `SHOW DATABASES`. pub fn show_databases( @@ -30,7 +30,7 @@ pub fn show_databases( let databases = catalog .list_databases() - .map_err(|e| ddl_err("XX000", format!("catalog list failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog list failed", &e))?; let columns = vec![ "name".to_string(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/show_lineage.rs b/nodedb/src/control/server/shared/ddl/neutral/database/show_lineage.rs index 21334a02b..d72067aba 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/show_lineage.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/show_lineage.rs @@ -42,7 +42,7 @@ pub fn show_database_lineage( let start_id = catalog .get_database_id_by_name(name) - .map_err(|e| ddl_err("XX000", format!("catalog lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))? .ok_or_else(|| ddl_err("3D000", format!("database '{name}' does not exist")))?; // Walk the parent_clone chain, bounded by MAX_CLONE_DEPTH to prevent @@ -54,12 +54,12 @@ pub fn show_database_lineage( for _ in 0..max_hops { let desc = catalog .get_database(current_id) - .map_err(|e| ddl_err("XX000", format!("catalog read failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog read failed", &e))? .ok_or_else(|| { - ddl_err( - "XX000", - format!("database id {} descriptor missing", current_id.as_u64()), - ) + DdlError::internal(format!( + "database id {} descriptor missing", + current_id.as_u64() + )) })?; let (as_of_lsn, clone_created_at_lsn) = match &desc.parent_clone { diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/show_quota.rs b/nodedb/src/control/server/shared/ddl/neutral/database/show_quota.rs index 4edb16541..14ae8ca99 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/show_quota.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/show_quota.rs @@ -32,12 +32,12 @@ pub fn show_database_quota( let db_id = catalog .get_database_id_by_name(name) - .map_err(|e| ddl_err("XX000", format!("catalog lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))? .ok_or_else(|| ddl_err("3D000", format!("database '{name}' does not exist")))?; let record = catalog .get_database_quota(db_id) - .map_err(|e| ddl_err("XX000", format!("quota read failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("quota read failed", &e))? .unwrap_or(QuotaRecord::DEFAULT); let columns = vec![ diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/show_usage.rs b/nodedb/src/control/server/shared/ddl/neutral/database/show_usage.rs index cf1772332..9e8ffbb01 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/show_usage.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/show_usage.rs @@ -36,12 +36,12 @@ pub fn show_database_usage( let db_id = catalog .get_database_id_by_name(name) - .map_err(|e| ddl_err("XX000", format!("catalog lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))? .ok_or_else(|| ddl_err("3D000", format!("database '{name}' does not exist")))?; let record = catalog .get_database_quota(db_id) - .map_err(|e| ddl_err("XX000", format!("quota read failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("quota read failed", &e))? .unwrap_or(QuotaRecord::DEFAULT); // Pull live gauges from the system metrics registry. Dimensions without a diff --git a/nodedb/src/control/server/shared/ddl/neutral/deferred_effects.rs b/nodedb/src/control/server/shared/ddl/neutral/deferred_effects.rs new file mode 100644 index 000000000..5ce0d914a --- /dev/null +++ b/nodedb/src/control/server/shared/ddl/neutral/deferred_effects.rs @@ -0,0 +1,193 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Run the engine side effects a transaction's index DDL deferred to COMMIT. +//! +//! COMMIT calls [`run_deferred_effects`] once the buffered catalog entries +//! landed, in statement order. Each effect runs through the same function its +//! statement runs in autocommit, so a transactional index ends in the same +//! engine state as an autocommit one. + +use crate::control::server::shared::session::ddl_effect::DeferredDdlEffect; +use crate::control::state::SharedState; + +use super::super::result::DdlError; +use super::collection::index::build::build_secondary_index; +use super::collection::index::kv_index::drop_kv_index; +use super::collection::index::teardown; +use super::kv_sorted_index::SortedIndexTarget; +use super::kv_sorted_index::dispatch::register_in_engine; +use super::kv_sorted_index::drop_in_engine; + +/// Run every effect in order. The first failure stops the run and returns +/// its error: the effects before it applied, and the ones after it did not. +pub(crate) async fn run_deferred_effects( + state: &SharedState, + effects: Vec, +) -> crate::Result<()> { + for effect in effects { + let collection = effect_collection(&effect).to_string(); + run_one(state, effect) + .await + .map_err(|error| effect_error(&collection, error))?; + } + Ok(()) +} + +/// The collection an effect changes. +fn effect_collection(effect: &DeferredDdlEffect) -> &str { + match effect { + DeferredDdlEffect::SecondaryIndexBuild(build) => &build.collection, + DeferredDdlEffect::EngineApply { collection, .. } + | DeferredDdlEffect::IndexTeardown { collection, .. } + | DeferredDdlEffect::SortedIndexRegister { collection, .. } + | DeferredDdlEffect::SortedIndexDrop { collection, .. } + | DeferredDdlEffect::KvIndexDrop { collection, .. } => collection, + } +} + +async fn run_one(state: &SharedState, effect: DeferredDdlEffect) -> Result<(), DdlError> { + match effect { + DeferredDdlEffect::SecondaryIndexBuild(build) => build_secondary_index(state, &build).await, + DeferredDdlEffect::EngineApply { + tenant_id, + database_id, + collection, + plan, + context, + } => { + crate::control::server::shared::ddl::engine_apply::apply_in_engine( + state, + tenant_id, + database_id, + &collection, + plan, + &context, + ) + .await + } + DeferredDdlEffect::IndexTeardown { + tenant_id, + database_id, + collection, + plan, + } => teardown::dispatch(state, tenant_id, database_id, &collection, plan, None).await, + DeferredDdlEffect::SortedIndexRegister { + tenant_id, + database_id, + collection, + plan, + } => { + let target = SortedIndexTarget { + tenant_id, + database_id, + collection: &collection, + }; + register_in_engine(state, &target, plan, "CREATE SORTED INDEX") + .await + .map(|_| ()) + } + DeferredDdlEffect::SortedIndexDrop { + tenant_id, + database_id, + collection, + index_name, + } => { + let target = SortedIndexTarget { + tenant_id, + database_id, + collection: &collection, + }; + drop_in_engine(state, &target, &index_name).await + } + DeferredDdlEffect::KvIndexDrop { + tenant_id, + database_id, + collection, + field, + } => drop_kv_index(state, tenant_id, database_id, &collection, &field).await, + } +} + +/// The COMMIT error for a failed effect. It keeps the effect's SQLSTATE, +/// code, details and cause, the class an autocommit statement reports. The +/// message names the collection. +fn effect_error(collection: &str, error: DdlError) -> crate::Error { + crate::Error::from(error.in_context(&format!( + "index DDL on '{collection}' committed, but its engine step failed" + ))) +} + +#[cfg(test)] +mod tests { + use nodedb_types::error::{ErrorCode, sqlstate}; + + use super::*; + use crate::control::server::native::dispatch::native_error_fields; + use crate::control::server::pgwire::types::error_to_sqlstate; + + /// A failed effect reports its own SQLSTATE and code at COMMIT, on + /// pgwire and native, with the collection in the message. + #[test] + fn a_failed_effect_keeps_its_class_at_commit() { + let refusals = [ + DdlError::from_error(&crate::Error::DataPlane( + crate::bridge::envelope::ErrorCode::RejectedConstraint { + constraint: "unique".into(), + detail: "duplicate 'a'".into(), + }, + )), + DdlError::from_error(&crate::Error::RejectedAuthz { + tenant_id: crate::types::TenantId::new(1), + resource: "collection 'users'".into(), + }), + DdlError::from_error(&crate::Error::CollectionNotFound { + tenant_id: crate::types::TenantId::new(1), + collection: "users".into(), + }), + DdlError::from_error(&crate::Error::RoleInUse { + role: "analyst".into(), + dependents: crate::control::security::role_assignment::RoleDependents::Users(vec![ + "bob".into(), + ]), + }), + DdlError::from_error(&crate::Error::DeadlineExceeded { + request_id: crate::types::RequestId::new(1), + }), + DdlError::new("42710", "index 'by_email' already exists"), + ]; + let expected = [ + (sqlstate::UNIQUE_VIOLATION, ErrorCode::CONSTRAINT_VIOLATION), + ( + sqlstate::INSUFFICIENT_PRIVILEGE, + ErrorCode::AUTHORIZATION_DENIED, + ), + (sqlstate::UNDEFINED_TABLE, ErrorCode::COLLECTION_NOT_FOUND), + ( + sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, + ErrorCode::DEPENDENT_OBJECTS_EXIST, + ), + ("57014", ErrorCode::DEADLINE_EXCEEDED), + ("42710", ErrorCode::ALREADY_EXISTS), + ]; + for (refusal, (state, code)) in refusals.into_iter().zip(expected) { + let error = effect_error("users", refusal); + let (_, pg_state, message) = error_to_sqlstate(&error); + assert_eq!(pg_state, state, "{error:?} on pgwire"); + assert!( + message.starts_with("index DDL on 'users' committed"), + "{message}" + ); + let native = native_error_fields(&error); + assert_eq!(native.sqlstate, state, "{error:?} native SQLSTATE"); + assert_eq!(native.code, code, "{error:?} native code"); + } + } + + /// An effect that failed with no typed class stays internal. + #[test] + fn an_untyped_effect_failure_stays_internal() { + let error = effect_error("users", DdlError::internal("core gone")); + assert_eq!(error_to_sqlstate(&error).1, sqlstate::INTERNAL_ERROR); + assert_eq!(native_error_fields(&error).code, ErrorCode::INTERNAL); + } +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs index 6e2790798..40e7d84d7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs @@ -87,7 +87,7 @@ pub async fn crdt_merge( }, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if source_bytes.is_empty() { return Err(ddl_err( "02000", @@ -97,8 +97,12 @@ pub async fn crdt_merge( let target_surrogate = state .surrogate_assigner - .assign(database_id, tenant_id, collection, target_id.as_bytes()) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .assign( + nodedb_types::CollectionKey::from_bare(database_id, collection), + tenant_id, + target_id.as_bytes(), + ) + .map_err(|e| DdlError::from_error(&e))?; let apply_plan = PhysicalPlan::Crdt(CrdtOp::Apply { collection: nodedb_types::QualifiedCollection::new(database_id, collection), @@ -114,7 +118,7 @@ pub async fn crdt_merge( }); let task = PhysicalTask { tenant_id, - vshard_id: crate::types::VShardId::from_collection_in_database(database_id, collection), + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(), database_id, plan: apply_plan, post_set_op: PostSetOp::None, @@ -131,16 +135,16 @@ pub async fn crdt_merge( .into_tasks() .into_iter() .next() - .ok_or_else(|| ddl_err("XX000", "authorization returned no capability"))?; + .ok_or_else(|| DdlError::internal("authorization returned no capability"))?; // Route the merge result through the Raft proposer gate so the applied delta // is quorum-durable under replication, not lost to followers on failover. // // RLS write policies are stored keyed by `db_qualified(database_id, // collection)`, so the policy is handed that same key or it silently - // misses a policy on a non-default database. `collection` itself stays - // bare: it also feeds vShard routing and the admission request's equality - // check against this same (unqualified) plan. + // misses a policy on a non-default database. The admission request takes + // the same qualified name: it must equal the plan's `CrdtOp::Apply` + // collection, and the preview addresses the Data Plane by it. let qualified_collection = crate::control::planner::sql_plan_convert::convert::db_qualified(database_id, collection); let policy = ExternalCrdtPostImagePolicy::from_identity( @@ -156,14 +160,14 @@ pub async fn crdt_merge( state, crate::control::crdt_admission::AuthorizedCrdtApplyAdmissionRequest { authorized, - collection, + collection: &qualified_collection, timeout: Duration::from_secs(state.tuning.network.default_deadline_secs), event_source: crate::event::EventSource::User, policy: &policy, }, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/sparse_index.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/sparse_index.rs index c69cff0a6..c998b4c0e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/sparse_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/sparse_index.rs @@ -64,7 +64,9 @@ pub fn create_sparse_index( .credentials .catalog() .get_index_record(database_id.as_u64(), tenant_id.as_u64(), &index_name) - .map_err(|e| ddl_err("XX000", format!("{CONTEXT}: read index registry: {e}")))? + .map_err(|e| { + DdlError::from_error_in_context(&format!("{CONTEXT}: read index registry"), &e) + })? { if stmt.header.if_not_exists && taken.kind == IndexKind::Sparse { return Ok(vec![DdlResult::Status { diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs index 0a551c4ff..84b2622c6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs @@ -15,6 +15,8 @@ use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::ddl::index_registry::{ IndexRegistration, propose_index_record, }; +use crate::control::server::shared::session::ddl_buffer; +use crate::control::server::shared::session::ddl_effect::DeferredDdlEffect; use crate::control::state::SharedState; use crate::types::DatabaseId; use nodedb_physical::physical_plan::TextOp; @@ -130,7 +132,9 @@ async fn create_text_index( .credentials .catalog() .get_index_record(database_id.as_u64(), tenant_id.as_u64(), &index_name) - .map_err(|e| ddl_err("XX000", format!("{command}: read index registry: {e}")))? + .map_err(|e| { + DdlError::from_error_in_context(&format!("{command}: read index registry"), &e) + })? { if stmt.header.if_not_exists && taken.kind == IndexKind::FullText { return Ok(vec![DdlResult::Status { @@ -183,16 +187,26 @@ async fn create_text_index( analyzer_name: analyzer_name.clone(), fuzzy_default, }); - crate::control::server::shared::ddl::engine_apply::apply_in_engine( - state, + // Inside an explicit transaction the binding waits for COMMIT, after + // the buffered index record lands. + let deferred = ddl_buffer::defer_effect(DeferredDdlEffect::EngineApply { tenant_id, database_id, - &collection, - set_config_plan, - "58000", - command, - ) - .await?; + collection: collection.clone(), + plan: set_config_plan.clone(), + context: command.to_string(), + }); + if !deferred { + crate::control::server::shared::ddl::engine_apply::apply_in_engine( + state, + tenant_id, + database_id, + &collection, + set_config_plan, + command, + ) + .await?; + } state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs index 47a498739..c68bc8501 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs @@ -20,6 +20,7 @@ use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::ddl::index_registry::{ IndexRegistration, propose_index_record, }; +use crate::control::server::shared::session::ddl_buffer; use crate::control::state::SharedState; use crate::types::DatabaseId; use nodedb_physical::physical_plan::VectorOp; @@ -119,7 +120,7 @@ pub async fn create_vector_index( collection, &field_name, ) - .map_err(|e| ddl_err("XX000", format!("read vector index params: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("read vector index params", &e))?; if existing.is_some() { if stmt.header.if_not_exists { return Ok(vec![status()]); @@ -140,7 +141,7 @@ pub async fn create_vector_index( .credentials .catalog() .get_index_record(database_id.as_u64(), tenant_id.as_u64(), index_name) - .map_err(|e| ddl_err("XX000", format!("read index registry: {e}")))? + .map_err(|e| DdlError::from_error_in_context("read index registry", &e))? { if stmt.header.if_not_exists && taken.kind == IndexKind::Vector { return Ok(vec![status()]); @@ -183,16 +184,31 @@ pub async fn create_vector_index( // client — this pre-flight is the only place the statement can fail // closed. The post-apply dispatch that follows re-installs the same // parameters on this node, which is a no-op on an unmaterialized index. - crate::control::server::shared::ddl::engine_apply::apply_in_engine( - state, - tenant_id, - database_id, - collection, - set_params_plan.clone(), - "42P16", - CONTEXT, - ) - .await?; + // + // Inside an explicit transaction the parameters install at COMMIT from + // the buffered catalog row, so the pre-flight changes nothing: it probes + // for a materialized index and refuses the same way. + if ddl_buffer::is_active() { + crate::control::server::shared::ddl::engine_apply::refuse_materialized_vector_index( + state, + tenant_id, + database_id, + collection, + &field_name, + CONTEXT, + ) + .await?; + } else { + crate::control::server::shared::ddl::engine_apply::apply_in_engine( + state, + tenant_id, + database_id, + collection, + set_params_plan, + CONTEXT, + ) + .await?; + } // Only now make it durable. The replicated catalog row re-registers the // index at boot via `seed_vector_index_params`, and each node's post-apply @@ -219,7 +235,11 @@ pub async fn create_vector_index( // record plus the fan-out that reaches every core, not just the one the // pre-flight dispatched to. if outcome.needs_local_apply() { - crate::control::catalog_entry::post_apply::install_vector_index_params(stored, state).await; + let shared = state + .self_arc() + .map_err(|e| DdlError::from_error_in_context("install vector index params", &e))?; + crate::control::catalog_entry::post_apply::install_vector_index_params(stored, shared) + .await; } propose_index_record( @@ -317,10 +337,16 @@ fn validate(options: &ParsedOptions) -> Result { )); } - if uses_pq && pq_m > 0 && !dim.is_multiple_of(pq_m) { + // An omitted PQ_M takes the engine default, which must divide dim too. + let effective_pq_m = if pq_m > 0 { + pq_m + } else { + nodedb_vector::index_config::DEFAULT_PQ_M + }; + if uses_pq && !dim.is_multiple_of(effective_pq_m) { return Err(ddl_err( "22023", - format!("{CONTEXT}: pq_m ({pq_m}) must divide dim ({dim}) evenly"), + format!("{CONTEXT}: pq_m ({effective_pq_m}) must divide dim ({dim}) evenly"), )); } diff --git a/nodedb/src/control/server/shared/ddl/neutral/estimate_count.rs b/nodedb/src/control/server/shared/ddl/neutral/estimate_count.rs index b1d278557..bf8d60d79 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/estimate_count.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/estimate_count.rs @@ -51,7 +51,7 @@ pub async fn estimate_count( gate.refuse_if_read_policy(&coll, "ESTIMATE_COUNT")?; gate.refuse_if_field_redacted(&coll, &field, "the estimated count")?; - let vshard = crate::types::VShardId::from_collection_in_database(database_id, &coll); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &coll).vshard(); let plan = PhysicalPlan::Document(DocumentOp::EstimateCount { collection: nodedb_types::QualifiedCollection::new(database_id, &coll), field, @@ -83,7 +83,7 @@ pub async fn estimate_count( ))]); } Err(e) => { - return Err(DdlError::new("XX000", e.to_string())); + return Err(DdlError::from_error(&e)); } } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/explain_ddl.rs b/nodedb/src/control/server/shared/ddl/neutral/explain_ddl.rs index 58d646541..ca4f88bbd 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/explain_ddl.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/explain_ddl.rs @@ -210,7 +210,7 @@ pub fn assert_visible( collection, scope.auth(), ) - .map_err(|e| DdlError::new("XX000", format!("rls compile: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("rls compile", &e))?; let visible = rls_bytes.is_some_and(|b| b.is_empty()); // No filters = visible. diff --git a/nodedb/src/control/server/shared/ddl/neutral/field_def.rs b/nodedb/src/control/server/shared/ddl/neutral/field_def.rs index c16f5947e..df47ab0e5 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/field_def.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/field_def.rs @@ -8,6 +8,8 @@ //! reads (VALUE computed fields). //! - `DEFINE EVENT ON WHEN THEN ` — //! stores an event definition in the catalog. +//! - `REMOVE EVENT ON ` — removes an event definition from +//! the catalog. //! //! Handlers build [`DdlResult`] directly and carry no pgwire wire types. @@ -124,7 +126,7 @@ pub fn define_field( database_id, &coll, ) { - return Err(err("XX000", &format!("save collection: {e}"))); + return Err(DdlError::from_error_in_context("save collection", &e)); } } _ => { @@ -227,7 +229,7 @@ pub fn define_event( database_id, &coll, ) { - return Err(err("XX000", &format!("save collection: {e}"))); + return Err(DdlError::from_error_in_context("save collection", &e)); } } _ => { @@ -251,3 +253,76 @@ pub fn define_event( rows_affected: None, }]) } + +/// Parse and apply a REMOVE EVENT statement. +/// +/// Syntax: REMOVE EVENT ON +/// +/// The collection descriptor is replicated without the definition, the same +/// way DEFINE EVENT replicates it with one. Inside a transaction the change +/// is held for COMMIT, as DEFINE EVENT's is. An undefined name is an error +/// with SQLSTATE 42704. +pub fn remove_event( + state: &SharedState, + identity: &AuthenticatedIdentity, + database_id: DatabaseId, + sql: &str, +) -> Result, DdlError> { + let parts: Vec<&str> = sql + .trim() + .trim_end_matches(';') + .split_whitespace() + .collect(); + if parts.len() != 5 || !parts[3].eq_ignore_ascii_case("ON") { + return Err(err("42601", "syntax: REMOVE EVENT ON ")); + } + let event_name = parse_ident_token(parts[2])?; + let collection = parse_ident_token(parts[4])?; + let tenant_id = identity.tenant_id; + + let audit = ArcAuditEmitter(std::sync::Arc::clone(&state.audit)); + authorize_collection( + identity, + database_id, + &collection, + Permission::Alter, + &state.permissions, + &state.roles, + &audit, + ) + .map_err(|error| err("42501", &format!("permission denied: {}", error.resource())))?; + + let catalog = state.credentials.catalog(); + let mut coll = match catalog.get_collection(database_id, tenant_id.as_u64(), &collection) { + Ok(Some(coll)) => coll, + Ok(None) => { + return Err(err( + "42P01", + &format!("collection '{collection}' does not exist"), + )); + } + Err(e) => return Err(DdlError::from_error_in_context("read collection", &e)), + }; + let before = coll.event_defs.len(); + coll.event_defs.retain(|e| e.name != event_name); + if coll.event_defs.len() == before { + return Err(err( + "42704", + &format!("event '{event_name}' on '{collection}' does not exist"), + )); + } + crate::control::catalog_entry::persist_collection_replicated(state, database_id, &coll) + .map_err(|e| DdlError::from_error_in_context("save collection", &e))?; + + state.audit_record( + crate::control::security::audit::AuditEvent::AdminAction, + Some(tenant_id), + &identity.username, + &format!("removed event '{event_name}' from '{collection}'"), + ); + + Ok(vec![DdlResult::Status { + command: "REMOVE EVENT".to_string(), + rows_affected: None, + }]) +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/function/alter.rs b/nodedb/src/control/server/shared/ddl/neutral/function/alter.rs index 8e060c2cf..2c234659e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/function/alter.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/function/alter.rs @@ -87,7 +87,7 @@ pub fn alter_function( let mut func = catalog .get_function_in_database(database_id, tenant_id, &name) - .map_err(|e| DdlError::new("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| DdlError::new("42883", format!("function '{name}' does not exist")))?; let old_owner = func.owner.clone(); @@ -129,7 +129,7 @@ fn alter_function_limits( let mut func = catalog .get_function_in_database(database_id, tenant_id, name) - .map_err(|e| DdlError::new("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| DdlError::new("42883", format!("function '{name}' does not exist")))?; // Parse SET (...) from remaining parts. diff --git a/nodedb/src/control/server/shared/ddl/neutral/function/create/handler.rs b/nodedb/src/control/server/shared/ddl/neutral/function/create/handler.rs index bed8f49bb..c44bb069c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/function/create/handler.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/function/create/handler.rs @@ -85,7 +85,7 @@ pub fn create_function( let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| DdlError::new("XX000", "system clock before UNIX epoch"))? + .map_err(|_| DdlError::internal("system clock before UNIX epoch"))? .as_secs(); let mut stored = StoredFunction { diff --git a/nodedb/src/control/server/shared/ddl/neutral/function/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/function/drop.rs index f8128f2d8..a5327e759 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/function/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/function/drop.rs @@ -36,7 +36,7 @@ pub fn drop_function( // Check if function exists. let func_exists = catalog .get_function_in_database(database_id, tenant_id, &name) - .map_err(|e| DdlError::new("XX000", format!("catalog read: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog read", &e))? .is_some(); if !func_exists && !if_exists { @@ -54,7 +54,7 @@ pub fn drop_function( // Check dependencies: block DROP if other objects depend on this function. let dependents = catalog .find_dependents(database_id, tenant_id, "function", &name) - .map_err(|e| DdlError::new("XX000", format!("dependency check: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("dependency check", &e))?; if !dependents.is_empty() { let dep_list: Vec = dependents .iter() @@ -79,7 +79,7 @@ pub fn drop_function( name: name.clone(), }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; crate::control::catalog_entry::apply::local::apply_locally_if_needed(state, &entry, outcome); // Broadcast deletion to connected Lite sessions. diff --git a/nodedb/src/control/server/shared/ddl/neutral/function/wasm_aggregate.rs b/nodedb/src/control/server/shared/ddl/neutral/function/wasm_aggregate.rs index 916168b53..a7675db7f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/function/wasm_aggregate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/function/wasm_aggregate.rs @@ -57,17 +57,16 @@ pub fn create_wasm_aggregate( .map_err(|e| DdlError::new("42601", e.to_string()))?; // Validate aggregate exports (init, accumulate, merge, finalize). - let runtime = - wasm::runtime::WasmRuntime::new().map_err(|e| DdlError::new("XX000", e.to_string()))?; + let runtime = wasm::runtime::WasmRuntime::new().map_err(|e| DdlError::from_error(&e))?; let module = runtime .get_or_compile(&wasm_bytes) - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; wasm::wit::validate_aggregate_exports(&module) .map_err(|e| DdlError::new("42601", e.to_string()))?; let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| DdlError::new("XX000", "system clock"))? + .map_err(|_| DdlError::internal("system clock"))? .as_secs(); // Store as a function with language=WASM. The "aggregate" nature is diff --git a/nodedb/src/control/server/shared/ddl/neutral/function/wasm_create.rs b/nodedb/src/control/server/shared/ddl/neutral/function/wasm_create.rs index ea10769d1..493e14ff1 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/function/wasm_create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/function/wasm_create.rs @@ -57,7 +57,7 @@ pub fn create_wasm_function( let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| DdlError::new("XX000", "system clock before UNIX epoch"))? + .map_err(|_| DdlError::internal("system clock before UNIX epoch"))? .as_secs(); let stored = StoredFunction { diff --git a/nodedb/src/control/server/shared/ddl/neutral/grant/database_permission.rs b/nodedb/src/control/server/shared/ddl/neutral/grant/database_permission.rs index 757a44869..df64a28aa 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/grant/database_permission.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/grant/database_permission.rs @@ -45,13 +45,12 @@ pub fn grant_database( let db_id = catalog .get_database_id_by_name(db_name) - .map_err(|e| DdlError::new("XX000", format!("catalog lookup: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup", &e))? .ok_or_else(|| DdlError::new("42704", format!("database '{db_name}' does not exist")))?; - // Resolve the target user_id from the grantee name. - let user_record = state - .credentials - .get_user(grantee) + // Resolve the target user_id from the grantee name, as this statement + // sees it: a user created earlier in the transaction counts. + let user_record = super::super::role_checks::visible_user(state, grantee) .ok_or_else(|| DdlError::new("42704", format!("user '{grantee}' does not exist")))?; let privileges: Vec<&str> = if privilege.eq_ignore_ascii_case("ALL") { @@ -69,12 +68,12 @@ pub fn grant_database( privilege: priv_name.to_string(), }, ) - .map_err(|e| DdlError::new("XX000", format!("catalog propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog propose", &e))?; if outcome.needs_local_apply() { catalog .put_database_grant(db_id, user_record.user_id, priv_name) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } } @@ -102,12 +101,10 @@ pub fn revoke_database( let db_id = catalog .get_database_id_by_name(db_name) - .map_err(|e| DdlError::new("XX000", format!("catalog lookup: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup", &e))? .ok_or_else(|| DdlError::new("42704", format!("database '{db_name}' does not exist")))?; - let user_record = state - .credentials - .get_user(grantee) + let user_record = super::super::role_checks::visible_user(state, grantee) .ok_or_else(|| DdlError::new("42704", format!("user '{grantee}' does not exist")))?; let privileges: Vec<&str> = if privilege.eq_ignore_ascii_case("ALL") { @@ -125,12 +122,12 @@ pub fn revoke_database( privilege: priv_name.to_string(), }, ) - .map_err(|e| DdlError::new("XX000", format!("catalog propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog propose", &e))?; if outcome.needs_local_apply() { catalog .delete_database_grant(db_id, user_record.user_id, priv_name) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/grant/permission.rs b/nodedb/src/control/server/shared/ddl/neutral/grant/permission.rs index 472a922bb..192cc7875 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/grant/permission.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/grant/permission.rs @@ -33,7 +33,9 @@ use super::support::{require_tenant_admin, status}; /// names that resolve to neither, so unresolved typos don't sink into the /// store as silently unenforceable rows. fn canonicalize_grantee(state: &SharedState, raw: &str) -> Result { - if state.credentials.get_user(raw).is_some() { + // Users and roles the statement sees: created earlier in the + // transaction counts, dropped earlier in it does not. + if super::super::role_checks::visible_user(state, raw).is_some() { return Ok(format!("user:{raw}")); } let parsed: Role = match raw.parse() { @@ -41,7 +43,7 @@ fn canonicalize_grantee(state: &SharedState, raw: &str) -> Result match e {}, }; let is_known_role = match &parsed { - Role::Custom(name) => state.roles.get_role(name).is_some(), + Role::Custom(name) => super::super::role_checks::visible_roles(state).contains_key(name), _ => true, }; if is_known_role { @@ -65,13 +67,13 @@ fn propose_grant( .prepare_permission(target, grantee, perm, granted_by); let entry = CatalogEntry::PutPermission(Box::new(stored.clone())); let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { { let catalog = state.credentials.catalog(); catalog .put_permission(&stored) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } state.permissions.install_replicated_permission(&stored); } @@ -91,13 +93,13 @@ fn propose_revoke( permission: perm_str.clone(), }; let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { { let catalog = state.credentials.catalog(); catalog .delete_permission(target, grantee, &perm_str) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } state .permissions @@ -117,7 +119,7 @@ fn resolve_tenant_id(state: &SharedState, name: &str) -> Result Result, DdlError> { - state - .credentials - .get_user(username) - .map(|r| r.roles) - .ok_or_else(|| DdlError::new("42704", format!("user '{username}' not found"))) +/// The roles and tenant of user `username` as this statement sees it: a user +/// created earlier in the transaction is visible, one dropped in it is not. +fn current_roles( + state: &SharedState, + username: &str, +) -> Result<(Vec, crate::types::TenantId), DdlError> { + let user = super::super::role_checks::visible_user_or_missing(state, username)?; + let roles = user.roles.iter().map(|name| parse_role(name)).collect(); + Ok((roles, crate::types::TenantId::new(user.tenant_id))) } fn propose_user_with_roles( state: &SharedState, username: &str, + tenant_id: crate::types::TenantId, new_roles: Vec, invalidation: crate::control::security::buses::SessionInvalidationReason, ) -> Result<(), DdlError> { + let base = super::super::role_checks::visible_user_or_missing(state, username)?; let stored = state .credentials - .prepare_user_update(username, None, Some(new_roles)) + .prepare_user_update_from(base, None, Some(new_roles.clone())) .map_err(|e| DdlError::new("42704", e.to_string()))?; let entry = CatalogEntry::PutUser(Box::new(stored.clone())); let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { { let catalog = state.credentials.catalog(); catalog .put_user(&stored) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } state .credentials .install_replicated_user(&stored, Some(invalidation)); + } else if outcome.is_replicated() { + super::super::role_checks::confirm_user_roles(state, username, &new_roles, tenant_id)?; } Ok(()) } @@ -79,9 +86,9 @@ pub fn grant_role( return Err(DdlError::new("42601", "GRANT: missing role name")); } - if state.credentials.get_user(grantee).is_some() { + if super::super::role_checks::visible_user(state, grantee).is_some() { grant_roles_to_user(state, identity, roles, grantee) - } else if state.roles.get_role(grantee).is_some() { + } else if super::super::role_checks::visible_roles(state).contains_key(grantee) { grant_role_to_role(state, identity, roles, grantee) } else { Err(DdlError::new( @@ -97,9 +104,12 @@ fn grant_roles_to_user( role_names: &[String], username: &str, ) -> Result, DdlError> { - let mut roles = current_roles(state, username)?; - for name in role_names { - let role = parse_role(name); + let (mut roles, tenant_id) = current_roles(state, username)?; + let granted: Vec = role_names.iter().map(|name| parse_role(name)).collect(); + // A role that is neither built in nor defined in the user's tenant would + // grant nothing: refuse it by name. + super::super::role_checks::check_user_roles(state, &granted, tenant_id)?; + for role in granted { if matches!(role, Role::Superuser) && !identity.is_superuser { return Err(DdlError::new( "42501", @@ -113,6 +123,7 @@ fn grant_roles_to_user( propose_user_with_roles( state, username, + tenant_id, roles, crate::control::security::buses::SessionInvalidationReason::RoleGranted, )?; @@ -181,9 +192,9 @@ pub fn revoke_role( )); } - if state.credentials.get_user(grantee).is_some() { + if super::super::role_checks::visible_user(state, grantee).is_some() { revoke_roles_from_user(state, identity, roles, grantee) - } else if state.roles.get_role(grantee).is_some() { + } else if super::super::role_checks::visible_roles(state).contains_key(grantee) { revoke_role_from_role(state, identity, roles, grantee) } else { Err(DdlError::new( @@ -199,7 +210,7 @@ fn revoke_roles_from_user( role_names: &[String], username: &str, ) -> Result, DdlError> { - let mut roles = current_roles(state, username)?; + let (mut roles, tenant_id) = current_roles(state, username)?; let revoked: Vec = role_names.iter().map(|n| parse_role(n)).collect(); for role in &revoked { if !roles.contains(role) { @@ -213,6 +224,7 @@ fn revoke_roles_from_user( propose_user_with_roles( state, username, + tenant_id, roles, crate::control::security::buses::SessionInvalidationReason::RoleRevoked, )?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/algo.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/algo.rs index 5826ce89f..06bcb95f2 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/algo.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/algo.rs @@ -137,7 +137,7 @@ pub async fn algo( }; return match result { Ok(payload) => Ok(algo_payload_to_rows(&payload, algorithm)?), - Err(e) => Err(ddl_err("XX000", e.to_string())), + Err(e) => Err(DdlError::from_error(&e)), }; } @@ -147,7 +147,7 @@ pub async fn algo( .await { Ok(resp) => Ok(algo_payload_to_rows(&resp.payload, algorithm)?), - Err(e) => Err(ddl_err("XX000", e.to_string())), + Err(e) => Err(DdlError::from_error(&e)), } } @@ -244,7 +244,7 @@ fn algo_payload_to_rows( let json_text = response_codec::decode_payload_to_json(payload); let rows: Vec = sonic_rs::from_str(&json_text) - .map_err(|e| ddl_err("XX000", format!("invalid algorithm result JSON: {e}")))?; + .map_err(|e| DdlError::internal(format!("invalid algorithm result JSON: {e}")))?; let mut shaped_rows = Vec::with_capacity(rows.len()); for row in &rows { diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs index 23e63430d..fcf346e98 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs @@ -23,14 +23,11 @@ use super::super::super::result::{DdlError, DdlResult}; use super::edge_parse::{properties_to_json, validate_edge_label}; use super::support::{data_plane_verdict, ddl_err}; -/// Read the affected count off a Data-Plane response, mapping a missing count -/// to a [`DdlError`] via `ddl_err` — never a default. +/// Read the affected count off a Data-Plane response. A missing count is an +/// error, never a default. fn response_affected(response: &crate::bridge::envelope::Response) -> Result { require_affected_count(response.payload.as_bytes()).map_err(|e| { - ddl_err( - "XX000", - format!("edge write response is missing its affected count: {e}"), - ) + DdlError::from_error_in_context("edge write response is missing its affected count", &e) }) } @@ -73,35 +70,22 @@ pub async fn insert_edge( &collection, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; // Dual-home: a cross-shard edge must be written on the home vShard of both src // and dst, or reverse/IN traversal never finds it. let vsrc = VShardId::from_key(src.as_bytes()); let vdst = VShardId::from_key(dst.as_bytes()); - let src_surrogate = assign_surrogate_routed( - state, - vsrc, - database_id, - tenant_id, - &collection, - src.as_bytes(), - TraceId::ZERO, - ) - .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; - let dst_surrogate = assign_surrogate_routed( - state, - vdst, - database_id, - tenant_id, - &collection, - dst.as_bytes(), - TraceId::ZERO, - ) - .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + let key = nodedb_types::CollectionKey::from_bare(database_id, &collection); + let src_surrogate = + assign_surrogate_routed(state, vsrc, key, tenant_id, src.as_bytes(), TraceId::ZERO) + .await + .map_err(|e| DdlError::from_error(&e))?; + let dst_surrogate = + assign_surrogate_routed(state, vdst, key, tenant_id, dst.as_bytes(), TraceId::ZERO) + .await + .map_err(|e| DdlError::from_error(&e))?; // Write policy decides the `PROPERTIES` image before staging: this handler // dispatches as trusted internal work, so nothing downstream resolves a policy. @@ -162,7 +146,7 @@ pub async fn insert_edge( crate::event::EventSource::User, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; data_plane_verdict(&response)?; response_affected(&response)? } else { @@ -176,11 +160,11 @@ pub async fn insert_edge( post_set_op: PostSetOp::None, txn_id: None, }; - let tx_class = build_static_tx_class(&[task], tenant_id, &[]) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + let tx_class = + build_static_tx_class(&[task], tenant_id, &[]).map_err(|e| DdlError::from_error(&e))?; let response = submit_calvin_routed(state, tx_class) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; match response { Some(response) => { data_plane_verdict(&response)?; @@ -192,8 +176,7 @@ pub async fn insert_edge( // participant's applied response. A missing deposit here is a // scheduler invariant violation, never a value to guess. None => { - return Err(ddl_err( - "XX000", + return Err(DdlError::internal( "cross-shard edge insert completed with no applied response to read \ its affected count from", )); @@ -256,28 +239,15 @@ pub async fn delete_edge( let vsrc = VShardId::from_key(src.as_bytes()); let vdst = VShardId::from_key(dst.as_bytes()); - let src_surrogate = assign_surrogate_routed( - state, - vsrc, - database_id, - tenant_id, - &collection, - src.as_bytes(), - TraceId::ZERO, - ) - .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; - let dst_surrogate = assign_surrogate_routed( - state, - vdst, - database_id, - tenant_id, - &collection, - dst.as_bytes(), - TraceId::ZERO, - ) - .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + let key = nodedb_types::CollectionKey::from_bare(database_id, &collection); + let src_surrogate = + assign_surrogate_routed(state, vsrc, key, tenant_id, src.as_bytes(), TraceId::ZERO) + .await + .map_err(|e| DdlError::from_error(&e))?; + let dst_surrogate = + assign_surrogate_routed(state, vdst, key, tenant_id, dst.as_bytes(), TraceId::ZERO) + .await + .map_err(|e| DdlError::from_error(&e))?; // A delete carries no image, so the policy compiles into the plan's write-gate // slot and is decided in the Data Plane against the edge's stored properties. @@ -336,7 +306,7 @@ pub async fn delete_edge( }; let response = crate::control::write_resolve::run_write_resolve(state, ctx, &*resolver) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; return Ok(vec![DdlResult::Status { command: "DELETE EDGE".to_string(), rows_affected: Some(response_affected(&response)?), @@ -357,7 +327,7 @@ pub async fn delete_edge( crate::event::EventSource::User, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; data_plane_verdict(&response)?; response_affected(&response)? } else { @@ -371,11 +341,11 @@ pub async fn delete_edge( post_set_op: PostSetOp::None, txn_id: None, }; - let tx_class = build_static_tx_class(&[task], tenant_id, &[]) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + let tx_class = + build_static_tx_class(&[task], tenant_id, &[]).map_err(|e| DdlError::from_error(&e))?; let response = submit_calvin_routed(state, tx_class) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; match response { Some(response) => { data_plane_verdict(&response)?; @@ -389,8 +359,7 @@ pub async fn delete_edge( // absent), so a missing deposit here is a scheduler invariant // violation, never a value to guess. None => { - return Err(ddl_err( - "XX000", + return Err(DdlError::internal( "cross-shard edge delete completed with no applied response to read \ its affected count from", )); @@ -433,28 +402,40 @@ pub async fn set_node_labels( }; // Single-keyed on `node_id`, so single-home: route to `from_key(node_id)`. - // No redb durability — a WAL record is the bitset's only backing. - crate::control::server::wal_dispatch::wal_append_if_write( - &state.wal, + // No redb durability — a WAL record is the bitset's only backing. The + // record's outcome-floor window opens before the append and closes from + // the dispatch's outcome. + let owner = crate::control::server::dispatch_utils::RecordOwner { tenant_id, + database_id: DatabaseId::DEFAULT, vshard_id, - DatabaseId::DEFAULT, + }; + let minted = crate::control::server::dispatch_utils::MintedRecords::open(&state.outcome_floor); + if let Err(e) = minted.append_plan( + &state.wal, + owner, &plan, - ) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + // The same source the edge write is dispatched with below. + crate::event::EventSource::User, + ) { + // Any record appended before the error never reaches a core. + minted + .cancel(&state.wal, owner, 0) + .await + .map_err(|c| DdlError::from_error(&c))?; + return Err(DdlError::from_error(&e)); + } let response = - crate::control::server::sync::raft_dispatch::dispatch_trusted_internal_sync_response( + crate::control::server::sync::raft_dispatch::dispatch_trusted_internal_minted_sync_response( state, - tenant_id, - DatabaseId::DEFAULT, - vshard_id, + owner, plan, - TraceId::ZERO, crate::event::EventSource::User, + minted, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; data_plane_verdict(&response)?; let tag = if remove { "UNLABEL" } else { "LABEL" }; diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_parse.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_parse.rs index 7735b21c5..4795a9372 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_parse.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_parse.rs @@ -57,7 +57,7 @@ pub(super) fn properties_to_json(properties: GraphProperties) -> Result sonic_rs::to_string(&nodedb_types::Value::Object(fields)) - .map_err(|e| ddl_err("XX000", format!("PROPERTIES serialize error: {e}"))), + .map_err(|e| DdlError::internal(format!("PROPERTIES serialize error: {e}"))), Some(Err(msg)) => Err(ddl_err( "42601", format!("PROPERTIES object literal error: {msg}"), diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_rls.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_rls.rs index 2b08b4220..436cfa378 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_rls.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_rls.rs @@ -22,7 +22,6 @@ use crate::control::state::SharedState; use crate::types::DatabaseId; use super::super::super::result::DdlError; -use super::support::ddl_err; /// Resolve the collection's write policy against a hand-built edge write. /// @@ -39,9 +38,8 @@ pub(super) fn resolve_edge_write_rls( CollectionReadGate::for_request(state, identity, database_id).inject_rls(&mut plan)?; match plan { PhysicalPlan::Graph(op) => Ok(op), - other => Err(ddl_err( - "XX000", - format!("edge write plan changed shape during RLS resolution: {other:?}"), - )), + other => Err(DdlError::internal(format!( + "edge write plan changed shape during RLS resolution: {other:?}" + ))), } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs index 1e627cfdd..5022b7d0c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs @@ -24,6 +24,7 @@ //! vShard set records both homes so ROLLBACK tears down both. use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::pgwire::types::error_to_sqlstate; use crate::control::server::shared::session::DmlTxnCtx; use crate::control::server::shared::session::staging_gate::{ InTxnRoute, StagingGateError, route_in_tx_write, @@ -150,20 +151,25 @@ pub(super) async fn stage_edge_write_in_txn( match routed { Ok(InTxnRoute::Staged(outcome)) => Ok(outcome.affected as u64), // Edge writes are stageable (`is_stageable_write`), so inside a - // transaction block the gate always returns `Staged`. `Read` (not in a - // block) and `Buffered` (non-stageable write) cannot occur for a - // caller that already checked `InBlock`; there is no affected count to - // report for either, so treat them as a no-op tag rather than panicking. - Ok(InTxnRoute::Read(_)) | Ok(InTxnRoute::Buffered) => Ok(0), - Err(StagingGateError::Dispatch(e)) => Err(ddl_err("XX000", e.to_string())), - Err(StagingGateError::Rejected { code }) => { - let (_, sqlstate, message) = match code { - Some(code) => { - crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate(&code) - } - None => ("ERROR", "XX000", "unknown data plane error".to_owned()), - }; + // transaction block the gate returns `Staged`. Any other route means + // the caller's `InBlock` check and the gate disagree. A report of zero + // rows drops the write silently, so the statement fails instead. + Ok(InTxnRoute::Read(_) | InTxnRoute::Autocommit(_) | InTxnRoute::Buffered) => { + Err(DdlError::internal( + "a graph edge write reached the transaction staging gate and was not staged", + )) + } + Err(StagingGateError::Dispatch(e)) => { + let (_, sqlstate, message) = error_to_sqlstate(&e); Err(ddl_err(sqlstate, message)) } + Err(StagingGateError::Rejected { code }) => Err(match code { + Some(code) => { + let (_, sqlstate, message) = + crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate(&code); + ddl_err(sqlstate, message) + } + None => DdlError::internal("unknown data plane error"), + }), } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/rag_fusion.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/rag_fusion.rs index 2d5f1b1c9..df6d66902 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/rag_fusion.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/rag_fusion.rs @@ -134,7 +134,7 @@ pub async fn rag_fusion( admission: user_dispatch::RequestAdmission::AlreadyAdmitted, }) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let json_text = response_codec::decode_payload_to_json(&payload); let mut row = Map::new(); diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/stats.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/stats.rs index 0f0e07a63..51ed521be 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/stats.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/stats.rs @@ -48,7 +48,6 @@ use nodedb_physical::physical_plan::GraphOp; use super::super::super::result::{DdlError, DdlResult}; use super::super::refuse_gate::RefusingReadGate; -use super::support::ddl_err; /// Names the collection-scoped stats read in the refusal a read policy raises. const STATS_WHAT: &str = "graph statistics, which are counters over the collection's edges"; @@ -120,10 +119,10 @@ pub async fn show_graph_stats( let resp = broadcast_to_all_cores(state, identity.tenant_id, database_id, plan, TraceId::ZERO) .await - .map_err(|e| ddl_err("58000", format!("graph stats dispatch failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("graph stats dispatch failed", &e))?; let merged: Vec = decode_merged_stats(resp.payload.as_bytes()) - .map_err(|e| ddl_err("XX000", format!("graph stats decode failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("graph stats decode failed", &e))?; let aggregated = aggregate_by_collection(merged); diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/support.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/support.rs index 4d14a25bc..2cf23600c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/support.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/support.rs @@ -36,11 +36,14 @@ pub(super) fn data_plane_verdict( if response.status != crate::bridge::envelope::Status::Error { return Ok(()); } - let (_, sqlstate, message) = match response.error_code.as_deref() { - Some(code) => super::super::super::sqlstate::error_code_to_sqlstate(code), - None => ("ERROR", "XX000", "unknown data plane error".to_owned()), - }; - Err(ddl_err(sqlstate, message)) + Err(match response.error_code.as_deref() { + Some(code) => { + let (_, sqlstate, message) = + super::super::super::sqlstate::error_code_to_sqlstate(code); + ddl_err(sqlstate, message) + } + None => DdlError::internal("unknown data plane error"), + }) } /// Gate a named collection on catalog `is_active`: a plain `DROP COLLECTION` diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/traverse.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/traverse.rs index 81ec6d094..60a63b52b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/traverse.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/traverse.rs @@ -168,7 +168,7 @@ pub async fn traverse( .await { Ok(resp) => Ok(payload_to_rows(&resp.payload)), - Err(e) => Err(ddl_err("XX000", e.to_string())), + Err(e) => Err(DdlError::from_error(&e)), } } @@ -229,7 +229,7 @@ pub async fn neighbors( .await { Ok(resp) => Ok(payload_to_rows(&resp.payload)), - Err(e) => Err(ddl_err("XX000", e.to_string())), + Err(e) => Err(DdlError::from_error(&e)), } } @@ -287,7 +287,7 @@ pub async fn shortest_path( .await { Ok(resp) => Ok(payload_to_rows(&resp.payload)), - Err(e) => Err(ddl_err("XX000", e.to_string())), + Err(e) => Err(DdlError::from_error(&e)), } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/inspect_audit.rs b/nodedb/src/control/server/shared/ddl/neutral/inspect_audit.rs index 6c8aba052..7ebd69c64 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/inspect_audit.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/inspect_audit.rs @@ -75,7 +75,7 @@ pub fn show_audit_log( let entries = catalog .load_recent_audit_entries(limit) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let (columns, column_types) = audit_columns(); let mut rows = Vec::with_capacity(entries.len()); @@ -236,7 +236,7 @@ pub fn show_audit_in_database( let db_id = catalog .get_database_id_by_name(db_name) - .map_err(|e| ddl_err("XX000", format!("catalog lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))? .ok_or_else(|| ddl_err("3D000", format!("database '{db_name}' does not exist")))?; let (columns, column_types) = audit_columns(); @@ -289,7 +289,7 @@ pub fn show_audit_in_database( let remaining = limit - rows.len(); let all_entries = catalog .load_recent_audit_entries(remaining * 10) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; for entry in all_entries.iter().rev() { if rows.len() >= limit { break; diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/dispatch.rs index 2316c3aa1..4863e4a11 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/dispatch.rs @@ -5,9 +5,12 @@ //! directory (`handlers` here, and the sibling `kv_sorted_index`, //! `weighted_pick`, `rate_gate`, `transfer` modules). +use std::sync::Arc; + use serde_json::{Map, Value as JsonValue}; use crate::bridge::envelope::Status; +use crate::control::security::audit::ArcAuditEmitter; use crate::control::security::identity::{AuthenticatedIdentity, Permission}; use crate::control::server::response_shape::types::ShapedRows; use crate::control::server::shared::session::DmlTxnCtx; @@ -23,27 +26,25 @@ use super::super::read_gate::CollectionReadGate; /// and return the JSON response as a single text-column row keyed by the /// lower-cased function name. /// -/// Outside a transaction block (or for the read half of the gate, which -/// never applies here since every `KvOp` this module builds is a write), -/// `route_in_tx_write` dispatches immediately -- byte-identical to the -/// pre-staging behavior. Inside a transaction, `KvOp::Incr` / `IncrFloat` / -/// `Cas` / `GetSet` are staged into the per-transaction overlay -/// (`is_stageable_write`) and this function reads the computed value back -/// from `StagedWriteOutcome::payload` (forwarded verbatim by the staging -/// gate for `StagedTagKind::RawPayload`), so a `SELECT KV_INCR(...)` inside -/// `BEGIN..COMMIT` returns the same value the staged overlay now holds, and -/// a following `SELECT KV_INCR(...)` on the same key in the same -/// transaction chains off it. +/// Outside a transaction block the gate answers `Autocommit`, and the op takes +/// the durable route every planned autocommit write takes: proposed through +/// Raft in cluster mode, otherwise the write funnel with `AppendHere`. Either +/// way a WAL record reproduces the value the op computed, and a replica +/// applies it. +/// +/// Inside a transaction, `KvOp::Incr` / `IncrFloat` / `Cas` / `GetSet` / +/// `Transfer` / `TransferItem` are staged into the per-transaction overlay +/// (`is_stageable_write`). This function reads the computed value back from +/// `StagedWriteOutcome::payload`, so a `SELECT KV_INCR(...)` inside +/// `BEGIN..COMMIT` returns the value the staged overlay now holds, and a +/// following `SELECT KV_INCR(...)` on the same key chains off it. /// /// `collections` names every collection the op touches, in the caller's own -/// words rather than read back out of the plan: these `KvOp`s carry no -/// collection the plan-classification helpers report, and `TRANSFER_ITEM` -/// touches two. Each is authorized here before the op is routed anywhere. +/// words rather than read back out of the plan: `TRANSFER_ITEM` touches two. +/// Each is authorized here before the op is routed anywhere. /// -/// Reused by the sibling `transfer.rs` module for the identical -/// in-transaction routing for `TRANSFER` / `TRANSFER_ITEM` instead of the -/// direct `dispatch_to_data_plane` call it used before those two `KvOp`s -/// became stageable. +/// Reused by the sibling `transfer.rs` module for `TRANSFER` / +/// `TRANSFER_ITEM`. pub(crate) async fn dispatch_and_respond( state: &SharedState, identity: &AuthenticatedIdentity, @@ -61,10 +62,10 @@ pub(crate) async fn dispatch_and_respond( let database_id = DatabaseId::DEFAULT; // Every caller here names its collections in the SQL text and reaches the - // Data Plane through a hand-built `KvOp`, which carries no identity and is - // never authorized downstream. The op reports the value it replaced or - // computed, so it is a read as much as a write and needs both grants — and - // a cross-collection move needs them on each side, hence the slice. + // Data Plane through a hand-built `KvOp`. The op reports the value it + // replaced or computed, so it is a read as much as a write and needs both + // grants — and a cross-collection move needs them on each side, hence the + // slice. let gate = CollectionReadGate::for_request(state, identity, database_id); for collection in collections { gate.authorize(collection)?; @@ -72,15 +73,13 @@ pub(crate) async fn dispatch_and_respond( } // Row-level security is resolved by the same injection pass the - // planner-driven path runs, against the very same op. That is the point of - // routing it through here rather than restating a verdict locally: these - // functions build the identical `KvOp`s a planned statement builds, so a - // local refusal here while the planner enforced — or the reverse — would - // give the same operation two different answers depending on which syntax - // reached it. The pass compiles the write policy into the op's own gate + // planner-driven path runs, against the very same op. These functions + // build the identical `KvOp`s a planned statement builds. A local refusal + // here while the planner enforced, or the reverse, gives the same + // operation two answers that depend on the syntax that reached it. The pass compiles the write policy into the op's own gate // slot for the Data Plane to decide the image against, injects the read - // filter where the reply is a row body, and still refuses outright where - // the reply is a value computed from a row the policy hides. + // filter where the reply is a row body, and refuses outright where the + // reply is a value computed from a row the policy hides. gate.inject_rls(&mut plan)?; let task = PhysicalTask { @@ -112,42 +111,24 @@ pub(crate) async fn dispatch_and_respond( .await; let payload = match routed { - Ok(InTxnRoute::Read(task)) => { - let task = *task; - match crate::control::server::dispatch_utils::dispatch_to_data_plane_with_txn( - state, - task.tenant_id, - task.database_id, - task.vshard_id, - task.plan, - TraceId::ZERO, - task.txn_id, - ) - .await - { - // A refused write comes back as `Ok(Response)` carrying - // `Status::Error` — `submit_write` reports the dispatch itself - // as having succeeded and puts the verdict inside the response. - // Its payload is empty, so forwarding it unchecked would answer - // `SELECT KV_INCR(...)` with one blank column and let the caller - // read a refusal as a completed write. Every terminal outcome - // these functions can produce arrives this way — a policy - // refusal, a type mismatch, an overflow, an insufficient - // balance, a missing key — so the status is what decides, - // never the payload's emptiness. - Ok(resp) if resp.status == Status::Error => { - return Err(data_plane_error(resp.error_code.map(|code| *code))); - } - Ok(resp) => resp.payload.as_ref().to_vec(), - Err(e) => return Err(ddl_err("XX000", e.to_string())), - } - } - // Every `KvOp` this module builds is stageable once in a - // transaction (`is_stageable_write`), so `Buffered` never occurs; - // handled defensively with an empty payload rather than a panic. - Ok(InTxnRoute::Buffered) => Vec::new(), + Ok(InTxnRoute::Autocommit(task)) => dispatch_autocommit(state, identity, *task).await?, Ok(InTxnRoute::Staged(outcome)) => outcome.payload, - Err(StagingGateError::Dispatch(e)) => return Err(ddl_err("XX000", e.to_string())), + // Every `KvOp` this module builds is a write, and a stageable one once + // in a transaction (`is_stageable_write`). A read route has no + // durable apply, and a buffered route has no value to answer with, so + // either one is a classification break, never an answer. + Ok(InTxnRoute::Read(_)) => { + return Err(DdlError::internal(format!( + "{func_name}: the staging gate classified this write as a read" + ))); + } + Ok(InTxnRoute::Buffered) => { + return Err(DdlError::internal(format!( + "{func_name}: the staging gate buffered this write, so it has no value to \ + return at the statement" + ))); + } + Err(StagingGateError::Dispatch(e)) => return Err(DdlError::from_error(&e)), Err(StagingGateError::Rejected { code }) => return Err(data_plane_error(code)), }; @@ -156,6 +137,70 @@ pub(crate) async fn dispatch_and_respond( Ok(vec![single_text_col(&col_name, payload_text)]) } +/// Apply one autocommit op on the durable route and return its payload. +/// +/// The task passes the same clone-write gate and authorization every +/// transport runs before a write dispatches. +async fn dispatch_autocommit( + state: &SharedState, + identity: &AuthenticatedIdentity, + task: PhysicalTask, +) -> Result, DdlError> { + use crate::control::server::shared::clone_write::{ + CloneCheckedOutcome, InterceptAndAuthorizeParams, intercept_and_authorize, + }; + + let emitter = ArcAuditEmitter(Arc::clone(&state.audit)); + let outcome = intercept_and_authorize(InterceptAndAuthorizeParams { + state, + task, + identity, + tenant_id: identity.tenant_id, + permissions: &state.permissions, + roles: &state.roles, + emitter: &emitter, + }) + .await + .map_err(|e| error_to_ddl(&e))?; + let response = match outcome { + CloneCheckedOutcome::Handled(resp) => resp, + CloneCheckedOutcome::Proceed(checked) => { + crate::control::server::dispatch_utils::dispatch_authorized_durable_write( + state, + checked, + TraceId::ZERO, + ) + .await + .map_err(|e| error_to_ddl(&e))? + } + }; + // A refused write comes back as `Ok(Response)` carrying `Status::Error`: + // the funnel reports the dispatch as successful and puts the verdict in the + // response. Its payload is empty, so an unchecked forward answers + // `SELECT KV_INCR(...)` with one blank column. Every terminal outcome these + // functions can produce arrives this way on the local route — a policy + // refusal, a type mismatch, an overflow, an insufficient balance, a + // missing key — so the status decides, never the payload's emptiness. + if response.status == Status::Error { + return Err(data_plane_error(response.error_code.map(|code| *code))); + } + Ok(response.payload.as_ref().to_vec()) +} + +/// Map a dispatch error to the client-facing error. A Data-Plane verdict +/// arrives here as `Error::DataPlane` on the replicated route, and it renders +/// the same way the local route's error status does. +fn error_to_ddl(error: &crate::Error) -> DdlError { + match error { + crate::Error::DataPlane(code) => data_plane_error(Some(code.clone())), + other => { + let (_, sqlstate, message) = + crate::control::server::pgwire::types::error_to_sqlstate(other); + ddl_err(sqlstate, message) + } + } +} + /// Translate a terminal Data-Plane verdict into the client-facing error. /// /// Shared by both routes a refusal can arrive on — the staging gate's @@ -167,11 +212,15 @@ pub(crate) async fn dispatch_and_respond( /// bug in whatever produced it; it is surfaced as an internal error rather than /// silently downgraded to success. fn data_plane_error(code: Option) -> DdlError { - let (_, sqlstate, message) = match code { - Some(code) => crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate(&code), - None => ("ERROR", "XX000", "unknown data plane error".to_owned()), + let Some(code) = code else { + return DdlError::internal("unknown data plane error"); }; - ddl_err(sqlstate, message) + let (_, sqlstate, message) = + crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate(&code); + // The public error carries the code a client classifies by and the + // details naming the collection, which the SQLSTATE alone cannot give. + let public = nodedb_types::NodeDbError::from(crate::Error::DataPlane(code)); + DdlError::from_public(sqlstate, message, &public) } /// Build a single-text-column row set carrying `text` under `col`. diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs index 46bea8487..af622d164 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs @@ -10,8 +10,9 @@ use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::session::DmlTxnCtx; use crate::control::state::SharedState; -use crate::types::{DatabaseId, VShardId}; -use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; +use crate::types::DatabaseId; +use nodedb_physical::physical_plan::{KvCounterShape, KvOp, PhysicalPlan}; +use nodedb_sql::planner::dml_helpers::KvCounterKind; use super::super::super::result::{DdlError, DdlResult}; use super::dispatch::{ @@ -51,16 +52,16 @@ pub async fn kv_incr( let ttl_ms = parse_optional_ttl(&args[3..])?; - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection).vshard(); let surrogate = state .surrogate_assigner .assign( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection), identity.tenant_id, - &collection, key.as_bytes(), ) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; + let shape = counter_shape(state, identity, &collection, &key, KvCounterKind::Integer)?; let plan = PhysicalPlan::Kv(KvOp::Incr { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, &collection), key: key.as_bytes().to_vec(), @@ -70,6 +71,7 @@ pub async fn kv_incr( // Filled by `dispatch_and_respond`, which runs the same RLS injection // pass the planner-driven path runs. rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape, }); dispatch_and_respond( @@ -104,23 +106,29 @@ pub async fn kv_incr_float( let collection = unquote(&args[0]).to_lowercase(); let key = unquote(&args[1]); - let delta: f64 = args[2].trim().parse().map_err(|_| { - ddl_err( + // The delta stays the client's decimal text, so the engine adds every + // digit the client wrote. + let delta = unquote(&args[2]).trim().to_string(); + if !nodedb_physical::kv_atomic::float_text::is_decimal_number(&delta) { + return Err(ddl_err( "42601", - format!("KV_INCR_FLOAT: delta must be a float, got '{}'", args[2]), - ) - })?; + format!( + "KV_INCR_FLOAT: delta must be a decimal number, got '{}'", + args[2] + ), + )); + } - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection).vshard(); let surrogate = state .surrogate_assigner .assign( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection), identity.tenant_id, - &collection, key.as_bytes(), ) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; + let shape = counter_shape(state, identity, &collection, &key, KvCounterKind::Float)?; let plan = PhysicalPlan::Kv(KvOp::IncrFloat { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, &collection), key: key.as_bytes().to_vec(), @@ -128,6 +136,7 @@ pub async fn kv_incr_float( surrogate, // Filled by `dispatch_and_respond` — see `kv_incr`. rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape, }); dispatch_and_respond( @@ -142,6 +151,43 @@ pub async fn kv_incr_float( .await } +/// The row an absent `key` in `collection` becomes, planned from the catalog. +/// +/// The caller's grants are checked first, the same pair `dispatch_and_respond` +/// checks: planning the row reads the catalog and can evaluate a DEFAULT, so +/// a caller refused the collection must be refused before either happens. +fn counter_shape( + state: &SharedState, + identity: &AuthenticatedIdentity, + collection: &str, + key: &str, + kind: KvCounterKind, +) -> Result { + let gate = super::super::read_gate::CollectionReadGate::for_request( + state, + identity, + DatabaseId::DEFAULT, + ); + gate.authorize(collection)?; + gate.authorize_permission( + collection, + crate::control::security::identity::Permission::Write, + )?; + crate::control::planner::sql_plan_convert::kv_counter_shape::kv_counter_shape( + state, + identity.tenant_id, + DatabaseId::DEFAULT, + collection, + key, + kind, + ) + .map_err(|error| { + let (_, sqlstate, message) = + crate::control::server::pgwire::types::error_to_sqlstate(&error); + DdlError::new(sqlstate, message) + }) +} + /// Handle `SELECT KV_CAS(collection, key, expected, new_value)` /// /// Returns `{"success": bool, "current_value": ""}` as a single text column. @@ -165,16 +211,15 @@ pub async fn kv_cas( let expected = unquote(&args[2]); let new_value = unquote(&args[3]); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection).vshard(); let surrogate = state .surrogate_assigner .assign( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection), identity.tenant_id, - &collection, key.as_bytes(), ) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let plan = PhysicalPlan::Kv(KvOp::Cas { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, &collection), key: key.as_bytes().to_vec(), @@ -219,16 +264,15 @@ pub async fn kv_getset( let key = unquote(&args[1]); let new_value = unquote(&args[2]); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection).vshard(); let surrogate = state .surrogate_assigner .assign( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection), identity.tenant_id, - &collection, key.as_bytes(), ) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let plan = PhysicalPlan::Kv(KvOp::GetSet { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, &collection), key: key.as_bytes().to_vec(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/mod.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/mod.rs index 43ce99325..2bb312865 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/mod.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/mod.rs @@ -19,6 +19,6 @@ pub mod dispatch; pub mod handlers; pub(crate) use dispatch::{ - ddl_err, dispatch_and_respond, parse_function_args, single_text_col, split_args, unquote, + dispatch_and_respond, parse_function_args, single_text_col, split_args, unquote, }; pub use handlers::{kv_cas, kv_getset, kv_incr, kv_incr_float}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/ddl.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/ddl.rs index e59dea581..c039056b2 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/ddl.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/ddl.rs @@ -14,6 +14,8 @@ use crate::control::security::audit::AuditEvent; use crate::control::security::catalog::IndexKind; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::ddl::sql_parse::parse_ident_token; +use crate::control::server::shared::session::ddl_buffer; +use crate::control::server::shared::session::ddl_effect::DeferredDdlEffect; use crate::control::state::SharedState; use crate::types::DatabaseId; use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; @@ -85,7 +87,7 @@ pub async fn create_sorted_index( .credentials .catalog() .get_collection(database_id, tenant_id.as_u64(), &collection) - .map_err(|e| ddl_err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .is_none() { return Err(ddl_err( @@ -108,22 +110,33 @@ pub async fn create_sorted_index( window_end_ms: window_end, }); + // Inside an explicit transaction the records are buffered first and the + // tree is built at COMMIT, so a ROLLBACK leaves no tree behind. + let in_transaction = ddl_buffer::is_active(); + // Routed by the collection, not by the index name: the backfill this plan // performs reads the collection's rows out of the `KvEngine` of whichever // core executes it, and every later write that must keep the tree current // lands on the collection's own core. Registering anywhere else builds an // empty tree that no write ever updates. - let response = register_in_engine( - state, - &SortedIndexTarget { - tenant_id, - database_id, - collection: &collection, - }, - plan, - "CREATE SORTED INDEX", - ) - .await?; + let response = if in_transaction { + vec![DdlResult::Status { + command: "CREATE SORTED INDEX".to_string(), + rows_affected: None, + }] + } else { + register_in_engine( + state, + &SortedIndexTarget { + tenant_id, + database_id, + collection: &collection, + }, + plan.clone(), + "CREATE SORTED INDEX", + ) + .await? + }; // Identity record: what resolves the index's owning collection on every // later read, what SHOW INDEXES lists, and what DROP INDEX resolves. @@ -149,6 +162,21 @@ pub async fn create_sorted_index( &identity.username, )?; + if in_transaction { + let deferred = ddl_buffer::defer_effect(DeferredDdlEffect::SortedIndexRegister { + tenant_id, + database_id, + collection: collection.clone(), + plan, + }); + if !deferred { + return Err(DdlError::internal( + "CREATE SORTED INDEX: the transaction buffer took no entry to defer the \ + index build on", + )); + } + } + state.audit_record( AuditEvent::AdminAction, Some(tenant_id), @@ -207,16 +235,18 @@ pub async fn drop_sorted_index( )); } - drop_in_engine( - state, - &SortedIndexTarget { - tenant_id, - database_id, - collection: &collection, - }, - &index_name, - ) - .await?; + // Autocommit drops the tree first, so a failed drop keeps the records a + // retry resolves through. Inside an explicit transaction the records are + // buffered first and the tree is dropped at COMMIT. + let target = SortedIndexTarget { + tenant_id, + database_id, + collection: &collection, + }; + let in_transaction = ddl_buffer::is_active(); + if !in_transaction { + drop_in_engine(state, &target, &index_name).await?; + } propose_delete_index_record(state, database_id, tenant_id, &index_name, &collection)?; crate::control::server::shared::ddl::owner::propose_delete_owner( @@ -227,6 +257,20 @@ pub async fn drop_sorted_index( &index_name, )?; + if in_transaction + && !ddl_buffer::defer_effect(DeferredDdlEffect::SortedIndexDrop { + tenant_id, + database_id, + collection: collection.clone(), + index_name: index_name.clone(), + }) + { + return Err(DdlError::internal( + "DROP SORTED INDEX: the transaction buffer took no entry to defer the index \ + drop on", + )); + } + Ok(vec![DdlResult::Status { command: "DROP SORTED INDEX".to_string(), rows_affected: None, diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs index 66a91a2d2..b33617619 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs @@ -17,6 +17,7 @@ use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; use super::super::super::result::{DdlError, DdlResult}; use super::parse::ddl_err; +use super::txn_read::SortedRead; /// Where one sorted index's Data Plane state lives. /// @@ -44,7 +45,7 @@ pub struct SortedIndexTarget<'a> { impl SortedIndexTarget<'_> { /// The vShard holding both the collection's rows and its index trees. fn vshard(&self) -> VShardId { - VShardId::from_collection_in_database(self.database_id, self.collection) + nodedb_types::CollectionKey::from_bare(self.database_id, self.collection).vshard() } } @@ -73,28 +74,31 @@ fn refusal(target: &SortedIndexTarget<'_>, resp: &Response) -> Option target.collection ), ), - Some(other) => ddl_err("XX000", format!("{other:?}")), - None => ddl_err("XX000", String::from_utf8_lossy(&resp.payload).into_owned()), + Some(other) => DdlError::from_error(&crate::Error::DataPlane(other.clone())), + None => DdlError::internal(String::from_utf8_lossy(&resp.payload)), }) } /// Dispatch a sorted-index read (`RANK` / `TOPK` / `RANGE` / `SORTED_COUNT` / -/// `ZSCORE`), which mints no durable record. -async fn dispatch_read( +/// score), which mints no durable record. A read inside a transaction carries +/// its transaction id, so the Data Plane folds in that transaction's staged +/// writes. +pub(super) async fn dispatch_read( state: &SharedState, target: &SortedIndexTarget<'_>, - plan: PhysicalPlan, + read: SortedRead, ) -> Result { - let resp = crate::control::server::dispatch_utils::dispatch_to_data_plane( + let resp = crate::control::server::dispatch_utils::dispatch_to_data_plane_with_txn( state, target.tenant_id, target.database_id, target.vshard(), - plan, + read.plan, TraceId::ZERO, + read.txn_id, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; match refusal(target, &resp) { Some(error) => Err(error), @@ -102,20 +106,26 @@ async fn dispatch_read( } } -/// Dispatch a sorted-index registration or teardown. +/// Dispatch a sorted-index registration or teardown on the durable route. +/// +/// In cluster mode the op is proposed through Raft, so every replica of the +/// collection's vShard builds or drops the tree from the committed entry. +/// Otherwise the write funnel appends its WAL record +/// (`kv_register_sorted_index` / `kv_drop_sorted_index`) under the +/// write-admission guard. The manager holds the tree only in memory, so that +/// record plus the KV checkpoint is all that carries a registration across a +/// restart. Dispatched as a read, the catalog keeps listing an index whose +/// tree no longer exists anywhere. /// -/// These go through the autocommit write funnel rather than the read path so -/// the funnel appends their WAL record (`kv_register_sorted_index` / -/// `kv_drop_sorted_index`) under the write-admission guard. The manager holds -/// the tree only in memory, so that record plus the KV checkpoint is all that -/// carries a registration across a restart: dispatched as a read, the catalog -/// would keep listing an index whose tree no longer exists anywhere. +/// A Data-Plane verdict from the replicated route arrives as +/// `Error::DataPlane`. It comes back here as the error-status response the +/// local route gives, so [`refusal`] reads both the one way. async fn dispatch_durable( state: &SharedState, target: &SortedIndexTarget<'_>, plan: PhysicalPlan, ) -> Result { - crate::control::server::dispatch_utils::dispatch_autocommit_write( + let dispatched = crate::control::server::dispatch_utils::dispatch_durable_autocommit_write( state, crate::control::server::dispatch_utils::AutocommitWrite { tenant_id: target.tenant_id, @@ -127,8 +137,28 @@ async fn dispatch_durable( txn_id: None, }, ) - .await - .map_err(|e| ddl_err("XX000", e.to_string())) + .await; + match dispatched { + Ok(resp) => Ok(resp), + Err(crate::Error::DataPlane(code)) => Ok(verdict_response(code)), + Err(e) => Err(DdlError::from_error(&e)), + } +} + +/// The error-status response the local route gives for a Data-Plane verdict. +fn verdict_response(code: ErrorCode) -> Response { + Response { + request_id: crate::types::RequestId::new(0), + status: Status::Error, + attempt: 0, + partial: false, + payload: crate::bridge::envelope::Payload::empty(), + watermark_lsn: crate::types::Lsn::ZERO, + error_code: Some(Box::new(code)), + read_set_valid: None, + read_version_lsn: crate::types::Lsn::ZERO, + write_set: Vec::new(), + } } /// Decode a row-shaped sorted-index reply. @@ -140,7 +170,7 @@ async fn dispatch_durable( /// report an empty leaderboard for every query, whatever the index held. fn decode_rows(payload: &[u8]) -> Result, DdlError> { crate::data::executor::response_codec::decode_payload(payload) - .map_err(|e| ddl_err("XX000", format!("sorted index reply: {e}"))) + .map_err(|e| DdlError::from_error_in_context("sorted index reply", &e)) } /// Build the index's tree on the core that owns its collection's rows, and @@ -152,7 +182,7 @@ fn decode_rows(payload: &[u8]) -> Result, DdlError> { /// an apply that did not happen files a record for an index that exists /// nowhere, and every read of it then answers from an index that was never /// built. -pub(super) async fn register_in_engine( +pub(crate) async fn register_in_engine( state: &SharedState, target: &SortedIndexTarget<'_>, plan: PhysicalPlan, @@ -194,30 +224,19 @@ pub async fn drop_in_engine( Ok(()) } -/// Dispatch plan and return a single-row JSON response. -pub(super) async fn dispatch_and_respond_json( - state: &SharedState, - target: &SortedIndexTarget<'_>, - plan: PhysicalPlan, - col_name: &str, -) -> Result, DdlError> { - let resp = dispatch_read(state, target, plan).await?; +/// A read's reply as a single-row JSON response. +pub(super) fn respond_json(resp: &Response, col_name: &str) -> Vec { let payload_text = crate::data::executor::response_codec::decode_payload_to_json(&resp.payload); let mut row = Map::new(); row.insert(col_name.to_string(), JsonValue::String(payload_text)); - Ok(vec![DdlResult::Rows(ShapedRows::text_rows( + vec![DdlResult::Rows(ShapedRows::text_rows( vec![col_name.to_string()], vec![row], - ))]) + ))] } -/// Dispatch plan and return multi-row response (for TOPK, RANGE). -pub(super) async fn dispatch_and_respond_rows( - state: &SharedState, - target: &SortedIndexTarget<'_>, - plan: PhysicalPlan, -) -> Result, DdlError> { - let resp = dispatch_read(state, target, plan).await?; +/// A read's reply as a multi-row response (for TOPK, RANGE). +pub(super) fn respond_rows(resp: &Response) -> Result, DdlError> { let rows_json = decode_rows(&resp.payload)?; let mut rows = Vec::with_capacity(rows_json.len()); diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/mod.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/mod.rs index 2c9582dff..729be04d7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/mod.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/mod.rs @@ -18,7 +18,9 @@ pub mod dispatch; pub mod gate; pub mod parse; pub mod query; +mod txn_read; pub use ddl::{create_sorted_index, drop_sorted_index}; pub use dispatch::{SortedIndexTarget, drop_in_engine}; +pub(crate) use query::run_read; pub use query::{select_range, select_rank, select_sorted_count, select_topk}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/query.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/query.rs index 24e4f6947..97f22a1c7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/query.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/query.rs @@ -4,24 +4,81 @@ //! //! Each names an index and returns keys, ranks, or counts drawn from the //! collection it was built over, so each resolves that collection and gates on -//! it before a plan is built (see [`super::gate`]). +//! it before a plan is built (see [`super::gate`]). The native sorted-index +//! read opcodes run through [`run_read`] too, so both protocols gate, route and +//! see the caller's transaction the same way. +use crate::bridge::envelope::Response; use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::shared::session::DmlTxnCtx; use crate::control::state::SharedState; use crate::types::DatabaseId; -use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; +use nodedb_physical::physical_plan::{PhysicalPlan, SortedIndexRead}; use super::super::super::result::{DdlError, DdlResult}; -use super::dispatch::{SortedIndexTarget, dispatch_and_respond_json, dispatch_and_respond_rows}; +use super::dispatch::{SortedIndexTarget, dispatch_read, respond_json, respond_rows}; use super::gate::gate_read; use super::parse::{ddl_err, parse_function_args, parse_score_arg, unquote}; +use super::txn_read::{ReadScope, plan_read}; + +/// What a read delivers instead of row bodies, for the refusal message. +fn what(read: &SortedIndexRead) -> &'static str { + match read { + SortedIndexRead::Rank { .. } => { + "RANK(), which returns a position in the sorted index rather than rows" + } + SortedIndexRead::TopK { .. } => { + "TOPK(), which returns the sorted index's ranked keys rather than rows" + } + SortedIndexRead::Range { .. } => { + "RANGE(), which returns the sorted index's ranked keys rather than rows" + } + SortedIndexRead::Count => { + "SORTED_COUNT(), which returns a count over the sorted index rather than rows" + } + SortedIndexRead::Score { .. } => { + "a sorted-index score read, which returns a sort key rather than rows" + } + } +} -/// What each function delivers instead of row bodies, for the refusal message. -const RANK_WHAT: &str = "RANK(), which returns a position in the sorted index rather than rows"; -const TOPK_WHAT: &str = "TOPK(), which returns the sorted index's ranked keys rather than rows"; -const RANGE_WHAT: &str = "RANGE(), which returns the sorted index's ranked keys rather than rows"; -const COUNT_WHAT: &str = - "SORTED_COUNT(), which returns a count over the sorted index rather than rows"; +/// Gate, plan and dispatch one sorted-index read. +/// +/// Gated on the index's owning collection, routed to the core that holds its +/// rows, and run in the caller's transaction when one is open. Returns the +/// plan that ran and the Data Plane reply. +pub(crate) async fn run_read( + state: &SharedState, + identity: &AuthenticatedIdentity, + database_id: DatabaseId, + txn_ctx: &DmlTxnCtx<'_>, + index_name: &str, + read: SortedIndexRead, +) -> Result<(PhysicalPlan, Response), DdlError> { + let collection = gate_read(state, identity, database_id, index_name, what(&read))?; + let sorted = plan_read( + &ReadScope { + txn_ctx, + tenant_id: identity.tenant_id, + database_id, + collection: &collection, + index_name, + }, + read, + ); + let plan = sorted.plan.clone(); + let response = dispatch_read( + state, + &SortedIndexTarget { + tenant_id: identity.tenant_id, + database_id, + collection: &collection, + }, + sorted, + ) + .await?; + Ok((plan, response)) +} /// Handle `SELECT RANK(index_name, 'key_value')` pub async fn select_rank( @@ -29,6 +86,7 @@ pub async fn select_rank( identity: &AuthenticatedIdentity, database_id: DatabaseId, sql: &str, + txn_ctx: &DmlTxnCtx<'_>, ) -> Result, DdlError> { let args = parse_function_args(sql)?; if args.len() < 2 { @@ -41,26 +99,11 @@ pub async fn select_rank( // A string literal is data: the index name resolves exactly as written, // with no case folding. let index_name = unquote(&args[0]); - let key_value = unquote(&args[1]); - - let collection = gate_read(state, identity, database_id, &index_name, RANK_WHAT)?; + let primary_key = unquote(&args[1]).into_bytes(); - let plan = PhysicalPlan::Kv(KvOp::SortedIndexRank { - index_name, - primary_key: key_value.into_bytes(), - }); - - dispatch_and_respond_json( - state, - &SortedIndexTarget { - tenant_id: identity.tenant_id, - database_id, - collection: &collection, - }, - plan, - "rank", - ) - .await + let read = SortedIndexRead::Rank { primary_key }; + let (_, response) = run_read(state, identity, database_id, txn_ctx, &index_name, read).await?; + Ok(respond_json(&response, "rank")) } /// Handle `SELECT * FROM TOPK(index_name, k)` or `SELECT TOPK(index_name, k)` @@ -69,6 +112,7 @@ pub async fn select_topk( identity: &AuthenticatedIdentity, database_id: DatabaseId, sql: &str, + txn_ctx: &DmlTxnCtx<'_>, ) -> Result, DdlError> { let args = parse_function_args(sql)?; if args.len() < 2 { @@ -88,20 +132,9 @@ pub async fn select_topk( ) })?; - let collection = gate_read(state, identity, database_id, &index_name, TOPK_WHAT)?; - - let plan = PhysicalPlan::Kv(KvOp::SortedIndexTopK { index_name, k }); - - dispatch_and_respond_rows( - state, - &SortedIndexTarget { - tenant_id: identity.tenant_id, - database_id, - collection: &collection, - }, - plan, - ) - .await + let read = SortedIndexRead::TopK { k }; + let (_, response) = run_read(state, identity, database_id, txn_ctx, &index_name, read).await?; + respond_rows(&response) } /// Handle `SELECT * FROM RANGE(index_name, score_min, score_max)` @@ -110,6 +143,7 @@ pub async fn select_range( identity: &AuthenticatedIdentity, database_id: DatabaseId, sql: &str, + txn_ctx: &DmlTxnCtx<'_>, ) -> Result, DdlError> { let args = parse_function_args(sql)?; if args.len() < 3 { @@ -122,27 +156,12 @@ pub async fn select_range( // A string literal is data: the index name resolves exactly as written, // with no case folding. let index_name = unquote(&args[0]); - let score_min = parse_score_arg(&args[1]); - let score_max = parse_score_arg(&args[2]); - - let collection = gate_read(state, identity, database_id, &index_name, RANGE_WHAT)?; - - let plan = PhysicalPlan::Kv(KvOp::SortedIndexRange { - index_name, - score_min, - score_max, - }); - - dispatch_and_respond_rows( - state, - &SortedIndexTarget { - tenant_id: identity.tenant_id, - database_id, - collection: &collection, - }, - plan, - ) - .await + let read = SortedIndexRead::Range { + score_min: parse_score_arg(&args[1]), + score_max: parse_score_arg(&args[2]), + }; + let (_, response) = run_read(state, identity, database_id, txn_ctx, &index_name, read).await?; + respond_rows(&response) } /// Handle `SELECT SORTED_COUNT(index_name)` @@ -151,6 +170,7 @@ pub async fn select_sorted_count( identity: &AuthenticatedIdentity, database_id: DatabaseId, sql: &str, + txn_ctx: &DmlTxnCtx<'_>, ) -> Result, DdlError> { let args = parse_function_args(sql)?; if args.is_empty() { @@ -163,20 +183,14 @@ pub async fn select_sorted_count( // A string literal is data: the index name resolves exactly as written, // with no case folding. let index_name = unquote(&args[0]); - - let collection = gate_read(state, identity, database_id, &index_name, COUNT_WHAT)?; - - let plan = PhysicalPlan::Kv(KvOp::SortedIndexCount { index_name }); - - dispatch_and_respond_json( + let (_, response) = run_read( state, - &SortedIndexTarget { - tenant_id: identity.tenant_id, - database_id, - collection: &collection, - }, - plan, - "sorted_count", + identity, + database_id, + txn_ctx, + &index_name, + SortedIndexRead::Count, ) - .await + .await?; + Ok(respond_json(&response, "sorted_count")) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/txn_read.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/txn_read.rs new file mode 100644 index 000000000..619285984 --- /dev/null +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/txn_read.rs @@ -0,0 +1,212 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The plan a sorted-index read runs, and the transaction it runs in. +//! +//! Outside a transaction block a read asks the index's registered tree. +//! Inside one it becomes `KvOp::SortedIndexTxnRead`. The Data Plane answers +//! that from a transaction-local tree over the collection's base rows and the +//! transaction's staged writes. An index this transaction created has no tree +//! before COMMIT, so its definition travels with the read. The definition +//! comes from the index build that the transaction deferred to COMMIT. + +use nodedb_physical::physical_plan::{KvOp, PhysicalPlan, SortedIndexRead, SortedIndexSpec}; +use nodedb_types::QualifiedCollection; + +use crate::control::server::shared::session::ddl_buffer; +use crate::control::server::shared::session::ddl_effect::DeferredDdlEffect; +use crate::control::server::shared::session::{DmlTxnCtx, TransactionState}; +use crate::types::{DatabaseId, TenantId, TxnId}; + +/// One sorted-index read: the plan and the transaction whose overlay it reads. +pub(super) struct SortedRead { + pub plan: PhysicalPlan, + pub txn_id: Option, +} + +/// Where a read runs and which index it names. +pub(super) struct ReadScope<'a> { + pub txn_ctx: &'a DmlTxnCtx<'a>, + pub tenant_id: TenantId, + pub database_id: DatabaseId, + /// The collection the index covers, resolved by the read gate. + pub collection: &'a str, + pub index_name: &'a str, +} + +/// The plan for `read`. +pub(super) fn plan_read(scope: &ReadScope<'_>, read: SortedIndexRead) -> SortedRead { + let txn_ctx = scope.txn_ctx; + if txn_ctx.sessions.transaction_state(txn_ctx.session_id) != TransactionState::InBlock { + return SortedRead { + plan: PhysicalPlan::Kv(autocommit_op(scope.index_name, read)), + txn_id: None, + }; + } + SortedRead { + plan: PhysicalPlan::Kv(KvOp::SortedIndexTxnRead { + collection: QualifiedCollection::new(scope.database_id, scope.collection), + index_name: scope.index_name.to_string(), + pending: pending_definition(scope.tenant_id, scope.database_id, scope.index_name), + read, + }), + txn_id: txn_ctx.sessions.tx_id(txn_ctx.session_id), + } +} + +/// The read outside a transaction block, against the registered tree. +fn autocommit_op(index_name: &str, read: SortedIndexRead) -> KvOp { + let index_name = index_name.to_string(); + match read { + SortedIndexRead::Rank { primary_key } => KvOp::SortedIndexRank { + index_name, + primary_key, + }, + SortedIndexRead::TopK { k } => KvOp::SortedIndexTopK { index_name, k }, + SortedIndexRead::Range { + score_min, + score_max, + } => KvOp::SortedIndexRange { + index_name, + score_min, + score_max, + }, + SortedIndexRead::Count => KvOp::SortedIndexCount { index_name }, + SortedIndexRead::Score { primary_key } => KvOp::SortedIndexScore { + index_name, + primary_key, + }, + } +} + +/// The definition of `index_name` when this transaction created it and has +/// not committed it. `None` for a committed index. +/// +/// Replays the deferred effects in statement order: a later drop of the same +/// name cancels an earlier create. +fn pending_definition( + tenant_id: TenantId, + database_id: DatabaseId, + index_name: &str, +) -> Option { + ddl_buffer::with_buffered(|items| { + let mut pending = None; + for effect in items.iter().flat_map(|item| item.effects.iter()) { + pending = step(pending, effect, tenant_id, database_id, index_name); + } + pending + }) + .flatten() +} + +/// Apply one deferred effect to the pending definition resolved so far. +fn step( + current: Option, + effect: &DeferredDdlEffect, + tenant_id: TenantId, + database_id: DatabaseId, + index_name: &str, +) -> Option { + match effect { + DeferredDdlEffect::SortedIndexRegister { + tenant_id: effect_tenant, + database_id: effect_db, + plan: + PhysicalPlan::Kv(KvOp::RegisterSortedIndex { + index_name: effect_index, + sort_columns, + key_column, + window_type, + window_timestamp_column, + window_start_ms, + window_end_ms, + .. + }), + .. + } if *effect_tenant == tenant_id + && *effect_db == database_id + && effect_index == index_name => + { + Some(SortedIndexSpec { + sort_columns: sort_columns.clone(), + key_column: key_column.clone(), + window_type: window_type.clone(), + window_timestamp_column: window_timestamp_column.clone(), + window_start_ms: *window_start_ms, + window_end_ms: *window_end_ms, + }) + } + DeferredDdlEffect::SortedIndexDrop { + tenant_id: effect_tenant, + database_id: effect_db, + index_name: effect_index, + .. + } if *effect_tenant == tenant_id + && *effect_db == database_id + && effect_index == index_name => + { + None + } + _ => current, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn register(index_name: &str) -> DeferredDdlEffect { + DeferredDdlEffect::SortedIndexRegister { + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + collection: "board".to_string(), + plan: PhysicalPlan::Kv(KvOp::RegisterSortedIndex { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "board"), + index_name: index_name.to_string(), + sort_columns: vec![("score".to_string(), "DESC".to_string())], + key_column: "id".to_string(), + window_type: "none".to_string(), + window_timestamp_column: String::new(), + window_start_ms: 0, + window_end_ms: 0, + }), + } + } + + fn drop_effect(index_name: &str) -> DeferredDdlEffect { + DeferredDdlEffect::SortedIndexDrop { + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + collection: "board".to_string(), + index_name: index_name.to_string(), + } + } + + fn replay(effects: &[DeferredDdlEffect], index_name: &str) -> Option { + effects.iter().fold(None, |current, effect| { + step( + current, + effect, + TenantId::new(1), + DatabaseId::DEFAULT, + index_name, + ) + }) + } + + #[test] + fn a_buffered_create_carries_its_definition() { + let spec = replay(&[register("lb")], "lb").expect("the create is pending"); + assert_eq!(spec.key_column, "id"); + assert_eq!( + spec.sort_columns, + vec![("score".to_string(), "DESC".to_string())] + ); + } + + #[test] + fn a_later_drop_cancels_the_create_and_other_names_are_ignored() { + assert!(replay(&[register("lb"), drop_effect("lb")], "lb").is_none()); + assert!(replay(&[register("other")], "lb").is_none()); + assert!(replay(&[drop_effect("lb"), register("lb")], "lb").is_some()); + } +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/last_value.rs b/nodedb/src/control/server/shared/ddl/neutral/last_value.rs index 9d12a5c17..4d48c6e8e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/last_value.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/last_value.rs @@ -51,7 +51,7 @@ pub async fn query_last_values( }, ) .await - .map_err(|e| ddl_err("XX000", format!("dispatch failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("dispatch failed", &e))?; // `meta_query_last_values` encodes with `response_codec::encode`, which is // MessagePack — `decode_payload` is its counterpart. A JSON parser on those @@ -60,7 +60,7 @@ pub async fn query_last_values( // them. let entries: Vec<(u64, i64, f64)> = crate::data::executor::response_codec::decode_payload(&payload) - .map_err(|e| ddl_err("XX000", format!("LAST_VALUES reply: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("LAST_VALUES reply", &e))?; let mut rows = Vec::with_capacity(entries.len()); for (series_id, ts, value) in &entries { @@ -120,13 +120,13 @@ pub async fn query_last_value( }, ) .await - .map_err(|e| ddl_err("XX000", format!("dispatch failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("dispatch failed", &e))?; // MessagePack, as in `query_last_values` above — an absent series is // encoded as a null (decoding to `None`), which is a different fact from a // payload that could not be read at all. let entry: Option<(i64, f64)> = crate::data::executor::response_codec::decode_payload(&payload) - .map_err(|e| ddl_err("XX000", format!("LAST_VALUE reply: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("LAST_VALUE reply", &e))?; let mut rows = Vec::new(); if let Some((ts, value)) = entry { @@ -148,7 +148,3 @@ pub async fn query_last_value( rows, ))]) } - -fn ddl_err(sqlstate: &str, message: impl Into) -> DdlError { - DdlError::new(sqlstate, message) -} diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/analyze.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/analyze.rs index 50a3c28d3..a0270d05a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/analyze.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/analyze.rs @@ -48,7 +48,7 @@ pub async fn handle_analyze( let coll = catalog .get_collection(database_id, tenant_id, &collection) - .map_err(|e| ddl_err("XX000", format!("catalog error: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog error", &e))? .ok_or_else(|| { ddl_err( "42P01", @@ -89,7 +89,7 @@ pub async fn handle_analyze( TraceId::ZERO, ) .await - .map_err(|error| ddl_err("XX000", format!("ANALYZE scan failed: {error}")))?; + .map_err(|error| DdlError::from_error_in_context("ANALYZE scan failed", &error))?; if !resp.payload.is_empty() { let json = crate::data::executor::response_codec::decode_payload_to_json(&resp.payload); push_scan_rows(&json, &mut rows); @@ -141,7 +141,7 @@ pub async fn handle_analyze( super::super::replicate::propose_and_apply(state, &entry, || { catalog .put_column_stats_batch(&local_rows) - .map_err(|e| ddl_err("XX000", format!("failed to store column stats: {e}"))) + .map_err(|e| DdlError::from_error_in_context("failed to store column stats", &e)) })?; state diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/auto_analyze.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/auto_analyze.rs index 36eefb70f..f4e40b619 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/auto_analyze.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/auto_analyze.rs @@ -23,7 +23,6 @@ use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; use super::super::super::result::DdlError; -use super::support::ddl_err; /// Duration estimate handed to the maintenance budget pre-screen. /// @@ -273,10 +272,7 @@ fn blocking_analyze( collection: &str, ) -> Result<(), DdlError> { let handle = tokio::runtime::Handle::try_current().map_err(|error| { - ddl_err( - "XX000", - format!("auto-ANALYZE needs a Tokio runtime: {error}"), - ) + DdlError::internal(format!("auto-ANALYZE needs a Tokio runtime: {error}")) })?; // `handle_analyze` reads the collection name off the second whitespace // token and lowercases it, so the bare name is what it expects. diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/reindex.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/reindex.rs index 9e6b98614..2a517dd17 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/reindex.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/reindex.rs @@ -5,9 +5,17 @@ //! Grammar: //! REINDEX [INDEX ] [CONCURRENTLY] //! -//! Non-concurrent path: dispatches `MetaOp::Checkpoint` (existing semantics). -//! Concurrent path: dispatches `MetaOp::RebuildIndex { concurrent: true }` to -//! every core and awaits the cross-core ACK barrier before returning. +//! Both forms dispatch `MetaOp::RebuildIndex` to every core and await the +//! cross-core ACK barrier. Without `INDEX ` every index the +//! collection has is rebuilt: HNSW, full-text and CSR. +//! +//! Both forms rebuild the same way: off the core, with the core serving +//! reads and writes meanwhile and swapping each rebuilt index in on a later +//! tick. They differ only in when a core answers: +//! +//! - Non-concurrent: once its cutovers are done. Past the statement +//! deadline it answers `DeadlineExceeded`; the rebuilds still complete. +//! - Concurrent: once its rebuilds have started. //! //! The grammar is parsed once by `nodedb_sql::ddl_ast::parse` into //! `NodedbStatement::Reindex { .. }`; this handler receives the already-parsed @@ -53,39 +61,26 @@ pub async fn handle_reindex( )); } - if concurrent { - // Concurrent path: broadcast to all cores and await per-core ACK. - let plan = crate::bridge::envelope::PhysicalPlan::Meta(MetaOp::RebuildIndex { - collection: nodedb_types::QualifiedCollection::new(database_id, &collection), - index_name, - concurrent: true, - }); - let trace_id = TraceId::generate(); - crate::control::server::broadcast::broadcast_register_to_all_cores( - state, - tenant_id, - database_id, - plan, - trace_id, - ) - .await - .map_err(|e| ddl_err("XX000", format!("REINDEX CONCURRENTLY failed: {e}")))?; + // Every core rebuilds the indexes it holds for the collection. A core + // answers the plain form after its cutovers, the concurrent form after + // its rebuilds start. + let plan = crate::bridge::envelope::PhysicalPlan::Meta(MetaOp::RebuildIndex { + collection: nodedb_types::QualifiedCollection::new(database_id, &collection), + index_name, + concurrent, + }); + let trace_id = TraceId::generate(); + crate::control::server::broadcast::broadcast_register_to_all_cores( + state, + tenant_id, + database_id, + plan, + trace_id, + ) + .await + .map_err(|e| DdlError::from_error_in_context("REINDEX failed", &e))?; - tracing::info!( - %collection, - concurrent = true, - "REINDEX CONCURRENTLY dispatched and acknowledged by all cores" - ); - } else { - // Non-concurrent path: fire-and-forget (same as legacy Checkpoint). - super::distributed::dispatch_maintenance_to_all_cores( - state, - tenant_id, - database_id, - MetaOp::Checkpoint, - ); - tracing::info!(%collection, concurrent = false, "REINDEX dispatched"); - } + tracing::info!(%collection, concurrent, "REINDEX acknowledged by all cores"); Ok(vec![DdlResult::Status { command: "REINDEX".to_string(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs index faf24988b..e63e2e2f7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs @@ -43,7 +43,7 @@ pub async fn handle_show_vector_index( // or: SHOW VECTOR INDEX status ON let (collection, field_name) = parse_collection_column(sql, " ON ")?; let tenant_id = identity.tenant_id; - let vshard = crate::types::VShardId::from_collection_in_database(database_id, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(); let plan = PhysicalPlan::Vector(VectorOp::QueryStats { collection: nodedb_types::QualifiedCollection::new(database_id, &collection), @@ -59,7 +59,7 @@ pub async fn handle_show_vector_index( TraceId::ZERO, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if resp.payload.is_empty() { return Err(ddl_err( @@ -69,11 +69,11 @@ pub async fn handle_show_vector_index( } let stats: nodedb_types::VectorIndexStats = zerompk::from_msgpack(&resp.payload) - .map_err(|e| ddl_err("XX000", format!("decode vector stats: {e}")))?; + .map_err(|e| DdlError::internal(format!("decode vector stats: {e}")))?; let columns = vec!["property".to_string(), "value".to_string()]; - let pairs: Vec<(&str, String)> = vec![ + let mut pairs: Vec<(&str, String)> = vec![ ("dimensions", stats.dimensions.to_string()), ("metric", stats.metric.clone()), ("index_type", stats.index_type.to_string()), @@ -94,6 +94,9 @@ pub async fn handle_show_vector_index( format!("{:.1}", stats.disk_bytes as f64 / (1024.0 * 1024.0)), ), ("build_in_progress", stats.build_in_progress.to_string()), + ("builds_queued", stats.builds_queued.to_string()), + ("builds_completed", stats.builds_completed.to_string()), + ("builds_failed", stats.builds_failed.to_string()), ("hnsw_m", stats.hnsw_m.to_string()), ("hnsw_m0", stats.hnsw_m0.to_string()), ( @@ -103,6 +106,17 @@ pub async fn handle_show_vector_index( ("seal_threshold", stats.seal_threshold.to_string()), ("mmap_segments", stats.mmap_segment_count.to_string()), ]; + if let Some(ivf) = &stats.ivf { + pairs.extend([ + ("ivf_training_threshold", ivf.training_threshold.to_string()), + ("ivf_trained", ivf.trained.to_string()), + ("ivf_trained_on", ivf.trained_on.to_string()), + ("ivf_trained_at_ms", ivf.trained_at_ms.to_string()), + ("ivf_indexed_vectors", ivf.indexed_vectors.to_string()), + ("ivf_cells", ivf.cells.to_string()), + ("ivf_nprobe", ivf.nprobe.to_string()), + ]); + } let rows: Vec> = pairs .into_iter() @@ -126,7 +140,7 @@ pub async fn handle_alter_vector_index_seal( ) -> Result, DdlError> { let (collection, field_name) = parse_collection_column(sql, " ON ")?; let tenant_id = identity.tenant_id; - let vshard = crate::types::VShardId::from_collection_in_database(database_id, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(); let plan = PhysicalPlan::Vector(VectorOp::Seal { collection: nodedb_types::QualifiedCollection::new(database_id, &collection), @@ -142,7 +156,7 @@ pub async fn handle_alter_vector_index_seal( TraceId::ZERO, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; Ok(vec![DdlResult::Status { command: "SEAL".to_string(), @@ -159,7 +173,7 @@ pub async fn handle_alter_vector_index_compact( ) -> Result, DdlError> { let (collection, field_name) = parse_collection_column(sql, " ON ")?; let tenant_id = identity.tenant_id; - let vshard = crate::types::VShardId::from_collection_in_database(database_id, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(); let plan = PhysicalPlan::Vector(VectorOp::CompactIndex { collection: nodedb_types::QualifiedCollection::new(database_id, &collection), @@ -175,7 +189,7 @@ pub async fn handle_alter_vector_index_compact( TraceId::ZERO, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; Ok(vec![DdlResult::Status { command: "COMPACT".to_string(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index_set.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index_set.rs index 5250632d3..b743cf1d7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index_set.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index_set.rs @@ -60,7 +60,7 @@ pub async fn handle_alter_vector_index_set( &collection, &field_name, ) - .map_err(|e| ddl_err("XX000", format!("read vector index params: {e}")))? + .map_err(|e| DdlError::from_error_in_context("read vector index params", &e))? .ok_or_else(|| { ddl_err( "42704", @@ -79,7 +79,11 @@ pub async fn handle_alter_vector_index_set( // Single node: no applier runs, so post-apply never fires. Run the // per-node install the post-apply lane runs everywhere else. if outcome.needs_local_apply() { - crate::control::catalog_entry::post_apply::install_vector_index_params(merged, state).await; + let shared = state + .self_arc() + .map_err(|e| DdlError::from_error_in_context("install vector index params", &e))?; + crate::control::catalog_entry::post_apply::install_vector_index_params(merged, shared) + .await; } state.audit_record( diff --git a/nodedb/src/control/server/shared/ddl/neutral/match_ops.rs b/nodedb/src/control/server/shared/ddl/neutral/match_ops.rs index ab45188ef..e8229b71d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/match_ops.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/match_ops.rs @@ -135,7 +135,7 @@ pub async fn match_query( // Serialize the MatchQuery for SPSC transport. let query_bytes = zerompk::to_msgpack_vec(&query) - .map_err(|e| DdlError::new("XX000", format!("serialize match query: {e}")))?; + .map_err(|e| DdlError::internal(format!("serialize match query: {e}")))?; let tenant_id = identity.tenant_id; @@ -189,7 +189,7 @@ pub async fn match_query( match_payload_to_rows(&outcome.rows_payload, &column_names) } } - Err(e) => Err(DdlError::new("XX000", e.to_string())), + Err(e) => Err(DdlError::from_error(&e)), }; } @@ -222,7 +222,7 @@ pub async fn match_query( match_payload_to_rows(&outcome.rows_payload, &column_names) } } - Err(e) => Err(DdlError::new("XX000", e.to_string())), + Err(e) => Err(DdlError::from_error(&e)), } } @@ -242,7 +242,7 @@ fn match_payload_to_rows( let json_text = response_codec::decode_payload_to_json(payload); let rows: Vec = sonic_rs::from_str(&json_text) - .map_err(|e| DdlError::new("XX000", format!("invalid match result JSON: {e}")))?; + .map_err(|e| DdlError::internal(format!("invalid match result JSON: {e}")))?; let mut out_rows = Vec::with_capacity(rows.len()); for row in &rows { diff --git a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/create.rs b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/create.rs index e2ee74129..b72f90b88 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/create.rs @@ -81,7 +81,7 @@ pub async fn create_materialized_view( // failed. if catalog .get_materialized_view(database_id.as_u64(), tenant_id.as_u64(), &name) - .map_err(|error| err("XX000", error.to_string()))? + .map_err(|error| DdlError::from_error(&error))? .is_some() { return Err(err( @@ -91,7 +91,7 @@ pub async fn create_materialized_view( } if catalog .get_collection(database_id, tenant_id.as_u64(), &name) - .map_err(|error| err("XX000", error.to_string()))? + .map_err(|error| DdlError::from_error(&error))? .is_some() { return Err(err("42P07", format!("collection '{name}' already exists"))); @@ -178,7 +178,7 @@ pub async fn create_materialized_view( propose_and_apply(state, &coll_entry)?; super::super::collection::dispatch_register_from_stored(state, &target) .await - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; tracing::info!( view = name, @@ -277,7 +277,7 @@ async fn create_streaming_mv( Box::new(def.clone()), ); let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|error| err("XX000", format!("metadata propose: {error}")))?; + .map_err(|error| DdlError::from_error_in_context("metadata propose", &error))?; crate::control::catalog_entry::apply::local::apply_locally_if_needed(state, &entry, outcome); if outcome.needs_local_apply() { state.permissions.install_replicated_owner( diff --git a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/drop.rs index b7b0c57a1..6b3daf6df 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/drop.rs @@ -70,7 +70,7 @@ pub fn drop_materialized_view( name: name.clone(), }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|error| err("XX000", format!("metadata propose: {error}")))?; + .map_err(|error| DdlError::from_error_in_context("metadata propose", &error))?; crate::control::catalog_entry::apply::local::apply_locally_if_needed( state, &entry, outcome, ); @@ -136,14 +136,14 @@ pub fn drop_materialized_view( None }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|error| err("XX000", format!("metadata propose: {error}")))?; + .map_err(|error| DdlError::from_error_in_context("metadata propose", &error))?; if outcome.needs_local_apply() { // No metadata Raft is active, so apply the same compound catalog // deletion locally and synchronously reclaim the implementation-owned // target collection. A reclaim failure after catalog deletion is // fatal: continuing would permit a same-name CREATE over stale rows. crate::control::catalog_entry::apply::apply_to(&entry, state.credentials.catalog()) - .map_err(|e| err("XX000", format!("catalog apply: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog apply", &e))?; let purge_lsn = state.wal.next_lsn().as_u64(); let purge_result = tokio::task::block_in_place(|| { tokio::runtime::Handle::current().block_on(async { diff --git a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/refresh.rs b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/refresh.rs index 89d9ee55f..1dfc83369 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/refresh.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/refresh.rs @@ -306,21 +306,38 @@ async fn dispatch_sql( checked } }; - crate::control::server::wal_dispatch::wal_append_if_write( + // The record's outcome-floor window opens before the append and + // closes from the task's outcome inside the funnel. + let owner = crate::control::server::dispatch_utils::RecordOwner { + tenant_id: identity.tenant_id, + database_id: checked.database_id(), + vshard_id: checked.vshard_id(), + }; + let minted = + crate::control::server::dispatch_utils::MintedRecords::open(&state.outcome_floor); + if let Err(e) = minted.append_plan( &state.wal, - identity.tenant_id, - checked.vshard_id(), - checked.database_id(), + owner, checked.plan(), - ) - .map_err(|e| err(sqlstate::IO_ERROR, format!("wal append: {e}")))?; - let response = crate::control::server::dispatch_utils::dispatch_authorized_to_data_plane( - state, - checked, - TraceId::ZERO, - ) - .await - .map_err(|e| err(sqlstate::CONNECTION_FAILURE, format!("dispatch: {e}")))?; + // The refresh is dispatched as a client write. + crate::event::EventSource::User, + ) { + // Any record appended before the error never reaches a core. + minted + .cancel(&state.wal, owner, 0) + .await + .map_err(|c| err(sqlstate::IO_ERROR, format!("cancel refresh record: {c}")))?; + return Err(err(sqlstate::IO_ERROR, format!("wal append: {e}"))); + } + let response = + crate::control::server::dispatch_utils::dispatch_authorized_minted_to_data_plane( + state, + checked, + TraceId::ZERO, + minted, + ) + .await + .map_err(|e| err(sqlstate::CONNECTION_FAILURE, format!("dispatch: {e}")))?; require_ok_response(&response)?; } Ok(()) diff --git a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/show.rs b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/show.rs index b3f1db060..efbb38f3c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/show.rs @@ -19,10 +19,6 @@ use crate::types::DatabaseId; use super::super::super::result::{DdlError, DdlResult}; -fn err(sqlstate: &str, message: String) -> DdlError { - DdlError::new(sqlstate, message) -} - pub fn show_materialized_views( state: &SharedState, identity: &AuthenticatedIdentity, @@ -49,7 +45,7 @@ pub fn show_materialized_views( .credentials .catalog() .list_materialized_views(database_id.as_u64(), tenant_id.as_u64()) - .map_err(|e| err("XX000", format!("catalog read failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog read failed", &e))?; let mut rows = Vec::new(); for view in &views { diff --git a/nodedb/src/control/server/shared/ddl/neutral/mod.rs b/nodedb/src/control/server/shared/ddl/neutral/mod.rs index 7d6d82e97..eae9717e6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/mod.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/mod.rs @@ -27,6 +27,7 @@ pub mod convert; pub mod crdt_ops; pub mod custom_type; pub mod database; +pub mod deferred_effects; pub mod dsl; pub mod emergency_ddl; pub mod estimate_count; @@ -63,6 +64,7 @@ pub mod replicate; pub mod retention_policy; pub mod rls; pub mod role; +mod role_checks; pub mod router; pub mod schedule; pub mod scope_ddl; diff --git a/nodedb/src/control/server/shared/ddl/neutral/oidc.rs b/nodedb/src/control/server/shared/ddl/neutral/oidc.rs index 44820fae6..192188157 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/oidc.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/oidc.rs @@ -70,7 +70,14 @@ fn has_ambiguous_issuer_route(existing_audience: Option<&str>, audience: Option< } } -fn validate_claim_mapping_roles(claim_mappings: &[OidcClaimMappingClause]) -> Result<(), DdlError> { +/// Refuse a claim mapping that grants superuser, or a role that is neither +/// built in nor defined in the provider's tenant: a login mapped to it would +/// hold nothing. A role dropped after this check refuses the login instead. +fn validate_claim_mapping_roles( + state: &SharedState, + claim_mappings: &[OidcClaimMappingClause], + tenant_id: Option, +) -> Result<(), DdlError> { if claim_mappings .iter() .flat_map(|mapping| mapping.add_roles.iter()) @@ -81,6 +88,19 @@ fn validate_claim_mapping_roles(claim_mappings: &[OidcClaimMappingClause]) -> Re "OIDC claim mappings cannot grant the database-owned superuser role", )); } + if let Some(tenant_id) = tenant_id { + let roles: Vec = claim_mappings + .iter() + .flat_map(|mapping| mapping.add_roles.iter()) + .map(String::as_str) + .map(crate::control::security::role_assignment::parse_role_name) + .collect(); + super::role_checks::check_user_roles( + state, + &roles, + crate::types::TenantId::new(tenant_id), + )?; + } Ok(()) } @@ -116,13 +136,12 @@ pub fn create_oidc_provider( if jwks_uri.is_empty() { return Err(DdlError::new("22023", "JWKS_URI must not be empty")); } - validate_claim_mapping_roles(claim_mappings)?; let catalog = state.credentials.catalog(); let tenant_exists = catalog .load_all_tenants() - .map_err(|e| DdlError::new("XX000", format!("tenant lookup: {e}")))? + .map_err(|e| DdlError::from_error_in_context("tenant lookup", &e))? .iter() .any(|tenant| tenant.tenant_id == tenant_id); if !tenant_exists { @@ -131,6 +150,7 @@ pub fn create_oidc_provider( format!("tenant '{tenant_id}' does not exist"), )); } + validate_claim_mapping_roles(state, claim_mappings, Some(tenant_id))?; // Check for duplicate by provider name. match catalog.get_oidc_provider(name) { @@ -142,7 +162,7 @@ pub fn create_oidc_provider( } Ok(None) => {} Err(e) => { - return Err(DdlError::new("XX000", format!("catalog read: {e}"))); + return Err(DdlError::from_error_in_context("catalog read", &e)); } } @@ -162,7 +182,7 @@ pub fn create_oidc_provider( } } Err(e) => { - return Err(DdlError::new("XX000", format!("catalog list: {e}"))); + return Err(DdlError::from_error_in_context("catalog list", &e)); } } @@ -189,11 +209,11 @@ pub fn create_oidc_provider( let entry = CatalogEntry::PutOidcProvider(Box::new(provider.clone())); let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { catalog .put_oidc_provider(&provider) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } state.audit_record( @@ -221,9 +241,9 @@ pub fn alter_oidc_provider_claim_mapping( let mut provider = catalog .get_oidc_provider(name) - .map_err(|e| DdlError::new("XX000", format!("catalog read: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog read", &e))? .ok_or_else(|| DdlError::new("42704", format!("OIDC provider '{name}' does not exist")))?; - validate_claim_mapping_roles(claim_mappings)?; + validate_claim_mapping_roles(state, claim_mappings, provider.tenant_id)?; let stored_mappings: Vec = claim_mappings .iter() @@ -240,11 +260,11 @@ pub fn alter_oidc_provider_claim_mapping( let entry = CatalogEntry::PutOidcProvider(Box::new(provider.clone())); let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { catalog .put_oidc_provider(&provider) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } state.audit_record( @@ -273,7 +293,7 @@ pub fn drop_oidc_provider( if catalog .get_oidc_provider(name) - .map_err(|e| DdlError::new("XX000", format!("catalog read: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog read", &e))? .is_none() { if if_exists { @@ -289,11 +309,11 @@ pub fn drop_oidc_provider( name: name.to_string(), }; let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { catalog .delete_oidc_provider(name) - .map_err(|e| DdlError::new("XX000", format!("catalog delete: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; } state.audit_record( @@ -317,7 +337,7 @@ pub fn show_oidc_providers( let providers = catalog .list_oidc_providers() - .map_err(|e| DdlError::new("XX000", format!("catalog list: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog list", &e))?; let columns = vec![ "name".to_string(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/org_ddl.rs b/nodedb/src/control/server/shared/ddl/neutral/org_ddl.rs index 2c4fb528a..85b9b9b7f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/org_ddl.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/org_ddl.rs @@ -109,7 +109,7 @@ fn alter_org( let found = state .orgs .set_status(org_id, &status_val) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if !found { return Err(err("42704", format!("org '{org_id}' not found"))); } @@ -141,7 +141,7 @@ fn drop_org( let found = state .orgs .drop_org(org_id) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if !found { return Err(err("42704", format!("org '{org_id}' not found"))); } diff --git a/nodedb/src/control/server/shared/ddl/neutral/period_lock.rs b/nodedb/src/control/server/shared/ddl/neutral/period_lock.rs index e61fe8265..eb62a6e27 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/period_lock.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/period_lock.rs @@ -100,13 +100,13 @@ pub fn add_period_lock( let mut coll = catalog .get_collection(DatabaseId::DEFAULT, tenant_id, &name) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", format!("collection '{name}' not found")))?; coll.period_lock = Some(def); persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); @@ -141,13 +141,13 @@ pub fn drop_period_lock( let mut coll = catalog .get_collection(DatabaseId::DEFAULT, tenant_id, &name) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", format!("collection '{name}' not found")))?; coll.period_lock = None; persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); diff --git a/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs b/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs index 3710d381b..e101f7f0a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs @@ -79,7 +79,7 @@ pub async fn set_permission_tree( let catalog = state.credentials.catalog(); let mut coll = catalog .get_collection(DatabaseId::DEFAULT, tenant_id.as_u64(), &collection) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", format!("collection '{collection}' does not exist")))?; if !coll.is_active { @@ -91,10 +91,12 @@ pub async fn set_permission_tree( // Serialize and persist. let def_json = sonic_rs::to_string(&def) - .map_err(|e| err("XX000", format!("serialize PERMISSION_TREE: {e}")))?; + .map_err(|e| DdlError::internal(format!("serialize PERMISSION_TREE: {e}")))?; coll.permission_tree_def = Some(def_json); persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; + + let sources = [collection.clone(), def.permission_table.clone()]; // Update in-memory cache. state @@ -103,6 +105,13 @@ pub async fn set_permission_tree( .await .register_tree_def(tenant_id.as_u64(), &collection, def); + // Rows already in the sources are grants and edges too. Hold the + // acknowledgement until every lease holder covers each source group + // through its current commit. + source_group_barrier(state, &sources) + .await + .map_err(|e| DdlError::from_error(&e))?; + // Audit. state .audit @@ -118,6 +127,31 @@ pub async fn set_permission_tree( Ok(status("ALTER COLLECTION")) } +/// Barrier on every Raft group homing one of `sources`, at a read index +/// taken now. A single node has no groups; its planning reloads the cache. +async fn source_group_barrier(state: &SharedState, sources: &[String]) -> crate::Result<()> { + let Some(timing) = state.authorization_fence.timing() else { + return Ok(()); + }; + let mut targets: Vec = Vec::new(); + for source in sources { + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, source).vshard(); + let group_id = + crate::control::security::auth_fence::cluster::group_of_vshard(state, vshard.as_u32())?; + if targets.iter().any(|target| target.group_id == group_id) { + continue; + } + let through = crate::control::security::auth_fence::cluster::confirmed_read_index( + state, + group_id, + timing.lease, + ) + .await?; + targets.push(nodedb_cluster::GroupCoverage { group_id, through }); + } + crate::control::security::auth_lease::authorization_barrier(state, targets).await +} + /// ALTER COLLECTION DROP PERMISSION_TREE pub async fn drop_permission_tree( state: &SharedState, @@ -134,12 +168,12 @@ pub async fn drop_permission_tree( let catalog = state.credentials.catalog(); let mut coll = catalog .get_collection(DatabaseId::DEFAULT, tenant_id.as_u64(), &collection) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", format!("collection '{collection}' does not exist")))?; coll.permission_tree_def = None; persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; // Update in-memory cache. state diff --git a/nodedb/src/control/server/shared/ddl/neutral/planning.rs b/nodedb/src/control/server/shared/ddl/neutral/planning.rs index acf185a58..039a7aeee 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/planning.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/planning.rs @@ -36,7 +36,10 @@ pub async fn plan_authorized_sql( ) -> Result<(Vec, OutputSchema, QueryLeaseScope), DdlError> { // Internal DDL scans still plan in the caller-selected database context. let scope = RequestAuthScope::for_database(identity, state.auth_stores(), database_id); - let permission_cache = state.permission_cache.read().await; + let permission_cache = + crate::control::security::auth_fence::permission_view(state, identity.tenant_id) + .await + .map_err(|error| DdlError::from_error(&error))?; let sec = PlanSecurityContext { identity, auth: scope.auth(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/procedure/call.rs b/nodedb/src/control/server/shared/ddl/neutral/procedure/call.rs index 71bf5f7fe..7296fa088 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/procedure/call.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/procedure/call.rs @@ -34,7 +34,7 @@ pub async fn call_procedure( let proc = catalog .get_procedure_in_database(database_id, tenant_id.as_u64(), &name) - .map_err(|e| DdlError::new("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| DdlError::new("42883", format!("procedure '{name}' does not exist")))?; // Validate argument count matches IN parameters. diff --git a/nodedb/src/control/server/shared/ddl/neutral/procedure/create/handler.rs b/nodedb/src/control/server/shared/ddl/neutral/procedure/create/handler.rs index dd5e07d46..0a3bbff7f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/procedure/create/handler.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/procedure/create/handler.rs @@ -50,7 +50,7 @@ pub fn create_procedure( let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| DdlError::new("XX000", "system clock before UNIX epoch"))? + .map_err(|_| DdlError::internal("system clock before UNIX epoch"))? .as_secs(); let routability = extract_routability(&parsed.body_sql); diff --git a/nodedb/src/control/server/shared/ddl/neutral/procedure/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/procedure/drop.rs index eaad03580..fa75f4579 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/procedure/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/procedure/drop.rs @@ -57,7 +57,7 @@ pub fn drop_procedure( // a clean no-op that never touches raft. let exists_before = catalog .get_procedure_in_database(database_id, tenant_id, &name) - .map_err(|e| DdlError::new("XX000", format!("catalog read: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog read", &e))? .is_some(); if !exists_before && !if_exists { return Err(DdlError::new( @@ -75,11 +75,11 @@ pub fn drop_procedure( name: name.clone(), }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { let _ = catalog .delete_procedure_in_database(database_id, tenant_id, &name) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } // Broadcast deletion to connected Lite sessions. diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs index ca33bc44f..81e5261c6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs @@ -11,7 +11,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::dispatch_utils; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TraceId, VShardId}; +use crate::types::{DatabaseId, TraceId}; use super::super::super::result::{DdlError, DdlResult}; use super::super::read_gate::CollectionReadGate; @@ -51,12 +51,16 @@ pub async fn balance_as_of( gate.refuse_if_field_redacted(&collection, &column, "the as-of balance")?; // Read current balance from the target document. - let vshard = VShardId::from_collection_in_database(database_id, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(); let pk_bytes = key.as_bytes().to_vec(); let surrogate = state .surrogate_assigner - .lookup(database_id, tenant_id, &collection, &pk_bytes) - .map_err(|e| err("XX000", &format!("surrogate lookup failed: {e}")))? + .lookup( + nodedb_types::CollectionKey::from_bare(database_id, &collection), + tenant_id, + &pk_bytes, + ) + .map_err(|e| DdlError::from_error_in_context("surrogate lookup failed", &e))? .unwrap_or(nodedb_types::Surrogate::ZERO); let mut get_plan = PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::PointGet { @@ -79,7 +83,7 @@ pub async fn balance_as_of( TraceId::ZERO, ) .await - .map_err(|e| err("XX000", &format!("point get failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("point get failed", &e))?; let doc_json = crate::data::executor::response_codec::decode_payload_to_json(&get_resp.payload); let doc: serde_json::Value = sonic_rs::from_str(&doc_json).unwrap_or(serde_json::Value::Null); @@ -93,7 +97,7 @@ pub async fn balance_as_of( let catalog = state.credentials.catalog(); let coll = catalog .get_collection(database_id, tenant_id.as_u64(), &collection) - .map_err(|e| err("XX000", &e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", &format!("collection '{collection}' not found")))?; let Some(mat_def) = coll @@ -115,7 +119,7 @@ pub async fn balance_as_of( // Scan the source collection for rows where join_column = key AND created_at > as_of. let source_vshard = - VShardId::from_collection_in_database(database_id, &mat_def.source_collection); + nodedb_types::CollectionKey::from_bare(database_id, &mat_def.source_collection).vshard(); let mut source_scan = PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::Scan { collection: nodedb_types::QualifiedCollection::new( @@ -145,7 +149,7 @@ pub async fn balance_as_of( TraceId::ZERO, ) .await - .map_err(|e| err("XX000", &format!("source scan failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("source scan failed", &e))?; let source_json = crate::data::executor::response_codec::decode_payload_to_json(&source_resp.payload); @@ -166,7 +170,7 @@ pub async fn balance_as_of( let src_doc = serde_json::Value::Object(obj.clone()); let created_at = crate::data::executor::enforcement::retention::extract_created_at_secs( &sonic_rs::to_vec(&src_doc) - .map_err(|e| err("XX000", &format!("serialization failed: {e}")))?, + .map_err(|e| DdlError::internal(format!("serialization failed: {e}")))?, ); if let Some(ts) = created_at { if ts <= as_of_secs { diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs index 81fc44d0b..653c3c79c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs @@ -11,7 +11,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::dispatch_utils; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TraceId, VShardId}; +use crate::types::{DatabaseId, TraceId}; use super::super::super::result::{DdlError, DdlResult}; use super::super::read_gate::CollectionReadGate; @@ -72,7 +72,7 @@ pub async fn convert_currency_lookup( let key_value = format!("{from_ccy}/{to_ccy}"); // Scan rate table to find latest rate where key_column == key_value AND time_column <= as_of. - let vshard = VShardId::from_collection_in_database(database_id, &rate_table); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &rate_table).vshard(); let mut scan_plan = PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::Scan { collection: nodedb_types::QualifiedCollection::new(database_id, &rate_table), limit: usize::MAX, @@ -98,7 +98,7 @@ pub async fn convert_currency_lookup( TraceId::ZERO, ) .await - .map_err(|e| err("XX000", &format!("rate table scan failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("rate table scan failed", &e))?; let payload_json = crate::data::executor::response_codec::decode_payload_to_json(&scan_resp.payload); diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/helpers.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/helpers.rs index 28cc868e6..dc3141fdd 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/helpers.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/helpers.rs @@ -94,7 +94,7 @@ pub fn single_result(value: &str) -> Vec { pub fn unwrap_scan_docs(docs: Vec) -> Result>, DdlError> { let mut rows = Vec::with_capacity(docs.len()); for doc in docs { - push_flat_rows(Value::from(doc), &mut rows).map_err(|e| err("XX000", &e.to_string()))?; + push_flat_rows(Value::from(doc), &mut rows).map_err(|e| DdlError::from_error(&e))?; } Ok(rows.iter().map(row_to_wire_json).collect()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs index 2093d0195..333b23d33 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs @@ -10,7 +10,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::dispatch_utils; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TraceId, VShardId}; +use crate::types::{DatabaseId, TraceId}; use super::super::super::result::{DdlError, DdlResult}; use super::super::read_gate::CollectionReadGate; @@ -46,7 +46,7 @@ pub async fn temporal_lookup( gate.require_document_engine(&table, "TEMPORAL_LOOKUP")?; // Scan the table. - let vshard = VShardId::from_collection_in_database(database_id, &table); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &table).vshard(); let mut scan_plan = PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::Scan { collection: nodedb_types::QualifiedCollection::new(database_id, &table), limit: usize::MAX, @@ -72,7 +72,7 @@ pub async fn temporal_lookup( TraceId::ZERO, ) .await - .map_err(|e| err("XX000", &format!("scan failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("scan failed", &e))?; let payload_json = crate::data::executor::response_codec::decode_payload_to_json(&scan_resp.payload); diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_audit_chain.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_audit_chain.rs index 682996ef0..a8aa3617a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_audit_chain.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_audit_chain.rs @@ -71,7 +71,7 @@ pub async fn verify_audit_chain( let audit_entries = state .wal .recover_audit_entries() - .map_err(|e| err("XX000", &format!("audit WAL recovery failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("audit WAL recovery failed", &e))?; let mut valid = true; let mut checked = 0u64; diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs index b6e7a25fa..33c50bb46 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs @@ -11,7 +11,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::dispatch_utils; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TraceId, VShardId}; +use crate::types::{DatabaseId, TraceId}; use super::super::super::result::{DdlError, DdlResult}; use super::super::read_gate::CollectionReadGate; @@ -51,7 +51,7 @@ pub async fn verify_balance( let catalog = state.credentials.catalog(); let coll = catalog .get_collection(database_id, tenant_id.as_u64(), &collection) - .map_err(|e| err("XX000", &e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", &format!("collection '{collection}' not found")))?; let Some(mat_def) = coll @@ -70,7 +70,7 @@ pub async fn verify_balance( gate.refuse_if_any_redaction(&mat_def.source_collection, "the balance verification")?; // Scan all target rows. - let target_vshard = VShardId::from_collection_in_database(database_id, &collection); + let target_vshard = nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(); let mut target_scan = PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::Scan { collection: nodedb_types::QualifiedCollection::new(database_id, &collection), @@ -96,7 +96,7 @@ pub async fn verify_balance( TraceId::ZERO, ) .await - .map_err(|e| err("XX000", &format!("target scan failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("target scan failed", &e))?; let target_json = crate::data::executor::response_codec::decode_payload_to_json(&target_resp.payload); let target_docs: Vec = sonic_rs::from_str(&target_json) @@ -107,7 +107,7 @@ pub async fn verify_balance( // Scan all source rows. let source_vshard = - VShardId::from_collection_in_database(database_id, &mat_def.source_collection); + nodedb_types::CollectionKey::from_bare(database_id, &mat_def.source_collection).vshard(); let mut source_scan = PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::Scan { collection: nodedb_types::QualifiedCollection::new( @@ -136,7 +136,7 @@ pub async fn verify_balance( TraceId::ZERO, ) .await - .map_err(|e| err("XX000", &format!("source scan failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("source scan failed", &e))?; let source_json = crate::data::executor::response_codec::decode_payload_to_json(&source_resp.payload); let source_docs: Vec = sonic_rs::from_str(&source_json) diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs index 4acda9e41..e1bfea4a1 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs @@ -11,7 +11,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::dispatch_utils; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TraceId, VShardId}; +use crate::types::{DatabaseId, TraceId}; use super::super::super::result::{DdlError, DdlResult}; use super::super::read_gate::CollectionReadGate; @@ -44,7 +44,7 @@ pub async fn verify_hash_chain( gate.refuse_if_any_redaction(&collection, "the hash chain")?; // Scan all documents. - let vshard = VShardId::from_collection_in_database(database_id, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(); let mut scan_plan = PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::Scan { collection: nodedb_types::QualifiedCollection::new(database_id, &collection), limit: usize::MAX, @@ -70,7 +70,7 @@ pub async fn verify_hash_chain( TraceId::ZERO, ) .await - .map_err(|e| err("XX000", &format!("scan failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("scan failed", &e))?; let payload_json = crate::data::executor::response_codec::decode_payload_to_json(&scan_resp.payload); @@ -118,7 +118,7 @@ pub async fn verify_hash_chain( obj.remove("_chain_hash"); } let doc_bytes = sonic_rs::to_vec(&doc_for_hash) - .map_err(|e| err("XX000", &format!("failed to serialize document: {e}")))?; + .map_err(|e| DdlError::internal(format!("failed to serialize document: {e}")))?; let expected = crate::data::executor::enforcement::hash_chain::compute_chain_hash( &prev_hash, &doc_id, &doc_bytes, diff --git a/nodedb/src/control/server/shared/ddl/neutral/quota_ddl.rs b/nodedb/src/control/server/shared/ddl/neutral/quota_ddl.rs index 566c09f7b..cb5cd6e95 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/quota_ddl.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/quota_ddl.rs @@ -129,7 +129,7 @@ pub fn define_quota( .credentials .catalog() .put_scope_quota(&stored) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; scope_quota_post_apply::put(&stored, state); Ok(()) }, @@ -185,7 +185,7 @@ pub fn drop_quota( .credentials .catalog() .delete_scope_quota(&scope_name) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; scope_quota_post_apply::delete(&scope_name, state); Ok(()) }, diff --git a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs index 21d5ed7c4..b9ec9ee32 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs @@ -14,8 +14,9 @@ //! `SELECT RATE_RESET(gate_name, key)` //! — Deletes the counter key (admin cooldown clear). -use crate::bridge::envelope::{PhysicalPlan, Status}; +use crate::bridge::envelope::{ErrorCode, PhysicalPlan}; use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::shared::response_payload::payload_or_typed_error; use crate::control::state::SharedState; use crate::types::{DatabaseId, TraceId, VShardId}; use nodedb_physical::physical_plan::KvOp; @@ -54,7 +55,8 @@ pub async fn rate_check( let rate_key = format!("_rate:{gate_name}:{key}"); let tenant_id = identity.tenant_id; - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, RATE_COLLECTION); + let vshard = + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, RATE_COLLECTION).vshard(); let ttl_ms = window_secs * 1000; // Fixed-window semantics: TTL is set ONLY on the first call (new key). @@ -69,7 +71,7 @@ pub async fn rate_check( ), key: rate_key.as_bytes().to_vec(), }); - match crate::control::server::dispatch_utils::dispatch_to_data_plane( + let result = crate::control::server::dispatch_utils::dispatch_to_data_plane( state, tenant_id, crate::types::DatabaseId::DEFAULT, @@ -77,16 +79,8 @@ pub async fn rate_check( check, TraceId::ZERO, ) - .await - { - Ok(resp) if resp.status == Status::Ok => { - let text = - crate::data::executor::response_codec::decode_payload_to_json(&resp.payload); - // ttl_ms == -2 means key does not exist. - !text.contains("-2") - } - _ => false, - } + .await; + ttl_from("RATE_CHECK", result)?.is_some() }; let actual_ttl = if key_exists { 0 } else { ttl_ms }; @@ -94,12 +88,11 @@ pub async fn rate_check( let surrogate = state .surrogate_assigner .assign( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, RATE_COLLECTION), tenant_id, - RATE_COLLECTION, rate_key.as_bytes(), ) - .map_err(|e| super::kv_atomic::ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error_in_context("RATE_CHECK", &e))?; let plan = PhysicalPlan::Kv(KvOp::Incr { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, RATE_COLLECTION), key: rate_key.as_bytes().to_vec(), @@ -111,51 +104,40 @@ pub async fn rate_check( // dispatch bypasses the injection pass by design rather than pending // a decision that will never come. rls_write_check: nodedb_types::RlsWriteCheck::system_internal_collection(), + // The rate-gate collection declares no columns: a counter is decimal + // text, which `rate_remaining` reads back. + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }); - match crate::control::server::dispatch_utils::dispatch_to_data_plane( - state, - tenant_id, - crate::types::DatabaseId::DEFAULT, - vshard, - plan, - TraceId::ZERO, - ) - .await - { - Ok(resp) if resp.status == Status::Ok => { - let payload_text = - crate::data::executor::response_codec::decode_payload_to_json(&resp.payload); - let current: i64 = sonic_rs::from_str::(&payload_text) - .ok() - .and_then(|v| v.get("value")?.as_i64()) - .unwrap_or(1); - - if current > max_count { - // Read TTL to compute retry_after_ms. - let ttl_remaining = read_ttl_ms(state, tenant_id, vshard, &rate_key).await; - Err(ddl_err( - "54001", - format!( - "rate limit exceeded for {gate_name}:{key}, retry after {ttl_remaining}ms (current={current}, max={max_count})" - ), - )) - } else { - let result = serde_json::json!({ - "allowed": true, - "current": current, - "max_count": max_count, - "remaining": max_count - current, - }); - Ok(vec![single_text_col("rate_check", result.to_string())]) - } - } - Ok(resp) => { - let payload_text = - crate::data::executor::response_codec::decode_payload_to_json(&resp.payload); - Err(ddl_err("XX000", payload_text)) - } - Err(e) => Err(ddl_err("XX000", e.to_string())), + let payload = counter_write_payload( + "RATE_CHECK", + dispatch_counter_write(state, tenant_id, vshard, plan).await, + )?; + let payload_text = crate::data::executor::response_codec::decode_payload_to_json(&payload); + let current: i64 = sonic_rs::from_str::(&payload_text) + .ok() + .and_then(|v| v.get("value")?.as_i64()) + .ok_or(DdlError::internal(format!( + "RATE_CHECK: counter '{rate_key}' increment answered no integer value" + )))?; + + if current > max_count { + // Read TTL to compute retry_after_ms. + let ttl_remaining = read_ttl_ms("RATE_CHECK", state, tenant_id, vshard, &rate_key).await?; + Err(ddl_err( + "53300", + format!( + "rate limit exceeded for {gate_name}:{key}, retry after {ttl_remaining}ms (current={current}, max={max_count})" + ), + )) + } else { + let result = serde_json::json!({ + "allowed": true, + "current": current, + "max_count": max_count, + "remaining": max_count - current, + }); + Ok(vec![single_text_col("rate_check", result.to_string())]) } } @@ -180,7 +162,8 @@ pub async fn rate_remaining( let rate_key = format!("_rate:{gate_name}:{key}"); let tenant_id = identity.tenant_id; - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, RATE_COLLECTION); + let vshard = + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, RATE_COLLECTION).vshard(); // Read current counter value (non-destructive). let plan = PhysicalPlan::Kv(KvOp::Get { @@ -190,7 +173,7 @@ pub async fn rate_remaining( surrogate_ceiling: None, }); - let current = match crate::control::server::dispatch_utils::dispatch_to_data_plane( + let result = crate::control::server::dispatch_utils::dispatch_to_data_plane( state, tenant_id, crate::types::DatabaseId::DEFAULT, @@ -198,17 +181,11 @@ pub async fn rate_remaining( plan, TraceId::ZERO, ) - .await - { - Ok(resp) if resp.status == Status::Ok && !resp.payload.is_empty() => { - // Counter is stored as MessagePack i64. - zerompk::from_msgpack::(&resp.payload).unwrap_or(0) - } - _ => 0, // Key doesn't exist yet — no usage. - }; + .await; + let current = counter_from(&rate_key, result)?; let ttl_remaining = if current > 0 { - read_ttl_ms(state, tenant_id, vshard, &rate_key).await + read_ttl_ms("RATE_REMAINING", state, tenant_id, vshard, &rate_key).await? } else { 0 }; @@ -242,7 +219,8 @@ pub async fn rate_reset( let rate_key = format!("_rate:{gate_name}:{key}"); let tenant_id = identity.tenant_id; - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, RATE_COLLECTION); + let vshard = + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, RATE_COLLECTION).vshard(); let plan = PhysicalPlan::Kv(KvOp::Delete { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, RATE_COLLECTION), @@ -251,45 +229,136 @@ pub async fn rate_reset( rls_write_check: nodedb_types::RlsWriteCheck::system_internal_collection(), returning: None, rls_filters: Vec::new(), + provenance: None, }); - match crate::control::server::dispatch_utils::dispatch_to_data_plane( + // A refusal means the counter is still there, so the reset did not happen. + counter_write_payload( + "RATE_RESET", + dispatch_counter_write(state, tenant_id, vshard, plan).await, + )?; + let result = serde_json::json!({ + "gate": gate_name, + "key": key, + "reset": true, + }); + Ok(vec![single_text_col("rate_reset", result.to_string())]) +} + +// ── Helpers ──────────────────────────────────────────────────────────── + +/// Apply a write to the rate-gate counter collection on the durable route: +/// Raft in cluster mode, else the write funnel's `AppendHere`. A counter +/// written any other way has no WAL record, so a crash resets the gate and a +/// replica never counts the call. +async fn dispatch_counter_write( + state: &SharedState, + tenant_id: crate::types::TenantId, + vshard: VShardId, + plan: PhysicalPlan, +) -> crate::Result { + crate::control::server::dispatch_utils::dispatch_durable_autocommit_write( state, - tenant_id, - crate::types::DatabaseId::DEFAULT, - vshard, - plan, - TraceId::ZERO, + crate::control::server::dispatch_utils::AutocommitWrite { + tenant_id, + database_id: DatabaseId::DEFAULT, + vshard_id: vshard, + plan, + trace_id: TraceId::ZERO, + event_source: crate::event::EventSource::User, + txn_id: None, + }, ) .await - { - Ok(_) => { - let result = serde_json::json!({ - "gate": gate_name, - "key": key, - "reset": true, - }); - Ok(vec![single_text_col("rate_reset", result.to_string())]) - } - Err(e) => Err(ddl_err("XX000", e.to_string())), +} + +/// The payload of a counter write, or its error with `context` before the +/// message. A refusal arrives as an error status inside an `Ok` response. A +/// coded refusal keeps its SQLSTATE and code. Only a refusal with no code is +/// `XX000`. +fn counter_write_payload( + context: &str, + result: crate::Result, +) -> Result, DdlError> { + result + .and_then(payload_or_typed_error) + .map_err(|e| DdlError::from_error_in_context(context, &e)) +} + +/// The `ttl_ms` a `GetTtl` read reports for a key that does not exist. +const TTL_ABSENT: i64 = -2; + +/// The payload of a counter read, or `None` when the key is absent. A +/// `NotFound` verdict is a result, not an error. Every other refusal or +/// dispatch error keeps its SQLSTATE and code, with `context` before the +/// message. +fn read_payload( + context: &str, + result: crate::Result, +) -> Result>, DdlError> { + match result.and_then(payload_or_typed_error) { + Ok(payload) => Ok(Some(payload)), + Err(crate::Error::DataPlane(ErrorCode::NotFound)) => Ok(None), + Err(e) => Err(DdlError::from_error_in_context(context, &e)), } } -// ── Helpers ──────────────────────────────────────────────────────────── +/// The TTL a `GetTtl` read reports, or `None` when the key is absent. +fn ttl_from( + context: &str, + result: crate::Result, +) -> Result, DdlError> { + let Some(payload) = read_payload(context, result)? else { + return Ok(None); + }; + let text = crate::data::executor::response_codec::decode_payload_to_json(&payload); + let ttl_ms = sonic_rs::from_str::(&text) + .ok() + .and_then(|v| v.get("ttl_ms")?.as_i64()) + .ok_or(DdlError::internal(format!( + "{context}: TTL read answered no integer ttl_ms: {text}" + )))?; + Ok((ttl_ms != TTL_ABSENT).then_some(ttl_ms)) +} + +/// The counter a `Get` read reports. An absent key has no usage, so it +/// reads as `0`. +fn counter_from( + rate_key: &str, + result: crate::Result, +) -> Result { + match read_payload("RATE_REMAINING", result)? { + None => Ok(0), + Some(payload) if payload.is_empty() => Ok(0), + // `KV_INCR` stores the counter as a raw body: its decimal text. + Some(payload) => std::str::from_utf8(&payload) + .ok() + .and_then(|text| text.parse::().ok()) + .ok_or(ddl_err( + "22P02", + format!( + "RATE_REMAINING: counter '{rate_key}' does not hold decimal text; \ + reset the gate with RATE_RESET" + ), + )), + } +} -/// Read TTL remaining for a KV key (in milliseconds). +/// Read TTL remaining for a KV key (in milliseconds). An absent key, or a +/// key with no expiry, reads as `0`. `context` names the calling function. async fn read_ttl_ms( + context: &str, state: &SharedState, tenant_id: crate::types::TenantId, vshard: VShardId, key: &str, -) -> u64 { +) -> Result { let plan = PhysicalPlan::Kv(KvOp::GetTtl { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, RATE_COLLECTION), key: key.as_bytes().to_vec(), }); - match crate::control::server::dispatch_utils::dispatch_to_data_plane( + let result = crate::control::server::dispatch_utils::dispatch_to_data_plane( state, tenant_id, crate::types::DatabaseId::DEFAULT, @@ -297,19 +366,17 @@ async fn read_ttl_ms( plan, TraceId::ZERO, ) - .await - { - Ok(resp) if resp.status == Status::Ok => { - let payload_text = - crate::data::executor::response_codec::decode_payload_to_json(&resp.payload); - sonic_rs::from_str::(&payload_text) - .ok() - .and_then(|v| v.get("ttl_ms")?.as_i64()) - .map(|ttl| if ttl > 0 { ttl as u64 } else { 0 }) - .unwrap_or(0) - } - _ => 0, - } + .await; + remaining_ttl_ms(context, result) +} + +/// The milliseconds a `GetTtl` read leaves on a key. An absent key, or a key +/// with no expiry (`-1`), reads as `0`. +fn remaining_ttl_ms( + context: &str, + result: crate::Result, +) -> Result { + Ok(ttl_from(context, result)?.map_or(0, |ttl| u64::try_from(ttl).unwrap_or(0))) } fn unquote(s: &str) -> String { @@ -345,3 +412,201 @@ fn parse_u64(s: &str, func: &str, param: &str) -> Result { fn ddl_err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } + +#[cfg(test)] +mod tests { + use nodedb_types::error::sqlstate; + + use super::*; + use crate::bridge::envelope::{Payload, Response, Status}; + use crate::types::{Lsn, RequestId}; + + fn refusal(code: Option) -> Response { + Response { + request_id: RequestId::new(1), + status: Status::Error, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: code.map(Box::new), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + /// A refused counter write keeps its SQLSTATE and code, with the function + /// name before the message. + #[test] + fn a_coded_refusal_keeps_its_sqlstate() { + let refused = refusal(Some(ErrorCode::Unsupported { + detail: "not on this engine".into(), + })); + let err = counter_write_payload("RATE_CHECK", Ok(refused)) + .expect_err("a refused counter write fails the call"); + assert_eq!(err.sqlstate, sqlstate::FEATURE_NOT_SUPPORTED, "{err:?}"); + assert_eq!(err.code, nodedb_types::error::ErrorCode::SQL_NOT_ENABLED); + assert!(err.message.starts_with("RATE_CHECK: "), "{}", err.message); + } + + /// A dispatch error keeps its own class too. + #[test] + fn a_dispatch_error_keeps_its_sqlstate() { + let deadline = crate::Error::DeadlineExceeded { + request_id: RequestId::new(1), + }; + let err = counter_write_payload("RATE_RESET", Err(deadline)) + .expect_err("a failed dispatch fails the call"); + assert_eq!(err.sqlstate, sqlstate::QUERY_CANCELED.0, "{err:?}"); + } + + /// A refusal with no code has no class of its own. + #[test] + fn a_refusal_with_no_code_is_internal() { + let err = counter_write_payload("RATE_RESET", Ok(refusal(None))) + .expect_err("a refused counter write fails the call"); + assert_eq!(err.sqlstate, sqlstate::INTERNAL_ERROR, "{err:?}"); + } + + fn answer(payload: Vec) -> Response { + Response { + status: Status::Ok, + payload: Payload::from_vec(payload), + error_code: None, + ..refusal(None) + } + } + + /// The msgpack map `{"ttl_ms": ttl}` a `GetTtl` read answers with. `ttl` + /// fits a msgpack fixint. + fn ttl_payload(ttl: i8) -> Vec { + let mut bytes = vec![0x81, 0xa6]; + bytes.extend_from_slice(b"ttl_ms"); + bytes.push(ttl.to_ne_bytes()[0]); + bytes + } + + fn deadline() -> crate::Error { + crate::Error::DeadlineExceeded { + request_id: RequestId::new(1), + } + } + + fn not_found() -> Response { + refusal(Some(ErrorCode::NotFound)) + } + + /// The existence check propagates a dispatch error and a coded refusal + /// with their own class, instead of reading them as "no key". + #[test] + fn the_existence_check_propagates_errors() { + let err = ttl_from("RATE_CHECK", Err(deadline())).expect_err("a failed read fails"); + assert_eq!(err.sqlstate, sqlstate::QUERY_CANCELED.0, "{err:?}"); + assert!(err.message.starts_with("RATE_CHECK: "), "{}", err.message); + + let refused = refusal(Some(ErrorCode::Unsupported { + detail: "not on this engine".into(), + })); + let err = ttl_from("RATE_CHECK", Ok(refused)).expect_err("a refused read fails"); + assert_eq!(err.sqlstate, sqlstate::FEATURE_NOT_SUPPORTED, "{err:?}"); + } + + /// An absent key reads as absent, whether the read answers `-2` or a + /// `NotFound` verdict. A live key reads as present. + #[test] + fn the_existence_check_reads_absence_as_a_result() { + assert_eq!( + ttl_from("RATE_CHECK", Ok(answer(ttl_payload(-2)))).expect("read succeeds"), + None + ); + assert_eq!( + ttl_from("RATE_CHECK", Ok(not_found())).expect("read succeeds"), + None + ); + assert_eq!( + ttl_from("RATE_CHECK", Ok(answer(ttl_payload(30)))).expect("read succeeds"), + Some(30) + ); + assert_eq!( + ttl_from("RATE_CHECK", Ok(answer(ttl_payload(-1)))).expect("read succeeds"), + Some(-1) + ); + } + + /// A TTL read that answers no `ttl_ms` is an internal error, never a guess. + #[test] + fn a_ttl_read_with_no_ttl_is_internal() { + let err = ttl_from("RATE_CHECK", Ok(answer(Vec::new()))).expect_err("no ttl_ms fails"); + assert_eq!(err.sqlstate, sqlstate::INTERNAL_ERROR, "{err:?}"); + } + + /// The counter read propagates a dispatch error instead of reading it as + /// no usage. + #[test] + fn the_counter_read_propagates_errors() { + let err = counter_from("_rate:g:k", Err(deadline())).expect_err("a failed read fails"); + assert_eq!(err.sqlstate, sqlstate::QUERY_CANCELED.0, "{err:?}"); + assert!( + err.message.starts_with("RATE_REMAINING: "), + "{}", + err.message + ); + } + + /// An absent counter reads as zero usage. A stored counter reads as its + /// decimal value. + #[test] + fn the_counter_read_reads_absence_as_zero() { + assert_eq!( + counter_from("_rate:g:k", Ok(answer(Vec::new()))).expect("read succeeds"), + 0 + ); + assert_eq!( + counter_from("_rate:g:k", Ok(not_found())).expect("read succeeds"), + 0 + ); + assert_eq!( + counter_from("_rate:g:k", Ok(answer(b"7".to_vec()))).expect("read succeeds"), + 7 + ); + } + + /// A counter that holds text other than a decimal is a stored value the + /// read cannot parse: `22P02`, never an internal error. + #[test] + fn a_non_decimal_counter_is_invalid_text() { + let err = counter_from("_rate:g:k", Ok(answer(b"abc".to_vec()))) + .expect_err("a non-decimal counter fails the read"); + assert_eq!( + err.sqlstate, + sqlstate::INVALID_TEXT_REPRESENTATION, + "{err:?}" + ); + assert_eq!(err.code, nodedb_types::error::ErrorCode::DATA_EXCEPTION); + } + + /// The TTL read behind `retry after` propagates a dispatch error instead + /// of reading it as no time left. + #[test] + fn the_ttl_read_propagates_errors() { + let err = + remaining_ttl_ms("RATE_REMAINING", Err(deadline())).expect_err("a failed read fails"); + assert_eq!(err.sqlstate, sqlstate::QUERY_CANCELED.0, "{err:?}"); + } + + /// An absent key and a key with no expiry leave no time. A live key + /// leaves its TTL. + #[test] + fn the_ttl_read_reads_absence_as_zero() { + assert_eq!( + remaining_ttl_ms("RATE_REMAINING", Ok(not_found())).expect("read succeeds"), + 0 + ); + for (ttl, expected) in [(-2, 0), (-1, 0), (30, 30)] { + let remaining = remaining_ttl_ms("RATE_REMAINING", Ok(answer(ttl_payload(ttl)))) + .expect("read succeeds"); + assert_eq!(remaining, expected, "ttl_ms {ttl}"); + } + } +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/read_gate.rs b/nodedb/src/control/server/shared/ddl/neutral/read_gate.rs index ded823621..8e8b9a2e6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/read_gate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/read_gate.rs @@ -50,8 +50,6 @@ const INSUFFICIENT_PRIVILEGE: &str = "42501"; const FEATURE_NOT_SUPPORTED: &str = "0A000"; /// SQLSTATE for a collection the catalog does not hold. const UNDEFINED_TABLE: &str = "42P01"; -/// SQLSTATE for a policy set that could not be compiled. -const INTERNAL_ERROR: &str = "XX000"; fn gate_err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) @@ -164,12 +162,15 @@ impl<'a> CollectionReadGate<'a> { &self.state.rls, self.scope.auth(), ) - .map_err(|error| { - let sqlstate = match &error { - crate::Error::RejectedAuthz { .. } => INSUFFICIENT_PRIVILEGE, - _ => FEATURE_NOT_SUPPORTED, - }; - gate_err(sqlstate, error.to_string()) + .map_err(|error| match &error { + crate::Error::RejectedAuthz { .. } => { + gate_err(INSUFFICIENT_PRIVILEGE, error.to_string()) + } + // The injection pass refuses a plan shape it cannot cover. A + // hand-built read of that shape is a feature this door lacks. + crate::Error::PlanError { .. } => gate_err(FEATURE_NOT_SUPPORTED, error.to_string()), + // Any other error keeps the class the SQLSTATE table gives it. + other => DdlError::from_error(other), }) } @@ -189,7 +190,7 @@ impl<'a> CollectionReadGate<'a> { self.tenant_id().as_u64(), collection, ) - .map_err(|e| gate_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; match stored { None => Err(gate_err( UNDEFINED_TABLE, @@ -219,7 +220,7 @@ impl<'a> CollectionReadGate<'a> { collection, self.scope.auth(), ) - .map_err(|e| gate_err(INTERNAL_ERROR, format!("rls compile: {e}")))? + .map_err(|e| DdlError::from_error_in_context("rls compile", &e))? .is_some_and(|filters| filters.is_empty()); if unrestricted { return Ok(()); diff --git a/nodedb/src/control/server/shared/ddl/neutral/redaction/create.rs b/nodedb/src/control/server/shared/ddl/neutral/redaction/create.rs index cad899278..791b56250 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/redaction/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/redaction/create.rs @@ -130,17 +130,17 @@ pub fn create_redaction_policy( }; let stored = StoredRedactionPolicy::from_runtime(&policy) - .map_err(|e| DdlError::new("XX000", format!("redaction serialize: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("redaction serialize", &e))?; let entry = CatalogEntry::PutRedactionPolicy(Box::new(stored.clone())); let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { { let catalog = state.credentials.catalog(); catalog .put_redaction_policy(&stored) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } state.redaction.install_replicated_policy(policy); } diff --git a/nodedb/src/control/server/shared/ddl/neutral/redaction/drop_show.rs b/nodedb/src/control/server/shared/ddl/neutral/redaction/drop_show.rs index c5323cc87..0baa11a0b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/redaction/drop_show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/redaction/drop_show.rs @@ -54,13 +54,13 @@ pub fn drop_redaction_policy( for_role: for_role.to_string(), }; let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { { let catalog = state.credentials.catalog(); catalog .delete_redaction_policy(tenant_id, &qualified_collection, for_role) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } state .redaction diff --git a/nodedb/src/control/server/shared/ddl/neutral/replicate.rs b/nodedb/src/control/server/shared/ddl/neutral/replicate.rs index d92262523..ddbefd261 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/replicate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/replicate.rs @@ -16,7 +16,7 @@ use super::super::result::DdlError; /// Propose `entry`, then run `local` when this node owns the catalog write. /// -/// A propose failure maps to SQLSTATE `XX000`. +/// A propose failure keeps the SQLSTATE class of its typed error. pub(crate) fn propose_and_apply( state: &SharedState, entry: &CatalogEntry, @@ -35,7 +35,7 @@ pub(crate) fn propose_and_apply_outcome( local: impl FnOnce() -> Result<(), DdlError>, ) -> Result { let outcome = propose_catalog_entry(state, entry) - .map_err(|e| DdlError::new("XX000", format!("catalog propose failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e))?; if outcome.needs_local_apply() { local()?; } diff --git a/nodedb/src/control/server/shared/ddl/neutral/retention_policy/create.rs b/nodedb/src/control/server/shared/ddl/neutral/retention_policy/create.rs index 0f87dbf35..5cb97b7c1 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/retention_policy/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/retention_policy/create.rs @@ -125,7 +125,7 @@ pub async fn create_retention_policy( let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| err("XX000", "system clock error".to_string()))? + .map_err(|_| DdlError::internal("system clock error"))? .as_secs(); let def = RetentionPolicyDef { @@ -170,15 +170,18 @@ pub async fn create_retention_policy( // Roll back through the same replicated path that created it, so the // policy disappears on every node, not only on this one. if let Err(rollback) = propose_delete(state, &def) { - return Err(err( - "XX000", - format!( - "failed to auto-wire aggregates: {e}; rollback left the policy in place: {}", + return Err(DdlError::from_error_in_context( + &format!( + "rollback left the policy in place: {}; failed to auto-wire aggregates", rollback.message ), + &e, )); } - return Err(err("XX000", format!("failed to auto-wire aggregates: {e}"))); + return Err(DdlError::from_error_in_context( + "failed to auto-wire aggregates", + &e, + )); } state.audit_record( diff --git a/nodedb/src/control/server/shared/ddl/neutral/retention_policy/replicate.rs b/nodedb/src/control/server/shared/ddl/neutral/retention_policy/replicate.rs index 47b8f00e6..4b5d42aba 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/retention_policy/replicate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/retention_policy/replicate.rs @@ -14,10 +14,6 @@ use crate::engine::timeseries::retention_policy::RetentionPolicyDef; use super::super::super::result::DdlError; use super::super::replicate::propose_and_apply; -fn err(sqlstate: &str, message: String) -> DdlError { - DdlError::new(sqlstate, message) -} - /// Propose the full policy record. CREATE and ALTER both re-put the row. /// /// The leader validates before proposing, so apply never rejects. @@ -28,7 +24,7 @@ pub(super) fn propose_put(state: &SharedState, def: &RetentionPolicyDef) -> Resu .credentials .catalog() .put_retention_policy(def) - .map_err(|e| err("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; post_apply::put(def, state); Ok(()) }) @@ -50,7 +46,7 @@ pub(super) fn propose_delete( .credentials .catalog() .delete_retention_policy(def.database_id, def.tenant_id, &def.name) - .map_err(|e| err("XX000", format!("catalog delete: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; post_apply::delete(def.database_id, def.tenant_id, &def.name, state); Ok(()) }) diff --git a/nodedb/src/control/server/shared/ddl/neutral/rls.rs b/nodedb/src/control/server/shared/ddl/neutral/rls.rs index 7cf67644d..eace83d3d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/rls.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/rls.rs @@ -80,7 +80,8 @@ fn compile_rls_predicate( crate::Error::CollectionNotFound { .. } => { DdlError::new("42P01", format!("collection '{collection}' does not exist")) } - other => DdlError::new("XX000", format!("catalog read: {other}")), + // Any other error keeps the class the SQLSTATE table gives it. + other => DdlError::from_error_in_context("catalog read", &other), })?; let compiled = compile_policy_predicate(predicate_str, &columns) .map_err(|e| DdlError::new("42601", e.to_string()))?; @@ -228,17 +229,17 @@ pub fn create_rls_policy( }; let stored = StoredRlsPolicy::from_runtime(&policy, database_id, predicate_raw) - .map_err(|e| DdlError::new("XX000", format!("rls serialize: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("rls serialize", &e))?; let entry = CatalogEntry::PutRlsPolicy(Box::new(stored.clone())); let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { { let catalog = state.credentials.catalog(); catalog .put_rls_policy(&stored) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } state.rls.install_replicated_policy(policy); } @@ -289,13 +290,13 @@ pub fn drop_rls_policy( name: name.to_string(), }; let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { { let catalog = state.credentials.catalog(); catalog .delete_rls_policy(tenant_id, &qualified_collection, name) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } state .rls diff --git a/nodedb/src/control/server/shared/ddl/neutral/role.rs b/nodedb/src/control/server/shared/ddl/neutral/role.rs index 911a13261..2e7a0daaf 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/role.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/role.rs @@ -40,8 +40,13 @@ pub fn create_role( let name = parts[2]; + // The roles this statement sees: committed ones, and inside a + // transaction those it created earlier, so a parent created in the same + // transaction resolves. COMMIT checks the whole batch again. + let visible = super::role_checks::visible_roles(state); + // `IF NOT EXISTS`: re-creating an existing role is a no-op success. - if if_not_exists && state.roles.get_role(name).is_some() { + if if_not_exists && visible.contains_key(name) { return Ok(status("CREATE ROLE")); } @@ -51,22 +56,27 @@ pub fn create_role( None }; - // Build the `StoredRole` on the proposer (runs the same - // validation as `create_role` but without touching state). - let stored = state - .roles - .prepare_role(name, identity.tenant_id, parent) - .map_err(|e| DdlError::new("42710", e.to_string()))?; + // Build the `StoredRole` on the proposer: the same validation as + // `create_role`, against the visible roles, without touching state. + let stored = crate::control::security::role::prepare_role_against( + name, + identity.tenant_id, + parent, + &visible, + ) + .map_err(|e| DdlError::new("42710", e.to_string()))?; let entry = crate::control::catalog_entry::CatalogEntry::PutRole(Box::new(stored.clone())); let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { let catalog = state.credentials.catalog(); catalog .put_role(&stored) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; state.roles.install_replicated_role(&stored); + } else if outcome.is_replicated() { + super::role_checks::confirm_role(state, name, parent)?; } state.audit_record( @@ -100,7 +110,7 @@ pub fn drop_role( } let name = parts[2]; - let exists_before = state.roles.get_role(name).is_some(); + let exists_before = super::role_checks::visible_roles(state).contains_key(name); if !exists_before { // `IF EXISTS`: dropping a missing role is a no-op success. if if_exists { @@ -112,21 +122,37 @@ pub fn drop_role( )); } + // As PostgreSQL does, a role that users hold or roles inherit from is + // not dropped: dropping it would leave them naming a role that grants + // nothing. + super::role_checks::check_role_droppable(state, name)?; + let entry = crate::control::catalog_entry::CatalogEntry::DeleteRole { name: name.to_string(), }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; let dropped = if outcome.needs_local_apply() { let catalog = state.credentials.catalog(); state .roles .drop_role(name, Some(catalog)) - .map_err(|e| DdlError::new("42704", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? + } else if outcome.is_replicated() { + // The synchronous post-apply removed the role from this node's + // cache before the applied index advanced. A role still present + // means the applier skipped the entry: a user or role came to + // depend on it before the drop committed. + if state.roles.get_role(name).is_some() { + super::role_checks::check_role_droppable(state, name)?; + return Err(DdlError::new( + "40001", + format!("transient: the drop of role '{name}' was superseded, retry"), + )); + } + true } else { - // Cluster mode: the raft entry committed, trust the - // log index. The in-memory cache update runs in a - // spawned tokio task and may not be visible yet. + // Buffered in an open transaction: COMMIT applies it. true }; @@ -159,11 +185,14 @@ pub fn alter_role_typed( ) -> Result, DdlError> { require_tenant_admin(identity, "alter roles")?; - // The role must exist before we mutate it. - state - .roles - .get_role(role_name) - .ok_or_else(|| DdlError::new("42704", format!("role '{role_name}' not found")))?; + // The role must exist before we mutate it: committed, or created earlier + // in this transaction. + if !super::role_checks::visible_roles(state).contains_key(role_name) { + return Err(DdlError::new( + "42704", + format!("role '{role_name}' not found"), + )); + } match sub_op { AlterRoleOp::Grant { @@ -220,17 +249,18 @@ pub fn set_role_parent( role_name: &str, parent: Option<&str>, ) -> Result<(), DdlError> { - let old_role = state - .roles - .get_role(role_name) + // Roles created earlier in this transaction count, as they do for + // `CREATE ROLE`. + let visible = super::role_checks::visible_roles(state); + let old_role = visible + .get(role_name) + .cloned() .ok_or_else(|| DdlError::new("42704", format!("role '{role_name}' not found")))?; if let Some(parent) = parent { - let parent_is_builtin = matches!( - parent, - "superuser" | "tenant_admin" | "readwrite" | "readonly" | "monitor" - ); - if !parent_is_builtin && state.roles.get_role(parent).is_none() { + let parent_is_builtin = + crate::control::security::role_assignment::is_builtin_role_name(parent); + if !parent_is_builtin && !visible.contains_key(parent) { return Err(DdlError::new( "42704", format!("parent role '{parent}' does not exist"), @@ -238,10 +268,10 @@ pub fn set_role_parent( } // Reject self-inheritance and multi-hop cycles, and enforce the // inheritance-depth cap — the same invariant `CREATE ROLE` checks. - state - .roles - .check_inheritance_cycle(role_name, parent) - .map_err(|e| DdlError::new("42P16", e.to_string()))?; + crate::control::security::role::check_inheritance_cycle_against( + role_name, parent, &visible, + ) + .map_err(|e| DdlError::new("42P16", e.to_string()))?; } let now = std::time::SystemTime::now() @@ -257,13 +287,15 @@ pub fn set_role_parent( let entry = crate::control::catalog_entry::CatalogEntry::PutRole(Box::new(stored.clone())); let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { let catalog = state.credentials.catalog(); catalog .put_role(&stored) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; state.roles.install_replicated_role(&stored); + } else if outcome.is_replicated() { + super::role_checks::confirm_role(state, role_name, parent)?; } Ok(()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/role_checks.rs b/nodedb/src/control/server/shared/ddl/neutral/role_checks.rs new file mode 100644 index 000000000..85d64cb77 --- /dev/null +++ b/nodedb/src/control/server/shared/ddl/neutral/role_checks.rs @@ -0,0 +1,248 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Role rules at the DDL entry points: a user may hold only a built-in role +//! or a custom role defined in its tenant, and a custom role may be dropped +//! only while no user holds it and no role inherits from it. +//! +//! Each statement checks before it proposes. The metadata applier runs the +//! same rules at the entry's log position and skips an entry that breaks +//! them, on every node alike. A statement whose entry was skipped, because +//! a racing change committed first, learns it here after the apply and +//! reports the refusal instead of a success. +//! +//! Inside a transaction a statement sees the committed users and roles +//! overlaid with the role and user entries its transaction buffered, in +//! order: a role created earlier in the transaction can be assigned, and a +//! role a user created earlier in it holds cannot be dropped. COMMIT checks +//! the whole batch again before it applies anything. + +use std::collections::HashMap; + +use crate::control::catalog_entry::CatalogEntry; +use crate::control::security::identity::Role; +use crate::control::security::role_assignment::{self, RoleRefusal}; +use crate::control::state::SharedState; +use crate::types::TenantId; + +use super::super::result::DdlError; + +/// The DDL error for a role refusal, with PostgreSQL's SQLSTATE. +pub(super) fn refusal(refusal: RoleRefusal) -> DdlError { + DdlError::new(refusal.sqlstate(), refusal.to_string()) +} + +/// Users and roles as the running statement sees them. +struct RoleView { + /// Active users: name to held role names. + users: HashMap>, + /// Custom roles: name to (tenant, parent, empty when none). + roles: HashMap, +} + +impl RoleView { + /// The committed users and roles, overlaid with the entries this + /// connection's open transaction buffered, in statement order. + fn current(state: &SharedState) -> Self { + let mut view = Self { + users: state + .credentials + .list_user_details() + .into_iter() + .map(|user| { + let held = user.roles.iter().map(ToString::to_string).collect(); + (user.username, held) + }) + .collect(), + roles: state + .roles + .list_roles() + .into_iter() + .map(|role| { + let parent = role.parent.unwrap_or_default(); + (role.name, (role.tenant_id.as_u64(), parent)) + }) + .collect(), + }; + crate::control::server::shared::session::ddl_buffer::with_buffered(|buffered| { + for item in buffered { + view.overlay(&item.entry); + } + }); + view + } + + fn overlay(&mut self, entry: &CatalogEntry) { + match entry { + CatalogEntry::PutUser(user) if user.is_active => { + self.users.insert(user.username.clone(), user.roles.clone()); + } + CatalogEntry::PutUser(user) => { + self.users.remove(&user.username); + } + CatalogEntry::PutTenantWithAdmin { admin, .. } => { + self.users + .insert(admin.username.clone(), admin.roles.clone()); + } + CatalogEntry::DropUser { username } => { + self.users.remove(username); + } + CatalogEntry::PutRole(role) => { + self.roles + .insert(role.name.clone(), (role.tenant_id, role.parent.clone())); + } + CatalogEntry::DeleteRole { name } => { + self.roles.remove(name); + } + // Every other entry leaves users and roles as they are. + _ => {} + } + } +} + +/// Active user `username` as the running statement sees it: the committed +/// record, overlaid with the user entries its transaction buffered, in +/// order. A user created earlier in the transaction is visible; one dropped +/// earlier in it is not. +pub(super) fn visible_user( + state: &SharedState, + username: &str, +) -> Option { + let committed = state.credentials.stored_user(username); + crate::control::server::shared::session::ddl_buffer::with_buffered(|buffered| { + let mut user = committed.clone(); + for item in buffered { + match &item.entry { + CatalogEntry::PutUser(stored) if stored.username == username => { + user = stored.is_active.then(|| (**stored).clone()); + } + CatalogEntry::PutTenantWithAdmin { admin, .. } if admin.username == username => { + user = Some((**admin).clone()); + } + CatalogEntry::DropUser { username: dropped } if dropped == username => { + user = None; + } + // Every other entry leaves this user as it is. + _ => {} + } + } + user + }) + .unwrap_or(committed) +} + +/// [`visible_user`], or the 42704 refusal a statement naming a missing user +/// reports. +pub(super) fn visible_user_or_missing( + state: &SharedState, + username: &str, +) -> Result { + visible_user(state, username) + .ok_or_else(|| DdlError::new("42704", format!("user '{username}' not found"))) +} + +/// Every custom role the running statement sees, keyed by name: the +/// committed ones overlaid with those its transaction buffered. +pub(super) fn visible_roles( + state: &SharedState, +) -> HashMap { + RoleView::current(state) + .roles + .into_iter() + .map(|(name, (tenant_id, parent))| { + let role = crate::control::security::role::CustomRole { + name: name.clone(), + tenant_id: TenantId::new(tenant_id), + parent: (!parent.is_empty()).then_some(parent), + created_at: 0, + }; + (name, role) + }) + .collect() +} + +/// Refuse any role in `roles` a user of `tenant_id` cannot hold. +pub(super) fn check_user_roles( + state: &SharedState, + roles: &[Role], + tenant_id: TenantId, +) -> Result<(), DdlError> { + let view = RoleView::current(state); + role_assignment::check_assignable(roles, tenant_id.as_u64(), |name| { + view.roles.get(name).map(|(tenant, _)| *tenant) + }) + .map_err(refusal) +} + +/// After a replicated user entry applied, confirm the user of `tenant_id` +/// holds exactly `roles`. The applier skips a user entry naming a role that +/// was dropped before the entry committed; that surfaces here as the role's +/// refusal. Any other mismatch is a concurrent change or a truncated entry, +/// which the client retries. +pub(super) fn confirm_user_roles( + state: &SharedState, + username: &str, + roles: &[Role], + tenant_id: TenantId, +) -> Result<(), DdlError> { + let held = state.credentials.get_user(username).map(|user| user.roles); + let held_exactly = held.as_ref().is_some_and(|held| { + roles.iter().all(|role| held.contains(role)) && held.iter().all(|role| roles.contains(role)) + }); + if held_exactly { + return Ok(()); + } + check_user_roles(state, roles, tenant_id)?; + Err(DdlError::new( + "40001", + format!( + "transient: the entry for user '{username}' was superseded or truncated by a \ + leader change, retry" + ), + )) +} + +/// Refuse to drop the custom role `name` while a user holds it or a role +/// inherits from it. +pub(super) fn check_role_droppable(state: &SharedState, name: &str) -> Result<(), DdlError> { + let view = RoleView::current(state); + role_assignment::check_droppable( + name, + view.users + .iter() + .map(|(user, held)| (user.as_str(), held.as_slice())), + view.roles + .iter() + .map(|(role, (_, parent))| (role.as_str(), parent.as_str())), + ) + .map_err(refusal) +} + +/// After a replicated role entry applied, confirm role `name` exists with +/// inheritance parent `parent`. The applier skips a role entry whose parent +/// was dropped before the entry committed; that surfaces here as the +/// parent's refusal. +pub(super) fn confirm_role( + state: &SharedState, + name: &str, + parent: Option<&str>, +) -> Result<(), DdlError> { + let installed = state + .roles + .get_role(name) + .is_some_and(|role| role.parent.as_deref() == parent); + if installed { + return Ok(()); + } + if let Some(parent) = parent + && !role_assignment::is_builtin_role_name(parent) + && state.roles.get_role(parent).is_none() + { + return Err(refusal(RoleRefusal::Undefined { + name: parent.to_string(), + })); + } + Err(DdlError::new( + "40001", + format!("transient: the entry for role '{name}' was superseded or truncated, retry"), + )) +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/router/dispatch.rs index bbc6b2254..059873586 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/dispatch.rs @@ -69,10 +69,10 @@ pub async fn try_dispatch( return Some(r); } - // Parse errors surface as a typed `DdlError` here: `UnsupportedConstraint` - // maps to `0A000` (feature_not_supported), every other parse error to - // `42601` (syntax error), with the parser's own `Display` text as the - // message. This is the sole parse-error gate for the DDL router; the + // Parse errors surface as a typed `DdlError` here, with the SQLSTATE the + // planner path gives the same `SqlError` (`UnsupportedConstraint` is + // `0A000`, a parse error `42601`) and the parser's own `Display` text as + // the message. This is the sole parse-error gate for the DDL router; the // GRAPH / MATCH / SHOW GRAPH STATS prefixed inputs that previously carried // their own parse-error reproduction are subsumed by this arm. // @@ -86,14 +86,13 @@ pub async fn try_dispatch( let stmt = match nodedb_sql::ddl_ast::parse(sql) { Some(Ok(stmt)) => stmt, Some(Err(e)) => { - // UnsupportedConstraint / ConflictingEngineClause → 0A000 (feature_not_supported). - // All other parse errors → 42601 (syntax error). - let sqlstate = match &e { - nodedb_sql::SqlError::UnsupportedConstraint { .. } - | nodedb_sql::SqlError::ConflictingEngineClause { .. } => "0A000", - _ => "42601", - }; - return Some(Err(DdlError::new(sqlstate, e.to_string()))); + // The SQLSTATE the planner path renders for the same error, so a + // parse refusal answers one class wherever it is raised. + let message = e.to_string(); + let (_, sqlstate, _) = crate::control::server::pgwire::types::error_to_sqlstate( + &crate::control::planner::plan_error_map::map_plan_error(e, identity.tenant_id), + ); + return Some(Err(DdlError::new(sqlstate, message))); } None => { // Bulk import: `COPY FROM STDIN [WITH (...)]`. The diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs b/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs index cbcb01808..ec80f408f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs @@ -96,23 +96,31 @@ pub(super) async fn try_string( // doc-object UPSERT body, string literal, or comment carrying the token // can never reach these arms. if upper.starts_with("SELECT RANK(") || upper.starts_with("SELECT RANK (") { - return Some(kv_sorted_index::select_rank(state, identity, database_id, sql).await); + return Some( + kv_sorted_index::select_rank(state, identity, database_id, sql, txn_ctx).await, + ); } if upper.starts_with("SELECT TOPK(") || upper.starts_with("SELECT TOPK (") || upper.starts_with("SELECT * FROM TOPK(") || upper.starts_with("SELECT * FROM TOPK (") { - return Some(kv_sorted_index::select_topk(state, identity, database_id, sql).await); + return Some( + kv_sorted_index::select_topk(state, identity, database_id, sql, txn_ctx).await, + ); } if upper.starts_with("SELECT SORTED_COUNT(") || upper.starts_with("SELECT SORTED_COUNT (") { - return Some(kv_sorted_index::select_sorted_count(state, identity, database_id, sql).await); + return Some( + kv_sorted_index::select_sorted_count(state, identity, database_id, sql, txn_ctx).await, + ); } // RANGE as a sorted index function (check it's not a standard SQL RANGE). if (upper.starts_with("SELECT * FROM RANGE(") || upper.starts_with("SELECT * FROM RANGE (")) && !upper.contains(" BETWEEN ") { - return Some(kv_sorted_index::select_range(state, identity, database_id, sql).await); + return Some( + kv_sorted_index::select_range(state, identity, database_id, sql, txn_ctx).await, + ); } // KV_INCR / KV_DECR / KV_INCR_FLOAT / KV_CAS / KV_GETSET — atomic KV operations. @@ -322,15 +330,17 @@ pub(super) async fn try_string( return Some(estimate_count::estimate_count(state, identity, database_id, sql).await); } - // `DEFINE FIELD …` / `DEFINE EVENT …` — string-recognized (no typed DDL - // variant); the pgwire schema string router dispatched both from the raw - // SQL. Replicate that exactly here, before the parse gate. + // `DEFINE FIELD …` / `DEFINE EVENT …` / `REMOVE EVENT …` — + // string-recognized (no typed DDL variant), before the parse gate. if upper.starts_with("DEFINE FIELD ") { return Some(field_def::define_field(state, identity, database_id, sql)); } if upper.starts_with("DEFINE EVENT ") { return Some(field_def::define_event(state, identity, database_id, sql)); } + if upper.starts_with("REMOVE EVENT ") { + return Some(field_def::remove_event(state, identity, database_id, sql)); + } // `EXPLAIN TIERS ON [RANGE …]` — string-recognized (no typed // DDL variant); the pgwire admin string router dispatched it from the raw @@ -354,10 +364,7 @@ fn crdt_apply_forbidden_in_transaction(txn_ctx: &DmlTxnCtx<'_>) -> bool { } fn crdt_transaction_error() -> DdlError { - DdlError::new( - "25001", - crate::Error::CrdtApplyForbiddenInTransaction.to_string(), - ) + DdlError::from_error(&crate::Error::CrdtApplyForbiddenInTransaction) } #[cfg(test)] @@ -385,6 +392,10 @@ mod tests { let error = crdt_transaction_error(); assert_eq!(error.sqlstate, "25001"); + assert_eq!( + error.code, + nodedb_types::error::ErrorCode::ACTIVE_SQL_TRANSACTION + ); assert_eq!( error.message, crate::Error::CrdtApplyForbiddenInTransaction.to_string() diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/string_versioning.rs b/nodedb/src/control/server/shared/ddl/neutral/router/string_versioning.rs index c507646f3..bd566e950 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/string_versioning.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/string_versioning.rs @@ -63,9 +63,8 @@ pub(super) async fn try_string( } if upper.starts_with("RESTORE ") && upper.contains("SET VERSION") { if restore_forbidden_in_transaction(txn_ctx) { - return Some(Err(DdlError::new( - "25001", - crate::Error::CrdtApplyForbiddenInTransaction.to_string(), + return Some(Err(DdlError::from_error( + &crate::Error::CrdtApplyForbiddenInTransaction, ))); } return Some( diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/typed_collection.rs b/nodedb/src/control/server/shared/ddl/neutral/router/typed_collection.rs index 9fab7bf78..6cc918345 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/typed_collection.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/typed_collection.rs @@ -65,7 +65,7 @@ pub(super) async fn try_typed( collection::dispatch_register_by_name(state, identity, name, database_id) .await .map(|()| resp) - .map_err(|e| DdlError::new("XX000", e.to_string())) + .map_err(|e| DdlError::from_error(&e)) } Err(e) => Err(e), }; @@ -106,7 +106,7 @@ pub(super) async fn try_typed( collection::dispatch_register_by_name(state, identity, name, database_id) .await .map(|()| resp) - .map_err(|e| DdlError::new("XX000", e.to_string())) + .map_err(|e| DdlError::from_error(&e)) } Err(e) => Err(e), }; diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/typed_database.rs b/nodedb/src/control/server/shared/ddl/neutral/router/typed_database.rs index b1d2fe469..72eb6ec46 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/typed_database.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/typed_database.rs @@ -179,10 +179,11 @@ pub(super) async fn try_typed( // USE DATABASE is intercepted in `execute_single_sql` before the DDL // router runs; reaching this arm means the intercept did not fire. - NodedbStatement::Database(DatabaseStmt::UseDatabase { name }) => Some(Err(DdlError::new( - "XX000", - format!("USE DATABASE {name}: reached router after expected intercept"), - ))), + NodedbStatement::Database(DatabaseStmt::UseDatabase { name }) => { + Some(Err(DdlError::internal(format!( + "USE DATABASE {name}: reached router after expected intercept" + )))) + } _ => None, } diff --git a/nodedb/src/control/server/shared/ddl/neutral/schedule/create.rs b/nodedb/src/control/server/shared/ddl/neutral/schedule/create.rs index 33be9087d..59be7aa7b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/schedule/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/schedule/create.rs @@ -78,7 +78,7 @@ pub fn create_schedule( let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| DdlError::new("XX000", "system clock error"))? + .map_err(|_| DdlError::internal("system clock error"))? .as_secs(); let target_collection = extract_target_collection(body_sql); diff --git a/nodedb/src/control/server/shared/ddl/neutral/schedule/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/schedule/drop.rs index d0c66b8b1..6848c0c3c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/schedule/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/schedule/drop.rs @@ -81,14 +81,14 @@ pub fn drop_schedule( name: name.clone(), }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { let _ = catalog .delete_schedule_in_database(database_id, tenant_id, &name) - .map_err(|e| DdlError::new("XX000", format!("catalog delete: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; catalog .delete_owner("schedule", database_id.as_u64(), tenant_id, &name) - .map_err(|e| DdlError::new("XX000", format!("catalog owner delete: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog owner delete", &e))?; state .schedule_registry .unregister(database_id, tenant_id, &name); diff --git a/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/define.rs b/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/define.rs index 7661e91fe..1897ea1a0 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/define.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/define.rs @@ -99,7 +99,7 @@ pub fn drop_scope( let found = state .scope_defs .drop_scope(name) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if !found { return Err(err("42704", format!("scope '{name}' not found"))); } diff --git a/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/grant.rs b/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/grant.rs index a746f6795..d4fb7d62a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/grant.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/grant.rs @@ -23,7 +23,7 @@ use super::support::{err, status}; /// the expiry sweep and the DDL handlers cannot drift apart; this wrapper only /// translates the error into what pgwire reports. fn propose_scope_grant(state: &SharedState, stored: &StoredScopeGrant) -> Result<(), DdlError> { - propose_grant(state, stored).map_err(|e| err("XX000", e.to_string())) + propose_grant(state, stored).map_err(|e| DdlError::from_error(&e)) } /// Replicate a scope-grant removal. Same dual path as [`propose_scope_grant`]. @@ -34,7 +34,7 @@ fn propose_scope_revoke( grantee_id: &str, ) -> Result<(), DdlError> { propose_revoke(state, scope_name, grantee_type, grantee_id) - .map_err(|e| err("XX000", e.to_string())) + .map_err(|e| DdlError::from_error(&e)) } /// GRANT SCOPE '' TO '' @@ -92,7 +92,7 @@ pub fn grant_scope( on_expire_action: &on_expire_action, conditions, }) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; propose_scope_grant(state, &stored)?; state.audit_record( @@ -168,7 +168,7 @@ pub fn renew_scope( let outcome = state .scope_grants .prepare_renew(scope_name, &grantee_type, grantee_id, extend_secs) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; match outcome { RenewOutcome::NotFound => return Err(err("42704", "scope grant not found")), // Nothing to move: a permanent grant has no deadline to extend. diff --git a/nodedb/src/control/server/shared/ddl/neutral/sequence.rs b/nodedb/src/control/server/shared/ddl/neutral/sequence.rs index 6622fec5d..fdde1d9a9 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/sequence.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/sequence.rs @@ -12,6 +12,8 @@ use serde_json::{Map, Value as JsonValue}; use crate::control::security::catalog::sequence_types::StoredSequence; use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::sequence::SequenceError; +use crate::control::sequence::error_map::sequence_error_to_error; use crate::control::server::response_shape::types::ShapedRows; use crate::control::server::shared::ddl::sql_parse::parse_ident_token; use crate::control::state::SharedState; @@ -132,10 +134,11 @@ pub fn create_sequence( let entry = crate::control::catalog_entry::CatalogEntry::PutSequence(Box::new(def.clone())); let outcome = propose_and_apply(state, &entry)?; if outcome.needs_local_apply() { + let name = def.name.clone(); state .sequence_registry .create(def) - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .map_err(|e| create_refusal(&name, e))?; } state.schema_version.bump(); @@ -143,6 +146,24 @@ pub fn create_sequence( Ok(status("CREATE SEQUENCE")) } +/// The DDL error for a registry refusal of a new sequence. A name taken since +/// the existence check is a duplicate object. Every other refusal keeps the +/// SQLSTATE the sequence error map gives it. +fn create_refusal(name: &str, error: SequenceError) -> DdlError { + match error { + SequenceError::AlreadyExists { .. } => DdlError::new("42P07", error.to_string()), + other @ (SequenceError::Exhausted { .. } + | SequenceError::NotYetCalled { .. } + | SequenceError::OutOfRange { .. } + | SequenceError::NotFound { .. } + | SequenceError::InvalidDefinition { .. } + | SequenceError::FormatParse { .. } + | SequenceError::InvalidResetScope { .. }) => { + DdlError::from_error(&sequence_error_to_error(name, other)) + } + } +} + /// Handle `ALTER SEQUENCE RESTART [WITH ] | FORMAT '